Merge branch 'main' into feat/relay-slack-parity

One conflict, gateway/relay/adapter.py send_for_platform: main added the
turn-final draft-seal interception (_sfp_metadata with the _interim_send
marker stripped, seal-or-fall-through); this branch added format-hint
stamping on the same frame. COMPOSED: the plain-send frame now stamps
_with_format_hints_for_platform over _sfp_metadata (the stripped copy),
so both the seal fall-through contract and the cron-lane block hints
hold. Note: the seal frame itself (op:draft final) does not stamp hints
— cron sends are never open drafts, so the flagship path is unaffected;
noted as a connector-PR follow-up for streamed interactive finals.
This commit is contained in:
Ben Barclay
2026-08-20 20:56:03 +10:00
911 changed files with 70514 additions and 17469 deletions
@@ -24,6 +24,12 @@ outputs:
docker_meta:
description: Docker setup and meta files have changed.
value: ${{ steps.classify.outputs.docker_meta }}
docker:
description: Files included in the docker image have changed.
value: ${{ steps.classify.outputs.docker }}
nix:
description: Run `nix flake check` (flake inputs, or any product Python change).
value: ${{ steps.classify.outputs.nix }}
site:
description: Build the Docusaurus docs site.
value: ${{ steps.classify.outputs.site }}
+6 -6
View File
@@ -52,14 +52,14 @@ jobs:
- name: Decide whether to build
id: gate
env:
# python_prod (not python): the image copies installed code, never
# tests/, so tests-only PRs skip the build.
PYTHON_PROD: ${{ steps.classify.outputs.python_prod }}
FRONTEND: ${{ steps.classify.outputs.frontend }}
DOCKER_META: ${{ steps.classify.outputs.docker_meta }}
# The docker lane derives from python_prod (not python: the image
# copies installed code, never tests/, so tests-only PRs skip the
# build), frontend and docker_meta. classify_changes.py owns the
# formula so this gate and the nix lane cannot drift apart.
DOCKER: ${{ steps.classify.outputs.docker }}
run: |
set -euo pipefail
if [ "$PYTHON_PROD" = "true" ] || [ "$FRONTEND" = "true" ] || [ "$DOCKER_META" = "true" ]; then
if [ "$DOCKER" = "true" ]; then
echo "build=true" >> "$GITHUB_OUTPUT"
else
echo "build=false" >> "$GITHUB_OUTPUT"
+1 -1
View File
@@ -44,7 +44,7 @@ jobs:
RUN_INFO=$(gh run list \
--repo "$REPO" \
--commit "$HEAD_SHA" \
--workflow ci.yml \
--workflow ci.yaml \
--limit 1 \
--json databaseId,status \
--jq '.[0] | "\(.databaseId) \(.status)"' 2>/dev/null || true)
+117
View File
@@ -0,0 +1,117 @@
name: Nix flake check
# Builds every output of the flake: the package, the devShell, and the 21
# checks under nix/checks.nix — module evaluation, option parity, .env
# assembly, service argv, and the rest.
#
# This workflow owns its triggers and ci.yml does not call it, for the reason
# docker.yml gives: a reusable-workflow call holds the caller run in progress
# for the full build, and GitHub refuses `gh run rerun` on a run that is still
# in progress. One slow advisory job in the CI lane blocks every rerun of the
# fast required jobs beside it. A separate run reruns and cancels on its own.
on:
pull_request:
push:
branches: [main]
permissions:
contents: read
# PR runs collapse to the newest commit. A push to main is never cancelled:
# each one saves the store cache that later PRs restore from, so cancelling a
# merge would leave the next PR to build from nothing.
concurrency:
group: nix-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
# A `paths:` filter cannot gate this workflow correctly. The flake packages
# the product, and nine of the checks then run the built binary, so a change
# to hermes_cli/ alone can fail `nix flake check` without touching one file
# under nix/. The `nix` lane therefore follows python_prod as well as the
# flake inputs. On push the classifier fails open and every lane is true.
detect:
name: Detect affected areas
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
nix: ${{ steps.classify.outputs.nix }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Detect affected areas
id: classify
uses: ./.github/actions/detect-changes
with:
github-token: ${{ github.token }}
flake-check:
name: nix flake check
needs: [detect]
if: needs.detect.outputs.nix == 'true'
# The build compiles the package and its whole dependency closure, so this
# is minutes and not seconds when the cache misses.
runs-on: ubuntu-latest
timeout-minutes: 60
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install Nix
uses: cachix/install-nix-action@630ae543ea3a38a9a4166f03376c02c50f408342 # v31.11.0
with:
extra_nix_config: |
experimental-features = nix-command flakes
# A store path that does not substitute is a cache miss and not a
# build failure. Build it here instead.
fallback = true
# Each source archive is fetched one time in a run, and not one
# time for each evaluation.
tarball-ttl = 3600
# Restores /nix/store from the GitHub Actions cache. The store holds the
# whole dependency closure, so a hit turns a build of several minutes
# into a short evaluation.
#
# The Magic Nix Cache is not an option here. Its free tier ended in
# February 2025 with the GitHub cache API that it was built on. This
# action uses the current API and needs no account and no secret.
- name: Restore and save the Nix store
uses: nix-community/cache-nix-action@7df957e333c1e5da7721f60227dbba6d06080569 # v7
with:
# The closure changes when the flake inputs change or when the
# dependencies of the project change. The key hashes both, so an
# edit to the source alone keeps the hit.
primary-key: nix-${{ runner.os }}-${{ hashFiles('flake.lock', 'nix/**', 'pyproject.toml', 'uv.lock') }}
# On a miss, restore the newest store for this runner. Most of the
# closure — Python, node, each transitive library — survives a bump
# of the lockfile, so an old store still removes most of the work.
restore-prefixes-first-match: nix-${{ runner.os }}-
# Save from main only. A cache that a PR writes is visible to that
# PR alone and never to another branch, so a save there spends the
# 10 GB quota of the repository and helps no later run. A PR still
# restores: it reads the cache that the merge to main wrote. This is
# the same rule that docker.yml applies to `cache-to`.
save: ${{ github.event_name != 'pull_request' }}
# Collect garbage before the save, so the store stays inside the
# 10 GB quota of the repository. Without a limit the store grows at
# each merge until GitHub removes the entry, and the next PR then
# gets nothing. This number is the size of the store and not the
# size of the compressed archive.
gc-max-store-size-linux: 5G
# Delete the caches that this key replaces. GitHub removes caches by
# least recent use across the whole repository, so a Nix store that
# is never purged pushes out the caches of the other workflows.
purge: true
purge-prefixes: nix-${{ runner.os }}-
purge-created: 0
purge-primary-key: never
- name: nix flake check
# --print-build-logs: a check that fails then prints the assertion
# that failed, and not only the derivation that failed to build.
run: nix flake check --print-build-logs
+41
View File
@@ -1342,6 +1342,47 @@ while the agent is blocked (e.g. approval prompts) MUST bypass BOTH
guards and be dispatched inline, not via `_process_message_background()`
(which races session lifecycle).
### Streaming delivery contract (stream-is-the-message adapters) — duplicate-final class
Adapters with `draft_stream_is_message = True` (relay Slack native streaming)
keep ONE cumulative native stream per turn; the stream IS the final message.
Four invariants, each learned from a live duplicate-final incident (NS-658
canary ledger, hermes#85796 / gateway-gateway#210). Violating any of them
re-creates a duplicate or a frozen stream:
1. **Draft frames must be prefix-stable.** The connector computes append-only
deltas: frame N must be a string prefix of frame N+1. NEVER mutate draft
frames per-tick — no fence-closing (`ensure_closed_code_fences`), no cursor
suffix, no segment-state resets at tool boundaries, no mrkdwn conversion.
Any non-prefix frame triggers a whole-snapshot re-append on the platform
("stacked copies"). The finalize path may still transform the real final.
2. **The consumer declares the final; the adapter never guesses.**
`finish(final_text)` carries the completed `final_response` (verifier
footer, completion explainer included) as the authoritative finalize
payload. New post-stream response augmentation MUST ride this payload —
if it mutates `final_response` after the stream sealed, it re-opens the
#11 bug (`delivered_final_matches` mismatch → corrective duplicate send).
3. **Interim sends must carry `_interim_send` metadata.** Any consumer-side
`adapter.send()` that is NOT the turn-final (commentary, segment-tail
flushes) must set `metadata["_interim_send"] = True`, or the relay
adapter's seal-interception will seal the live stream with interim text.
Seal-interception exists at BOTH egress doors (`send()` AND
`send_for_platform()`); a new egress door needs the same two checks.
4. **Reconcile by edit, never by plain send.** Any lane that delivers a final
beside an already-sealed stream (queued follow-ups, media-accompanied
finals, future lanes) must first try `edit_message` on the consumer's
`message_id`; plain `send()` is the fallback only when no editable message
exists. A sealed native stream is a regular message — `chat.update` on it
works (live-verified).
Contract tests: `tests/gateway/test_stream_final_contract.py` (all four
invariants, mutation-checked). Slack streaming API ground truth (live-probed,
also encoded in connector comments/tests): `chat.*Stream` speaks STANDARD
markdown, not mrkdwn; `stopStream.markdown_text` APPENDS (never replaces);
`startStream`/`stopStream` are rate-limit Tier 2 (~20/min).
Guard style note: check `draft_stream_is_message` with `is True` — MagicMock
adapters in older tests auto-create truthy attributes.
### Squash merges from stale branches silently revert recent fixes
Before squash-merging a PR, ensure the branch is up to date with `main`
(`git fetch origin main && git reset --hard origin/main` in the worktree,
+1 -1
View File
@@ -582,7 +582,7 @@ test(tools): añadir tests unitarios para file_operations
## Reportar Issues
- Usa [GitHub Issues](https://github.com/NousResearch/hermes-agent/issues)
- Incluye: SO, versión de Python, versión de Hermes (`hermes version`), traza de error completa
- Incluye: SO, versión de Python, versión de Hermes (`hermes --version`), traza de error completa
- Incluye pasos para reproducir
- Verifica los issues existentes antes de crear duplicados
- Para vulnerabilidades de seguridad, por favor reporta de forma privada
+1 -1
View File
@@ -973,7 +973,7 @@ test(tools): add unit tests for file_operations
## Reporting Issues
- Use [GitHub Issues](https://github.com/NousResearch/hermes-agent/issues)
- Include: OS, Python version, Hermes version (`hermes version`), full error traceback
- Include: OS, Python version, Hermes version (`hermes --version`), full error traceback
- Include steps to reproduce
- Check existing issues before creating duplicates
- For security vulnerabilities, please report privately
+1 -1
View File
@@ -16,7 +16,7 @@ Un informe útil incluye:
- Una descripción concisa y evaluación de severidad.
- El componente afectado, identificado por ruta de archivo y rango de líneas
(ej. `path/to/file.py:120-145`).
- Detalles del entorno (`hermes version`, SHA del commit, SO, versión de Python).
- Detalles del entorno (`hermes --version`, SHA del commit, SO, versión de Python).
- Una reproducción contra `main` o el último release.
- Una declaración de qué límite de confianza del §2 se cruza.
+1 -1
View File
@@ -16,7 +16,7 @@ A useful report includes:
- A concise description and severity assessment.
- The affected component, identified by file path and line range
(e.g. `path/to/file.py:120-145`).
- Environment details (`hermes version`, commit SHA, OS, Python
- Environment details (`hermes --version`, commit SHA, OS, Python
version).
- A reproduction against `main` or the latest release.
- A statement of which trust boundary in §2 is crossed.
+59
View File
@@ -490,6 +490,25 @@ def _merge_custom_provider_extra_body(agent, custom_providers: List[Dict[str, An
agent.request_overrides = overrides
def _normalize_run_budget_seconds(value) -> Optional[float]:
"""Normalize a wall-clock run budget value to a positive float or None.
None / absent / non-numeric / non-positive all resolve to ``None``
(feature off) so a malformed config value can never activate the
deadline machinery, only leave it dormant. ``bool`` is rejected because
YAML ``true`` would otherwise become a 1-second budget.
"""
if value is None or isinstance(value, bool):
return None
try:
seconds = float(value)
except (TypeError, ValueError):
return None
if seconds != seconds or seconds <= 0: # NaN or non-positive
return None
return seconds
def init_agent(
agent,
base_url: str = None,
@@ -527,8 +546,10 @@ def init_agent(
clarify_callback: callable = None,
read_terminal_callback: callable = None,
read_preview_callback: callable = None,
drive_preview_callback: callable = None,
read_window_below_callback: callable = None,
setup_mcp_callback: callable = None,
tour_callback: callable = None,
step_callback: callable = None,
stream_delta_callback: callable = None,
interim_assistant_callback: callable = None,
@@ -559,6 +580,7 @@ def init_agent(
session_db=None,
parent_session_id: str = None,
iteration_budget: "IterationBudget" = None,
run_budget_seconds: Optional[float] = None,
fallback_model: Dict[str, Any] = None,
credential_pool=None,
checkpoints_enabled: bool = False,
@@ -822,8 +844,10 @@ def init_agent(
agent.clarify_callback = clarify_callback
agent.read_terminal_callback = read_terminal_callback
agent.read_preview_callback = read_preview_callback
agent.drive_preview_callback = drive_preview_callback
agent.read_window_below_callback = read_window_below_callback
agent.setup_mcp_callback = setup_mcp_callback
agent.tour_callback = tour_callback
agent.step_callback = step_callback
agent.stream_delta_callback = stream_delta_callback
agent.interim_assistant_callback = interim_assistant_callback
@@ -911,6 +935,10 @@ def init_agent(
# Model response configuration
agent.max_tokens = max_tokens # None = use model default
agent.reasoning_config = reasoning_config # None = use default (medium for OpenRouter)
# Per-provider reasoning_content echo opt-in (see _reasoning_echo_opt_in).
# Read once at init; switch_model / try_activate_fallback / restore
# keep it in sync with the active provider.
agent._reasoning_echo_flag = agent._read_reasoning_echo_from_config()
agent.service_tier = service_tier
agent.request_overrides = dict(request_overrides or {})
agent.prefill_messages = prefill_messages or [] # Prefilled conversation turns
@@ -966,6 +994,17 @@ def init_agent(
agent._budget_exhausted_injected = False
agent._budget_grace_call = False
# Optional wall-clock run budget (seconds per run_conversation turn).
# Explicit constructor arg wins; else resolved from config.yaml
# (agent.run_budget_seconds) further below. None = feature fully off:
# no clock reads, no injection, no stale-timeout capping.
agent.run_budget_seconds = _normalize_run_budget_seconds(run_budget_seconds)
# Wall-clock start of the CURRENT run_conversation turn. Set by
# turn_context.prepare_turn when a run budget is active; None otherwise.
agent._run_budget_started_at = None
# One-shot latch for the 80% wrap-up notice (reset each turn).
agent._run_budget_wrapup_injected = False
# Activity tracking — updated on each API call, tool execution, and
# stream chunk. Used by the gateway timeout handler to report what the
# agent was doing when it was killed, and by the "still working"
@@ -1890,6 +1929,20 @@ def init_agent(
_agent_section = {}
agent._tool_use_enforcement = _agent_section.get("tool_use_enforcement", "auto")
# Execution-discipline guidance gate: "auto" (default — matches
# EXECUTION_GUIDANCE_MODELS), true (always), false (never), or list of
# model-name substrings. Independent of tool_use_enforcement — see
# agent/system_prompt.py for the injection gate.
agent._execution_guidance = _agent_section.get("execution_guidance", "auto")
# Wall-clock run budget from config (agent.run_budget_seconds) — only
# consulted when the constructor arg was not given. Absent/None/invalid
# keeps the feature fully off (zero behavior change in the default path).
if agent.run_budget_seconds is None:
agent.run_budget_seconds = _normalize_run_budget_seconds(
_agent_section.get("run_budget_seconds")
)
# Empty-response retry guard config (NS-503): additive
# ``agent.empty_response_guard`` subsection. Resolution is tolerant —
# a malformed section falls back to the schema defaults (guard on,
@@ -1906,6 +1959,11 @@ def init_agent(
# conversation loop's intent-ack block.
agent._intent_ack_continuation = _agent_section.get("intent_ack_continuation", "auto")
# Runtime anti-stall guards (identical-call loop-breaker notice on tool
# results + continue-intent extension of the empty-response recovery).
# Single boolean gate, default True. Notice-only — never blocks a call.
agent._stall_guards = bool(_agent_section.get("stall_guards", True))
# Universal task-completion guidance toggle. Default True. Surfaced
# as a separate flag from tool_use_enforcement because the guidance
# applies to ALL models, not just the model families enforcement
@@ -2949,6 +3007,7 @@ def init_agent(
"client_kwargs": dict(agent._client_kwargs),
"use_prompt_caching": agent._use_prompt_caching,
"use_native_cache_layout": agent._use_native_cache_layout,
"reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False),
# Context engine state that _try_activate_fallback() overwrites.
# Use getattr for model/base_url/api_key/provider since plugin
# engines may not have these (they're ContextCompressor-specific).
+88 -1
View File
@@ -99,7 +99,7 @@ def _ra():
AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset(
{"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "read_window_below", "setup_mcp", "delegate_task"}
{"todo", "session_search", "memory", "clarify", "read_terminal", "read_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"}
)
@@ -1349,6 +1349,7 @@ def try_recover_primary_transport(
if hasattr(agent, "_transport_cache"):
agent._transport_cache.clear()
agent.api_key = rt["api_key"]
agent._reasoning_echo_flag = rt.get("reasoning_echo_flag", False)
if agent.api_mode == "anthropic_messages":
from agent.anthropic_adapter import build_anthropic_client
@@ -1579,6 +1580,7 @@ def restore_primary_runtime(agent) -> bool:
if hasattr(agent, "_transport_cache"):
agent._transport_cache.clear()
agent.api_key = rt["api_key"]
agent._reasoning_echo_flag = rt.get("reasoning_echo_flag", False)
agent._client_kwargs = dict(rt["client_kwargs"])
agent._use_prompt_caching = rt["use_prompt_caching"]
# Default to native layout when the restored snapshot predates the
@@ -2652,6 +2654,7 @@ def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mo
"_anthropic_base_url",
"_is_anthropic_oauth",
"_config_context_length",
"_reasoning_echo_flag",
)
}
# _client_kwargs is a dict — snapshot a shallow copy so mutating the
@@ -2685,6 +2688,9 @@ def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mo
agent.model = new_model
agent.provider = new_provider
agent.requested_provider = new_provider
# Re-read reasoning_echo from config so the flag reflects the new
# primary model's setting (see _reasoning_echo_opt_in).
agent._reasoning_echo_flag = agent._read_reasoning_echo_from_config()
# Use the new base_url when provided. When it's empty AND the
# provider is actually changing, do NOT fall back to the current
# (old provider's) URL — that silently pairs the new provider label
@@ -2970,6 +2976,7 @@ def switch_model(agent, new_model, new_provider, api_key='', base_url='', api_mo
"use_prompt_caching": agent._use_prompt_caching,
"use_native_cache_layout": agent._use_native_cache_layout,
"reasoning_config": dict(agent.reasoning_config) if getattr(agent, "reasoning_config", None) else None,
"reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False),
"compressor_model": getattr(_cc, "model", agent.model) if _cc else agent.model,
"compressor_base_url": getattr(_cc, "base_url", agent.base_url) if _cc else agent.base_url,
"compressor_api_key": getattr(_cc, "api_key", "") if _cc else "",
@@ -3200,6 +3207,7 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i
question=next_args.get("question", ""),
choices=next_args.get("choices"),
multi_select=next_args.get("multi_select", False),
questions=next_args.get("questions"),
callback=agent.clarify_callback,
),
next_args,
@@ -3226,6 +3234,37 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i
),
next_args,
)
elif function_name == "drive_preview":
def _execute(next_args: dict) -> Any:
from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool
return _finish_agent_tool(
_drive_preview_tool(
action=next_args.get("action", ""),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
text=next_args.get("text"),
key=next_args.get("key"),
submit=next_args.get("submit"),
amount=next_args.get("amount"),
to=next_args.get("to"),
limit=next_args.get("max"),
callback=getattr(agent, "drive_preview_callback", None),
),
next_args,
)
elif function_name == "annotate_preview":
def _execute(next_args: dict) -> Any:
from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool
return _finish_agent_tool(
_annotate_preview_tool(
action=next_args.get("action", "add"),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
label=next_args.get("label"),
callback=getattr(agent, "drive_preview_callback", None),
),
next_args,
)
elif function_name == "read_window_below":
def _execute(next_args: dict) -> Any:
from tools.read_window_tool import read_window_below_tool as _read_window_below_tool
@@ -3235,6 +3274,23 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i
),
next_args,
)
elif function_name == "tour":
def _execute(next_args: dict) -> Any:
from tools.tour_tool import tour_tool as _tour_tool
return _finish_agent_tool(
_tour_tool(
action=next_args.get("action", ""),
surface=next_args.get("surface"),
selector=next_args.get("selector"),
title=next_args.get("title"),
text=next_args.get("text"),
side=next_args.get("side"),
steps=next_args.get("steps"),
step_index=next_args.get("step_index"),
callback=getattr(agent, "tour_callback", None),
),
next_args,
)
elif function_name == "setup_mcp":
def _execute(next_args: dict) -> Any:
from tools.setup_mcp_tool import setup_mcp_tool as _setup_mcp_tool
@@ -3838,6 +3894,37 @@ def looks_like_codex_intermediate_ack(
return user_targets_workspace or assistant_targets_workspace
# Conservative "trailing continue-intent" detector for the said-continue-but-
# stopped stall guard (agent.stall_guards). Matches only when the message TAIL
# announces an immediate next action ("Let me now…", "I will now…",
# "Next, I…"), which is the observed stall shape: the model narrates the next
# step and then ends the turn with no tool call. Kept deliberately narrow so
# ordinary answers that merely contain "I will" mid-sentence never trip it.
_TRAILING_CONTINUE_INTENT_RE = re.compile(
r"(?:\blet me now\b|\bi(?:['\u2019])?ll now\b|\bi will now\b"
r"|\bnow i(?:['\u2019]ll| will)\b|\bnext[,:] i\b)"
r"[^.!?\n]{0,100}[.:\u2026]?\s*$",
re.IGNORECASE,
)
# Content longer than this is a substantive reply, not a dangling ack.
_TRAILING_CONTINUE_INTENT_MAX_CHARS = 400
def trailing_continue_intent(text: str) -> bool:
"""Whether ``text`` is a short reply ENDING on an announced next action.
Used by the stall-guard extension of the intent-ack continuation path in
``agent.conversation_loop``: when a turn is about to end with this shape
(no tool calls, short content, trailing intent), the loop re-prompts via
the existing bounded continuation mechanism instead of stopping.
"""
t = (text or "").strip()
if not t or len(t) > _TRAILING_CONTINUE_INTENT_MAX_CHARS:
return False
return bool(_TRAILING_CONTINUE_INTENT_RE.search(t[-160:]))
def intent_ack_continuation_mode(agent) -> str:
"""Classify the resolved intent-ack continuation mode for this turn.
+197 -6
View File
@@ -1551,10 +1551,16 @@ class _CodexCompletionsAdapter:
# Codex backend, which rejects e.g. {"effort": null}
# with a 400.
effort = reasoning_cfg.get("effort") or "medium"
# Codex backend rejects "minimal"; clamp to "low" to
# match the main-agent Codex transport behavior.
if effort == "minimal":
effort = "low"
# Same declared vocabulary + shared clamp as the main
# Codex transport (agent.reasoning_effort): per-model —
# "max" is gpt-5.6-only, "minimal"/"ultra" always
# rejected (live-verified, #68365).
from agent.reasoning_effort import (
clamp_effort,
codex_supported_efforts,
)
effort = clamp_effort(effort, codex_supported_efforts(model))
resp_kwargs["reasoning"] = {
"effort": effort,
"summary": "auto",
@@ -1956,6 +1962,39 @@ class AsyncCodexAuxiliaryClient:
self._real_client = sync_wrapper._real_client
def _translate_anthropic_response_format(
anthropic_kwargs: Dict[str, Any], response_format: Any,
) -> None:
"""Merge an OpenAI response format into Anthropic ``output_config``."""
if not isinstance(response_format, dict):
return
format_type = response_format.get("type")
if format_type == "json_schema":
json_schema = response_format.get("json_schema")
if not isinstance(json_schema, dict) or "schema" not in json_schema:
return
native_format = {
"type": "json_schema",
"schema": json_schema["schema"],
}
elif format_type == "json_object":
# Anthropic SDK 0.87.0 exposes only JSONOutputFormatParam, whose
# required type is ``json_schema``; it has no schema-less JSON mode.
native_format = {
"type": "json_schema",
"schema": {"type": "object"},
}
else:
return
output_config = anthropic_kwargs.get("output_config")
if not isinstance(output_config, dict):
output_config = {}
anthropic_kwargs["output_config"] = output_config
output_config["format"] = native_format
class _AnthropicCompletionsAdapter:
"""OpenAI-client-compatible adapter for Anthropic Messages API."""
@@ -2056,18 +2095,38 @@ class _AnthropicCompletionsAdapter:
# form is the documented Anthropic SDK passthrough for non-standard
# request body keys; merge on top of whatever build_anthropic_kwargs
# already produced (e.g. fast-mode ``speed``) so call-time settings
# survive. Two exclusions:
# survive. Three exclusions:
# - ``reasoning``: the OpenAI-shaped config dict is TRANSLATED into
# the native ``thinking`` field above (build_anthropic_kwargs);
# forwarding the raw field alongside would double-specify
# reasoning and 400 on strict gateways.
# - ``response_format``: the OpenAI structured-output shape is
# TRANSLATED into top-level ``output_config.format`` below;
# forwarding the raw field 400s on strict Anthropic gateways.
# - ``_``-prefixed keys: private Hermes plumbing (_reasoning_config
# et al.), never wire fields.
caller_extra_body = kwargs.get("extra_body")
# A top-level ``response_format`` kwarg (the OpenAI SDK's documented
# call shape) must get the same translation as the extra_body form.
# The adapter builds the Messages body from a fixed allow-list of
# kwargs, so before this an unrecognized top-level kwarg was dropped
# on the floor: the request succeeded but the schema contract
# silently became prompt compliance (#85626 review, point 2). When
# both shapes are present, the extra_body form wins — it is the shape
# every in-tree caller uses.
top_level_response_format = kwargs.get("response_format")
if top_level_response_format is not None:
_translate_anthropic_response_format(
anthropic_kwargs, top_level_response_format,
)
if caller_extra_body and isinstance(caller_extra_body, dict):
_translate_anthropic_response_format(
anthropic_kwargs, caller_extra_body.get("response_format"),
)
passthrough = {
k: v for k, v in caller_extra_body.items()
if k != "reasoning" and not str(k).startswith("_")
if k not in {"reasoning", "response_format"}
and not str(k).startswith("_")
}
if passthrough:
existing = anthropic_kwargs.get("extra_body") or {}
@@ -4315,6 +4374,71 @@ def _is_unsupported_temperature_error(exc: Exception) -> bool:
return _is_unsupported_parameter_error(exc, "temperature")
def _is_structured_output_rejection(exc: Exception) -> bool:
"""Detect provider 400s that reject the structured-output request field.
One predicate covers the field on both wires, because both come from the
same caller-supplied ``response_format``:
- OpenAI wire: the provider rejects ``response_format`` itself. vLLM
gateways translate the field into ``guided_grammar`` and fail when the
grammar backend is absent (``compile_grammar_error: No module named
'xgrammar'``, #82816). Other endpoints answer ``This response_format
type is unavailable now``.
- Anthropic wire: the adapter translates ``response_format`` into
``output_config.format``. Gateways that predate structured outputs
(the documented case is the ``bedrock-mantle`` Messages endpoint)
reject that field: ``output_config: Extra inputs are not permitted``.
Callers tolerate an unconstrained reply — the title prompt demands bare
JSON and ``_extract_title_text`` has a loose-JSON fallback — so the right
reaction is one retry without the field, not a hard failure.
"""
status = getattr(exc, "status_code", None)
if status is not None and status not in {400, 422}:
return False
err_lower = str(exc).lower()
# vLLM grammar-backend failures name the translated parameter, not ours.
if "guided_grammar" in err_lower or "xgrammar" in err_lower or (
"compile_grammar_error" in err_lower
):
return True
if "extra inputs are not permitted" in err_lower and (
"response_format" in err_lower or "output_config" in err_lower
):
return True
if "response_format" in err_lower and "unavailable" in err_lower:
return True
return (
_is_unsupported_parameter_error(exc, "response_format")
or _is_unsupported_parameter_error(exc, "output_config")
)
def _without_structured_output_format(kwargs: dict) -> Optional[dict]:
"""Copy *kwargs* without any ``response_format`` request field.
Removes the top-level kwarg and the ``extra_body`` entry. Returns None
when the kwargs carry no such field, so call sites do not retry a
request that the removal did not change.
"""
changed = False
retry_kwargs = dict(kwargs)
if retry_kwargs.pop("response_format", None) is not None:
changed = True
extra_body = retry_kwargs.get("extra_body")
if isinstance(extra_body, dict) and "response_format" in extra_body:
remaining = {
k: v for k, v in extra_body.items() if k != "response_format"
}
if remaining:
retry_kwargs["extra_body"] = remaining
else:
retry_kwargs.pop("extra_body", None)
changed = True
return retry_kwargs if changed else None
def _is_model_not_found_error(exc: Exception) -> bool:
"""Detect "the requested model doesn't exist" errors (404 / invalid model).
@@ -9488,6 +9612,39 @@ def _call_llm_impl(
first_err = retry_err
kwargs = retry_kwargs
if _is_structured_output_rejection(first_err):
retry_kwargs = _without_structured_output_format(kwargs)
if retry_kwargs is not None:
logger.info(
"Auxiliary %s: provider rejected the structured-output "
"format field; retrying once without it (schema "
"enforcement degrades to prompt compliance): %s",
task or "call", first_err,
)
try:
return _validate_llm_response(
_relay_sync_completion(
client,
retry_kwargs,
provider=resolved_provider,
api_mode=resolved_api_mode,
), task)
except Exception as retry_err:
# Same contract as the temperature rung: fall through to
# the max_tokens / payment / auth chains below with the
# stripped kwargs; re-raise anything those chains do not
# handle.
if not (
_is_payment_error(retry_err)
or _is_connection_error(retry_err)
or _is_auth_error(retry_err)
or "max_tokens" in str(retry_err)
or "unsupported_parameter" in str(retry_err)
):
raise
first_err = retry_err
kwargs = retry_kwargs
err_str = str(first_err)
# ZAI vision models (glm-4v-flash etc.) return error code 1210
# ("API 调用参数有误") when max_tokens is passed on multimodal
@@ -10200,6 +10357,40 @@ async def _async_call_llm_impl(
first_err = retry_err
kwargs = retry_kwargs
if _is_structured_output_rejection(first_err):
retry_kwargs = _without_structured_output_format(kwargs)
if retry_kwargs is not None:
logger.info(
"Auxiliary %s (async): provider rejected the "
"structured-output format field; retrying once without "
"it (schema enforcement degrades to prompt "
"compliance): %s",
task or "call", first_err,
)
try:
return _validate_llm_response(
await _relay_async_completion(
client,
retry_kwargs,
provider=resolved_provider,
api_mode=resolved_api_mode,
), task)
except Exception as retry_err:
# Same contract as the temperature rung: fall through to
# the max_tokens / payment / auth chains below with the
# stripped kwargs; re-raise anything those chains do not
# handle.
if not (
_is_payment_error(retry_err)
or _is_connection_error(retry_err)
or _is_auth_error(retry_err)
or "max_tokens" in str(retry_err)
or "unsupported_parameter" in str(retry_err)
):
raise
first_err = retry_err
kwargs = retry_kwargs
err_str = str(first_err)
# ZAI vision models (glm-4v-flash etc.) return error code 1210
# ("API 调用参数有误") when max_tokens is passed on multimodal
+4
View File
@@ -2652,6 +2652,10 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool
agent.requested_provider = fb_provider
agent.base_url = fb_base_url
agent.api_mode = fb_api_mode
# Per-provider reasoning_content echo opt-in (see _reasoning_echo_opt_in).
# Read from the fallback entry so the flag travels with the active
# provider; restore_primary_runtime will revert it from the snapshot.
agent._reasoning_echo_flag = bool(fb.get("reasoning_echo", False))
if hasattr(agent, "_transport_cache"):
agent._transport_cache.clear()
agent._fallback_activated = True
+60 -1
View File
@@ -1305,6 +1305,62 @@ def _consume_codex_event_stream(
return final
def _sanitize_consumer_codex_request(
agent: Any,
request: dict[str, Any],
) -> dict[str, Any]:
"""Drop fields the ChatGPT OAuth Codex endpoint does not accept.
This guard intentionally lives at the final wire boundary, after Relay or
other request middleware has had a chance to transform the request. The
normal transport builder already omits ``prompt_cache_retention`` for this
endpoint, but a late mutation must not be allowed to turn a valid tool
follow-up into a non-retryable HTTP 400.
Explicit ``request_overrides`` are subject to the same endpoint contract:
unsupported retention is dropped with a warning instead of being sent and
rejected by the provider. The check covers both the top-level kwarg and a
nested ``extra_body`` entry — the OpenAI SDK merges ``extra_body`` into
the outgoing JSON body, so either shape reaches the endpoint.
"""
sanitized = dict(request)
# Resolved defensively on purpose: run_codex_stream is also driven with
# lightweight stand-in agents that carry only the attributes a given path
# needs (see tests/agent/test_codex_request_transport_diagnostics.py), so a
# bare agent._is_codex_backend() here would raise AttributeError on them.
backend_predicate = getattr(agent, "_is_codex_backend", None)
is_consumer_codex = (
bool(backend_predicate()) if callable(backend_predicate) else False
)
if not is_consumer_codex:
return sanitized
dropped_from: list[str] = []
if "prompt_cache_retention" in sanitized:
sanitized.pop("prompt_cache_retention")
dropped_from.append("top-level")
# The OpenAI SDK merges ``extra_body`` into the outgoing JSON body, so a
# nested ``extra_body.prompt_cache_retention`` reaches the endpoint just
# like the top-level field would. Copy before editing — the caller's
# mapping must not be mutated — and drop the mapping when it empties.
extra_body = sanitized.get("extra_body")
if isinstance(extra_body, dict) and "prompt_cache_retention" in extra_body:
extra_body = dict(extra_body)
extra_body.pop("prompt_cache_retention")
if extra_body:
sanitized["extra_body"] = extra_body
else:
sanitized.pop("extra_body")
dropped_from.append("extra_body")
if dropped_from:
logger.warning(
"Dropped unsupported prompt_cache_retention at consumer Codex "
"wire boundary (model=%s, via %s).",
sanitized.get("model", getattr(agent, "model", "unknown")),
", ".join(dropped_from),
)
return sanitized
def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta=None):
"""Execute one streaming Responses API request and return the final response.
@@ -1347,7 +1403,10 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
writer_token = {"value": None}
def _open_codex_stream(next_api_kwargs: dict[str, Any]):
stream_kwargs = dict(next_api_kwargs)
stream_kwargs = _sanitize_consumer_codex_request(
agent,
next_api_kwargs,
)
stream_kwargs["stream"] = True
return active_client.responses.create(**stream_kwargs)
+169
View File
@@ -350,6 +350,147 @@ _SUMMARY_END_MARKER = (
_MERGED_PRIOR_CONTEXT_HEADER = "[PRIOR CONTEXT — for reference only; not a new message]"
_MERGED_SUMMARY_DELIMITER = "[END OF PRIOR CONTEXT — COMPACTION SUMMARY BELOW]"
_SALVAGE_SUMMARY_MAX_CHARS = 8_000
_SALVAGE_KEEP_RECENT_TOOLS = 2
def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool:
# Only cap a standalone handoff. Merged carriers preserve a real tail ask
# in the same content string; truncating those could delete live user text.
if not content.rstrip().endswith(_SUMMARY_END_MARKER):
return False
if content.startswith(_MERGED_PRIOR_CONTEXT_HEADER):
return False
# Content heuristics alone must never authorize mutating a live turn.
# Compressor-generated summaries carry this private marker; ordinary
# user input — and live assistant replies or kept tool bodies that
# merely quote a summary header/marker — do not. Tool messages are
# handled exclusively by the stub/keep-recent pass, never the cap.
if msg.get("role") == "tool":
return False
if (
msg.get("role") in ("user", "assistant")
and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY)
):
return False
head = content[:280]
return (
bool(msg.get(COMPRESSED_SUMMARY_METADATA_KEY))
or "CONTEXT COMPACTION" in head
or "[CONTEXT COMPACTION]" in head
or "Conversation Summary" in head
)
def _salvage_reduce_todo_snapshot(out: List[Dict[str, Any]]) -> None:
"""Last-resort shrink: reduce or drop the synthetic todo snapshot.
The snapshot is the only in-transcript todo re-injection at a compaction
boundary, and since 7a16840add the pruned-skill reload notice is coupled
into the same string — so it is only touched when the cheaper shrink ops
could not get under budget. When the snapshot carries a reload notice,
keep just the notice (the coupling must survive salvage); otherwise drop
the row entirely.
"""
from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER
for i in range(len(out) - 1, -1, -1):
msg = out[i]
if not isinstance(msg, dict):
continue
if msg.get("_todo_snapshot_synthetic") and msg.get("role") == "user":
content = msg.get("content")
notice_idx = (
content.find(_PRUNED_SKILL_RELOAD_NOTICE_HEADER)
if isinstance(content, str)
else -1
)
if isinstance(content, str) and notice_idx >= 0:
msg["content"] = content[notice_idx:]
else:
del out[i]
return
def salvage_grown_transcript(
original: List[Dict[str, Any]],
candidate: List[Dict[str, Any]],
budget: Optional[int] = None,
) -> Optional[List[Dict[str, Any]]]:
"""Mechanically shrink a compression candidate, or return ``None``.
Already-compacted middles can be summarized slightly larger while retained
tool bodies, stale reasoning, or a synthetic todo snapshot tip the final
candidate over the input size. Work on copies and admit the salvage only
when the same rough estimator proves it is strictly smaller than the input.
Shrink order is cheapest-information-loss first: stale reasoning keys and
codex replay sidecars, then old tool bodies, then an oversized summary cap.
The synthetic todo snapshot (which carries the pruned-skill reload notice,
see ``_salvage_reduce_todo_snapshot``) is only reduced as a LAST resort
when everything else still leaves the candidate at or over budget.
"""
if not candidate or not original:
return None
if budget is None:
budget = estimate_messages_tokens_rough(original)
if budget <= 0:
return None
out: List[Dict[str, Any]] = []
tool_indices: List[int] = []
last_assistant_idx = -1
for msg in candidate:
if not isinstance(msg, dict):
out.append(msg)
continue
copied = dict(msg)
out.append(copied)
role = copied.get("role")
if role == "tool":
tool_indices.append(len(out) - 1)
elif role == "assistant":
last_assistant_idx = len(out) - 1
salvage_reasoning_keys = _NEWEST_TURN_ONLY_BUDGET_KEYS + ("reasoning_details",)
keep_tools = set(tool_indices[-_SALVAGE_KEEP_RECENT_TOOLS:])
for index, msg in enumerate(out):
if not isinstance(msg, dict):
continue
if msg.get("role") == "assistant" and index != last_assistant_idx:
for key in salvage_reasoning_keys:
msg.pop(key, None)
if msg.get("role") == "tool" and index not in keep_tools:
content = msg.get("content")
if isinstance(content, str) and len(content) > _PRUNE_MIN_CHARS:
msg["content"] = _PRUNED_TOOL_PLACEHOLDER
content = msg.get("content")
if (
isinstance(content, str)
and len(content) > _SALVAGE_SUMMARY_MAX_CHARS
and _looks_like_compaction_summary(msg, content)
):
msg["content"] = (
content[:_SALVAGE_SUMMARY_MAX_CHARS].rstrip()
+ "\n…[summary truncated so compaction can shrink]\n\n"
+ _SUMMARY_END_MARKER
)
# Heavier codex replay sidecars (encrypted reasoning blobs) — reuse the
# proven prune with its last-user-turn safety boundary (#71058).
_prune_stale_reasoning_replay(out)
if estimate_messages_tokens_rough(out) >= budget:
_salvage_reduce_todo_snapshot(out)
if not any(
isinstance(message, dict) and message.get("role") == "user"
for message in out
):
return None
if estimate_messages_tokens_rough(out) < budget:
return out
return None
# Handoff prefixes that shipped in earlier releases. A summary persisted under
# one of these can be inherited into a resumed lineage (#35344); when it is
# re-normalized on re-compaction we must strip the OLD prefix too, otherwise the
@@ -1885,6 +2026,7 @@ class ContextCompressor(ContextEngine):
self._cooldown_persist_failed = False
self._last_summary_error = None
self._last_compress_aborted = False
self._last_compress_refused_would_grow = False
self.last_real_prompt_tokens = 0
self.last_compression_rough_tokens = 0
self.last_rough_tokens_when_real_prompt_fit = 0
@@ -2167,6 +2309,7 @@ class ContextCompressor(ContextEngine):
self._summary_failure_cooldown_until = 0.0
self._cooldown_persist_failed = False
self._last_compress_aborted = False
self._last_compress_refused_would_grow = False
self._context_probed = False
self._context_probe_persistable = False
self.last_real_prompt_tokens = 0
@@ -2371,6 +2514,31 @@ class ContextCompressor(ContextEngine):
self._ineffective_compression_count = count
self._persist_ineffective_compression_count()
def record_rejected_compaction(self) -> None:
"""Record one compaction whose result was REJECTED before committing.
The anti-growth guard in the commit layer (conversation_compression)
discards a candidate that would grow the transcript and keeps the
original. Without recording the attempt, the anti-thrash breaker
never sees a strike, so automatic compression retries the SAME
unchanged transcript on every turn — same summary request, same
refusal, same user-facing warning (#88568). This counts one
ineffective strike (persisted, so the normal >= 2 latch and its
recovery window apply) WITHOUT arming post-compaction real-usage
verification — nothing was committed, so there is no new compaction
to verify — and without touching the fallback-summary streak (no
summary was accepted).
"""
self._record_ineffective_compression_verdict(
self._ineffective_compression_count + 1
)
if not self.quiet_mode:
logger.warning(
"Compaction rejected before commit (would grow the "
"transcript); ineffective_compression_count=%d",
self._ineffective_compression_count,
)
def record_completed_compaction(
self, *, used_fallback: bool = False, feasibility_skip: bool = False,
) -> None:
@@ -6927,6 +7095,7 @@ This compaction should PRIORITISE preserving all information related to the focu
self._last_aux_model_failure_error = None
self._last_aux_model_failure_model = None
self._last_compress_aborted = False
self._last_compress_refused_would_grow = False
self._last_compression_made_progress = False
# NOTE: do NOT reset _last_summary_auth_failure or
# _last_summary_network_failure here. These flags are set by
+46
View File
@@ -3386,6 +3386,27 @@ def compress_context(
# transcript stays untouched and durable.
_rough_in = estimate_messages_tokens_rough(messages)
_rough_out = estimate_messages_tokens_rough(compressed)
if _rough_out > _rough_in:
# Todo refresh and user-turn anchoring happen after the
# compressor's own size check, so they can tip a break-even
# candidate over. Give it one mechanical salvage pass.
from agent.context_compressor import salvage_grown_transcript
_salvaged = salvage_grown_transcript(
messages, compressed, budget=_rough_in
)
if _salvaged is not None:
_salv_est = estimate_messages_tokens_rough(_salvaged)
if _salv_est < _rough_in:
logger.info(
"Compression salvage recovered a shrinking "
"transcript (session=%s, ~%s -> ~%s tokens)",
agent.session_id or "none",
f"{_rough_in:,}",
f"{_salv_est:,}",
)
compressed = _salvaged
_rough_out = _salv_est
if _rough_out > _rough_in:
logger.warning(
"Compression refused: compressed transcript would be "
@@ -3395,6 +3416,17 @@ def compress_context(
f"{_rough_in:,}",
f"{_rough_out:,}",
)
# Flag the refusal on the compressor state so manual
# /compress feedback can report it honestly. Without this,
# the CLI compared the returned list against its pre-call
# snapshot, saw a difference (durable-snapshot adoption can
# legitimately change the count), and printed
# "✅ Compressed: 8 → 14 messages" directly under the
# refusal warning (Aug 2026 full-surface CLI QA sweep).
try:
agent.context_compressor._last_compress_refused_would_grow = True
except Exception:
pass
try:
agent._emit_warning(
"⚠️ Compression refused: the generated summary "
@@ -3414,6 +3446,20 @@ def compress_context(
split_status="aborted",
failure_class="would_grow",
)
# Record the rejected attempt as an ineffective
# compaction strike so the anti-thrash breaker latches
# after the normal threshold. Without this, the unchanged
# transcript stays over the compression threshold and
# automatic compression retries the identical summary
# request on every turn (#88568). Manual /compress keeps
# bypassing the latch (force=True skips the guards).
try:
agent.context_compressor.record_rejected_compaction()
except Exception:
logger.debug(
"could not record rejected-compaction strike",
exc_info=True,
)
_release_lock()
return messages, _existing_sp
+174 -3
View File
@@ -107,6 +107,64 @@ logger = logging.getLogger(__name__)
_INTERRUPT_SCAFFOLD_MARKER = "[This response was interrupted by a user correction.]"
# One-time wrap-up notice appended when a wall-clock run budget crosses its
# 80% threshold (agent.run_budget_seconds / --run-budget). Mirrors the Codex
# CLI budget wrap-up template: stop new work, deliver from current state.
RUN_BUDGET_WRAPUP_NOTICE = (
"[SYSTEM NOTICE — run time budget nearly exhausted] "
"Run time budget nearly exhausted. Stop new discovery/verification work "
"now. Produce the required final deliverable (answer/JSON/summary) from "
"the state you already have, completing only mandatory writes."
)
def _maybe_inject_run_budget_wrapup(agent: Any, messages: List[Dict[str, Any]]) -> bool:
"""Inject the one-time wall-clock wrap-up notice when past 80% of budget.
Cache-safe delivery: the notice is appended to the NEWEST ``role:"tool"``
message (the same channel /steer uses) — no synthetic user message is
inserted mid-loop and no past context is rewritten, so role alternation
and the prompt-cache prefix survive. Latches ``_run_budget_wrapup_injected``
only on a successful append, so a first iteration without tool results
retries on the next iteration. Returns True when the notice was injected.
Dormant unless ``agent.run_budget_seconds`` is set AND the turn stamped
``_run_budget_started_at`` (see ``turn_context.prepare_conversation_turn``).
"""
budget = getattr(agent, "run_budget_seconds", None)
if not budget:
return False
if getattr(agent, "_run_budget_wrapup_injected", False):
return False
started = getattr(agent, "_run_budget_started_at", None)
if not started:
return False
if (time.time() - started) < 0.8 * float(budget):
return False
for i in range(len(messages) - 1, -1, -1):
msg = messages[i]
if isinstance(msg, dict) and msg.get("role") == "tool":
existing = msg.get("content", "")
if isinstance(existing, str):
msg["content"] = existing + f"\n\n{RUN_BUDGET_WRAPUP_NOTICE}"
else:
# Multimodal content blocks — append a text block.
try:
blocks = list(existing) if existing else []
blocks.append({"type": "text", "text": RUN_BUDGET_WRAPUP_NOTICE})
msg["content"] = blocks
except Exception:
return False
agent._run_budget_wrapup_injected = True
logger.info(
"Run budget wrap-up notice injected (budget=%.0fs, elapsed=%.0fs)",
float(budget),
time.time() - started,
)
return True
return False
def _restore_user_after_reference_handoff(
messages: List[Dict[str, Any]], user_message: Any
) -> bool:
@@ -331,6 +389,19 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text
}
if not visible:
placeholder["display_kind"] = "hidden"
# Keep the transcript hidden and empty, but give the historical
# API projection a non-empty neutral assistant turn so the
# pre-call sanitizer (repair_empty_non_final_messages) does not
# re-heal this row on every later call (#88955). display_kind is
# stripped before sanitization, while api_content is projected
# back into content for historical assistant rows. Use the
# canonical neutral interruption placeholder, never
# _INTERRUPT_SCAFFOLD_MARKER: replaying the scaffold as assistant
# text made the model echo it and self-replicate ghost rows
# (#81841).
from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER
placeholder["api_content"] = _INTERRUPTED_PLACEHOLDER
append_message(messages, placeholder)
append_message(
messages,
@@ -2028,6 +2099,15 @@ def run_conversation(
existing = getattr(agent, "_pending_steer", None)
agent._pending_steer = (existing + "\n" + _pre_api_steer) if existing else _pre_api_steer
# ── Wall-clock run-budget wrap-up notice ───────────────────────
# One-shot: when a run budget (agent.run_budget_seconds /
# --run-budget) is active and 80% of it has elapsed, ask the model
# to wrap up and deliver from the state it already has. Same
# cache-safe channel as /steer (appended to the newest tool
# result); dormant when no budget is set.
if getattr(agent, "run_budget_seconds", None):
_maybe_inject_run_budget_wrapup(agent, messages)
# Prepare messages for API call
# If we have an ephemeral system prompt, prepend it to the messages
# Note: Reasoning is embedded in content via <think> tags for trajectory storage.
@@ -2123,9 +2203,29 @@ def run_conversation(
# from every outgoing copy so strict OpenAI-compatible backends
# don't reject the request after a model switch or resumed typed
# event row enters the live history.
api_msg.pop("display_kind", None)
_display_kind = api_msg.pop("display_kind", None)
api_msg.pop("display_metadata", None)
# Legacy hidden redirect placeholders (#88955): rows persisted
# BEFORE the writer-side api_content stamp in
# _apply_active_turn_redirect are content="" with no sidecar.
# Once display_kind is stripped the pre-call sanitizer
# (repair_empty_non_final_messages) would re-heal such a row on
# every call forever, since the durable transcript is never
# mutated. Give the wire copy the same neutral payload here so
# old sessions converge too. Never the interrupt scaffold —
# replaying scaffold bytes as assistant text is #81841.
if (
_display_kind == "hidden"
and api_msg.get("role") == "assistant"
and not _api_content
and not (api_msg.get("content") or "").strip()
and not api_msg.get("tool_calls")
):
from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER
api_msg["content"] = _INTERRUPTED_PLACEHOLDER
# Durable row identity stamped by _rows_to_conversation so the
# desktop can address a specific persisted message (reactions).
# Bookkeeping, never a provider field — only the chat-completions
@@ -2665,6 +2765,33 @@ def run_conversation(
request_pressure_tokens,
int(getattr(_compressor, "threshold_tokens", 0) or 0),
)
elif not agent.compression_enabled and len(messages) > 1:
# Uncompressed session guard (#89297): compression is disabled, so
# nothing shrinks a growing session. Reuse the unconditionally
# computed request estimate (zero marginal cost — this site runs
# before every provider request, covering turn-start AND mid-turn
# tool-result growth) and surface a deduped, actionable warning
# when the request exceeds the model context window. The dedup is
# re-armed by the turn-context preflight once the session is back
# under the window (manual /compress works with compression
# disabled), so the guard warns again on a later re-overflow.
# context_compressor always exists (agent_init constructs it even
# when compression is disabled) and its context_length property
# hard-floors at a positive default — no metadata re-resolution
# needed here.
_ctx_len = getattr(
getattr(agent, "context_compressor", None), "context_length", None
)
if (
isinstance(_ctx_len, int)
and _ctx_len > 0
and request_pressure_tokens > _ctx_len
):
_warn_fn = getattr(
agent, "_warn_uncompressed_context_overflow", None
)
if callable(_warn_fn):
_warn_fn(request_pressure_tokens, _ctx_len)
# Thinking spinner for quiet mode (animated during API call)
thinking_spinner = None
@@ -5204,6 +5331,21 @@ def run_conversation(
FailoverReason.billing,
FailoverReason.upstream_rate_limit,
}
# Relay-wrapped output-cap errors: some gateways wrap an
# upstream "[400]: max_tokens (...) exceeds model's maximum
# output tokens (...)" as HTTP 429, which classifies as
# rate_limit. The failure is a deterministic request-shape
# problem — falling back to another provider (or burning
# generic retries) can't fix it, but the output-cap clamp
# below can, in one retry (#72281). Parse once here; the
# result gates both the eager-fallback exemption and the
# widened is_context_length_error entry, and is reused as
# available_out inside the handler.
_wrapped_output_cap_budget = (
parse_available_output_tokens_from_error(error_msg)
if classified.reason == FailoverReason.rate_limit
else None
)
_is_transport_failure = classified.reason in {
FailoverReason.timeout,
FailoverReason.overloaded,
@@ -5221,7 +5363,7 @@ def run_conversation(
if _is_zai_coding_overload:
max_retries = max(max_retries, zai_coding_overload_retry_ceiling())
_should_fallback = (
is_rate_limited
(is_rate_limited and _wrapped_output_cap_budget is None)
or (_is_transport_failure and retry_count >= 2)
)
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
@@ -5514,6 +5656,11 @@ def run_conversation(
# server disconnect + large session pattern (#2153).
is_context_length_error = (
classified.reason == FailoverReason.context_overflow
# Relay-wrapped output-cap 429s (parsed once above, where
# the eager-fallback exemption is gated) route into the
# output-cap clamp below instead of provider failover or
# generic retries (#72281).
or _wrapped_output_cap_budget is not None
)
if is_context_length_error:
@@ -7831,10 +7978,28 @@ def run_conversation(
from agent.agent_runtime_helpers import (
intent_ack_continuation_mode,
trailing_continue_intent,
)
_ack_mode = intent_ack_continuation_mode(agent)
if (
# Said-continue-but-stopped guard (agent.stall_guards): the
# model ended the turn with no tool calls but its short reply
# TAILS with an announced next action ("Let me now…",
# "I will now…"). Unlike the intent-ack detector below, this
# fires mid-task too (after tool results), which is exactly
# where eval traces show the stall. It reuses the SAME bounded
# continuation path and counter (max 2 per turn), so the
# alternation-safe interim-assistant + user-nudge mechanism —
# not a new parallel one — carries the recovery.
_stall_continue_intent = (
bool(getattr(agent, "_stall_guards", True))
and agent.valid_tool_names
and codex_ack_continuations < 2
and trailing_continue_intent(
agent._strip_think_blocks(final_response or "")
)
)
if _stall_continue_intent or (
_ack_mode != "off"
and agent.valid_tool_names
and codex_ack_continuations < 2
@@ -7845,6 +8010,12 @@ def run_conversation(
require_workspace=(_ack_mode == "codex_only"),
)
):
if _stall_continue_intent:
logger.info(
"Stall guard: turn ending on trailing continue-"
"intent with no tool calls — re-prompting to act "
"(%d/2)", codex_ack_continuations + 1,
)
codex_ack_continuations += 1
interim_msg = agent._build_assistant_message(assistant_message, "incomplete")
append_message(messages, interim_msg)
+12
View File
@@ -139,6 +139,18 @@ def get_active_provider() -> Optional[ImageGenProvider]:
except Exception as exc:
logger.debug("Could not read image_gen.provider from config: %s", exc)
# The managed "Nous Subscription" selection is serviced by the FAL
# plugin through the managed fal-queue gateway (the legacy FAL pipeline
# routes managed when the stored selection is "nous").
if configured:
try:
from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER
if configured.lower() == NOUS_MANAGED_PROVIDER:
configured = "fal"
except Exception: # pragma: no cover — helpers are in-repo
pass
with _lock:
snapshot = dict(_providers)
snapshot.update(_scoped_providers.get(hermes_home_key(), {}))
+20 -2
View File
@@ -53,6 +53,11 @@ def summarize_manual_compression(
compression_state is not None
and getattr(compression_state, "_last_compress_aborted", False) is True
)
refused_would_grow = (
compression_state is not None
and getattr(compression_state, "_last_compress_refused_would_grow", False)
is True
)
fallback_used = (
compression_state is not None
and getattr(compression_state, "_last_summary_fallback_used", False) is True
@@ -65,7 +70,12 @@ def summarize_manual_compression(
if not isinstance(failure_reason, str) or not failure_reason.strip():
failure_reason = None
if aborted:
if refused_would_grow:
headline = (
f"Compression refused (summary would grow the conversation): "
f"{before_count} messages preserved"
)
elif aborted:
headline = f"Compression aborted: {before_count} messages preserved"
elif fallback_used:
headline = (
@@ -78,6 +88,8 @@ def summarize_manual_compression(
if noop and after_tokens == before_tokens:
token_line = f"Approx request size: ~{before_tokens:,} tokens (unchanged)"
elif refused_would_grow:
token_line = f"Approx request size: ~{before_tokens:,} tokens (unchanged)"
else:
token_line = (
f"Approx request size: ~{before_tokens:,} → "
@@ -85,7 +97,12 @@ def summarize_manual_compression(
)
note = None
if aborted:
if refused_would_grow:
note = (
"The generated summary was larger than what it would replace; "
"no messages were removed."
)
elif aborted:
note = "Summary generation failed; no messages were removed."
elif fallback_used:
dropped_count = getattr(
@@ -113,6 +130,7 @@ def summarize_manual_compression(
return {
"noop": noop,
"aborted": aborted,
"refused_would_grow": refused_would_grow,
"fallback_used": fallback_used,
"headline": headline,
"token_line": token_line,
+37
View File
@@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
# The input itself fits — this is purely an output-cap error, so reduce
# max_tokens and retry; do NOT compress.
"range of max_tokens should be" in error_lower
) or (
# OpenAI-compatible relays may reject a request whose output cap exceeds
# the model's separate completion-token limit, e.g.
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
# This is independent of the input context window.
"exceeds model" in error_lower
and "maximum output tokens" in error_lower
)
if not is_output_cap_error:
return None
# Generic model-output-cap form:
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
_m_max_output = re.search(
r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?',
error_lower,
)
if _m_max_output:
_cap = int(_m_max_output.group(1))
if _cap >= 1:
return _cap
# DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]".
# The upper bound is the available output cap.
_m_range = re.search(
@@ -1726,11 +1744,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
# Available output = window - input. When the input alone is at or over
# the window this stays None, so the caller correctly falls through to
# compression instead of futilely shrinking the output cap.
#
# Caveat: when max_tokens is the BINDING constraint, vLLM does not report
# the real prompt size at all. It back-computes a lower bound from the
# constraint itself -- "at least N input tokens" where
# N == window + 1 - requested_output -- so window - N is always exactly
# requested_output - 1. Subtracting the caller's safety margin then walks
# the cap down ~65 tokens per retry while the reported input walks up by
# the same amount, burning every compression attempt without ever fitting.
# Detect that degenerate case and halve the requested cap instead: it
# carries the same guarantee (strictly below what was rejected) and
# converges in one or two retries.
_m_vllm_input = re.search(
r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower
)
if _m_ctx_tok and _m_vllm_input:
_available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1))
_m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower)
if 'at least' in error_lower and _m_requested_out:
_requested_out = int(_m_requested_out.group(1))
if _available >= _requested_out - 1:
# The budget is derived from the constraint, not measured.
return max(1, _requested_out // 2)
if _available >= 1:
return _available
@@ -1782,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool:
or "should be" in error_lower # generic "max_tokens should be <= N"
or "less than or equal" in error_lower
or "must be" in error_lower
or ("exceeds model" in error_lower
and "maximum output tokens" in error_lower)
)
if not output_cap_signal:
return False
+67 -2
View File
@@ -191,6 +191,21 @@ MEMORY_GUIDANCE = (
"workflows belong in skills, not memory."
)
USER_PROFILE_GUIDANCE = (
"You have a persistent user profile across sessions. Save durable facts about "
"the user with the memory tool (target='user'): name, role, preferences, "
"corrections, and communication style. The profile is injected into every turn, "
"so keep it compact and focused on facts that will still matter later.\n"
"The built-in memory notes store is disabled — write only to the user profile "
"(target='user'), never target='memory'.\n"
"Prioritize what reduces future user steering — the most valuable entry is one "
"that prevents the user from having to correct or remind you again.\n"
"Write entries as declarative facts, not instructions to yourself. "
"'User prefers concise responses' ✓ — 'Always respond concisely' ✗. "
"Imperative phrasing gets re-read as a directive in later sessions and can "
"cause repeated work or override the user's current request."
)
SESSION_SEARCH_GUIDANCE = (
"When the user references something from a past conversation or you suspect "
"relevant cross-session context exists, use session_search to recall it before "
@@ -358,6 +373,25 @@ TOOL_USE_ENFORCEMENT_GUIDANCE = (
# Add new patterns here when a model family needs explicit steering.
TOOL_USE_ENFORCEMENT_MODELS = ("gpt", "codex", "gemini", "gemma", "grok", "glm", "qwen", "deepseek")
# Model name substrings whose sessions receive OPENAI_MODEL_EXECUTION_GUIDANCE
# (execution discipline: tool persistence, mandatory tool use for arithmetic,
# external-write read-back, count reconciliation, literal preservation,
# verification-gated completion) when agent.execution_guidance is "auto".
#
# gpt/codex/grok are the historical set; deepseek/kimi/qwen/glm/minimax/
# mimo/mistral were added after Composio agentic-eval traces showed the same
# failure modes on those families (financial math in prose, no read-back after
# external writes, identifier "repair", completeness claims despite count
# mismatches). GLM's tool-calls-as-plain-text stall (#53847) and MiMo (#41874)
# are covered here too. Gemini/Gemma are excluded — they get the more specific
# GOOGLE_MODEL_OPERATIONAL_GUIDANCE block instead. Claude is excluded because
# it does not exhibit these failure modes; users can opt any model in via
# config.yaml `agent.execution_guidance: true` or a substring list.
EXECUTION_GUIDANCE_MODELS = (
"gpt", "codex", "grok",
"deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral",
)
# Universal "finish the job" guidance — applied to ALL models, not gated
# by model family. Addresses two cross-model failure modes:
# 1. Stopping after a stub: writing a tiny file or running one command
@@ -438,13 +472,22 @@ PARALLEL_TOOL_CALL_GUIDANCE = (
# without tool calls, suggests workarounds instead of using existing tools,
# replies with plans/suggestions instead of executing). The body is
# family-agnostic; the OPENAI_ prefix reflects origin, not exclusivity.
#
# As of the Composio agentic-eval follow-up, the block is no longer fenced to
# gpt/codex/grok: eval traces showed DeepSeek/Kimi doing financial math in
# prose, skipping read-back verification after external writes, "repairing"
# malformed identifiers, and claiming completeness despite count mismatches —
# exactly the failure modes this block targets. The injection gate lives in
# agent/system_prompt.py and is controlled by config.yaml
# ``agent.execution_guidance`` (auto/true/false/list); "auto" matches the
# EXECUTION_GUIDANCE_MODELS substring tuple below.
OPENAI_MODEL_EXECUTION_GUIDANCE = (
"# Execution discipline\n"
"<tool_persistence>\n"
"- Use tools whenever they improve correctness, completeness, or grounding.\n"
"- Do not stop early when another tool call would materially improve the result.\n"
"- If a tool returns empty or partial results, retry with a different query or "
"strategy before giving up.\n"
"- If a tool returns empty, partial, or suspiciously narrow results, retry "
"with a broader or different query or strategy before concluding.\n"
"- Keep calling tools until: (1) the task is complete, AND (2) you have verified "
"the result.\n"
"</tool_persistence>\n"
@@ -487,8 +530,30 @@ OPENAI_MODEL_EXECUTION_GUIDANCE = (
"- Formatting: does the output match the requested format or schema?\n"
"- Safety: if the next step has side effects (file writes, commands, API calls), "
"confirm scope before executing.\n"
"- Completion: 'done' means every named acceptance criterion is verified — "
"never a plausible subset. Completing your plan is not itself the answer; "
"the requested output must appear in your response.\n"
"</verification>\n"
"\n"
"<external_state_verification>\n"
"- After any state-changing write to an external system (API call, message "
"post, record update), verify the effect by reading back the exact target "
"before claiming success — a successful tool call is not a successful task. "
"Do NOT re-verify internal file edits a tool already confirmed.\n"
"- Declared totals in responses (total, reply_count, has_more, '...N more') "
"are hard assertions. If your enumerated count disagrees, re-fetch or parse "
"programmatically — never finalize on 'go with what I have'.\n"
"- When building write payloads, set fields explicitly rather than relying "
"on provider defaults that could contradict intent.\n"
"</external_state_verification>\n"
"\n"
"<literal_preservation>\n"
"- Preserve identifiers, commands, and values exactly as given — never "
"'repair' or normalize a token that fails a stated format. A successful "
"lookup does not validate a malformed source token; validate format first, "
"then look up.\n"
"</literal_preservation>\n"
"\n"
"<missing_context>\n"
"- If required context is missing, do NOT guess or hallucinate an answer.\n"
"- Use the appropriate lookup tool when missing information is retrievable "
+213
View File
@@ -0,0 +1,213 @@
"""Canonical reasoning-effort vocabulary and wire clamping.
Hermes' internal effort ladder (``hermes_constants.VALID_REASONING_EFFORTS``
plus the ``none`` disable level) is wider than what any single provider wire
accepts. Historically every transport and provider profile hand-rolled its own
translation map, and the class of bugs that produced was constant: a new
internal level (``ultra``) leaking to a wire that rejects it with HTTP 400
(#89503, #70058), or an unknown level being dropped to a weak default so the
strongest ask resolved *weaker* than an explicit ``high`` — a ladder
inversion (#74295, #87279).
This module is the single source of truth both kinds of code use instead:
- :data:`EFFORT_LADDER` — canonical low→high ordering.
- :func:`clamp_effort` — the one clamping policy: keep a supported level
verbatim, otherwise take the **nearest weaker** supported level (never
silently escalate cost above what was asked), and only when nothing weaker
exists take the weakest supported level (a provider whose minimum thinking
level is ``high`` serves ``high`` for a ``low`` ask — GLM-5.2's shape).
- Named wire-vocabulary constants for the common OpenAI-compatible surfaces,
so call sites declare *data* ("this route accepts these levels") rather
than logic.
Rules for call sites:
1. **Wire shape stays local.** Whether a route wants ``extra_body.reasoning``,
a top-level ``reasoning_effort`` string, or a ``thinking`` toggle is the
caller's business. Only the *vocabulary math* lives here.
2. **Unset stays unset.** ``clamp_effort`` translates an explicit request; it
does not invent one. When the user expressed no effort, prefer omitting
the field so the server default applies.
3. **Never patch a predicate.** When a provider rejects a level, fix its
declared supported set (data), never add another vendor-name special case
at the call site.
"""
from __future__ import annotations
import re
from typing import Optional, Sequence
#: K3 slug detector — matches ``k3`` as a delimited token (``k3``,
#: ``k3-256k``, ``kimi-k3``, ``kimi-k3-cot``) without matching K2-era names
#: (``kimi-k2.6``). From #76427 by @ruizanthony.
_KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)")
# Canonical low→high ordering used for nearest-level clamping. Superset of
# hermes_constants.VALID_REASONING_EFFORTS ("none" included so an explicit
# disable can be clamped too when a provider publishes it as a level).
EFFORT_LADDER: tuple[str, ...] = (
"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra",
)
# ``ultra`` is Hermes-internal ladder vocabulary (the Codex product tier); no
# provider wire accepts it verbatim anywhere. Every declared wire set below
# therefore stops at ``max`` — ``ultra`` always clamps down.
#: The widest OpenAI-compatible wire vocabulary (OpenRouter, Nous Portal):
#: exactly max|xhigh|high|medium|low|minimal|none.
OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = (
"none", "minimal", "low", "medium", "high", "xhigh", "max",
)
#: OpenAI/Codex Responses backend — per-model vocabulary, live-verified
#: (Aug 2026): ``minimal`` is rejected by both generations (clamps to low);
#: ``max`` is gpt-5.6-only — gpt-5.5 rejects it with "Supported values are:
#: 'none', 'low', 'medium', 'high', 'xhigh'" (#68365's premise, confirmed).
CODEX_GPT56_EFFORTS: tuple[str, ...] = (
"none", "low", "medium", "high", "xhigh", "max",
)
CODEX_LEGACY_EFFORTS: tuple[str, ...] = (
"none", "low", "medium", "high", "xhigh",
)
def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
"""Supported effort set for an OpenAI/Codex Responses model."""
if "gpt-5.6" in (model or "").lower():
return CODEX_GPT56_EFFORTS
return CODEX_LEGACY_EFFORTS
#: Backward-compat alias (pre-#68365-verification name).
CODEX_RESPONSES_EFFORTS: tuple[str, ...] = CODEX_GPT56_EFFORTS
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Actual Computer relays (SGLang/vLLM): none/low/medium/high/max.
ACTUAL_RELAY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "max")
#: Moonshot/Kimi K3: low/high/max (server default high).
KIMI_K3_EFFORTS: tuple[str, ...] = ("low", "high", "max")
#: Moonshot/Kimi K2-era models: low/medium/high.
KIMI_K2_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Tencent TokenHub: low/medium/high.
TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Kimi K3's vendor-documented translation quirks (platform.kimi.ai
#: thinking-model guide): ``high`` is K3's positional middle AND server
#: default, so ``medium`` rounds to it rather than down to ``low``; ``xhigh``
#: rounds up to ``max`` (K3's top tier), matching the kimi-coding plugin.
KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"}
#: GLM-5.2 native reasoning_effort knob: exactly two enabled levels,
#: ``high`` (its minimum thinking level) and ``max`` (per Z.AI/BigModel
#: docs). ``xhigh`` requests the top tier, not the floor.
GLM52_EFFORTS: tuple[str, ...] = ("high", "max")
GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: DeepSeek V4 OpenAI-compat endpoint: low/medium/high/max; ``xhigh``
#: requests the top tier (matches the shipped profile mapping).
DEEPSEEK_V4_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
DEEPSEEK_V4_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: Ollama Cloud /v1/chat/completions: accepts {none, low, medium, high, max};
#: rejects ``minimal`` with HTTP 400. ``xhigh`` requests the top tier.
OLLAMA_CLOUD_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "max")
OLLAMA_CLOUD_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: Meta Model API (Muse): minimal..xhigh; rejects ``none``.
META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh")
#: Upstage Solar Pro/Open: low/medium/high.
SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
"""Supported effort set for a Moonshot/Kimi model slug.
K3 is served as the bare slug ``k3``, plan variants like ``k3-256k``,
and the ``kimi-k3*`` aliases; its documented set is low/high/max.
Everything earlier speaks low/medium/high. Boundary-matched so K2-era
names (``kimi-k2.6``) never match (detection regex from #76427 by
@ruizanthony).
"""
m = (model or "").strip().lower().split("/")[-1]
if _KIMI_K3_SLUG_RE.search(m):
return KIMI_K3_EFFORTS
return KIMI_K2_EFFORTS
def clamp_effort(
effort: Optional[str],
supported: Optional[Sequence[str]],
overrides: Optional[dict[str, str]] = None,
) -> Optional[str]:
"""Clamp a requested reasoning effort onto a wire's supported levels.
``overrides`` is an optional declared mapping consulted first, for routes
whose vendor documents a translation that differs from nearest-weaker
(Kimi K3 documents ``medium → high``: high is its positional middle and
server default). Overrides are data, not logic — a call site never adds
vendor ``if``\\ s around this function.
Otherwise: returns the requested effort unchanged when it is supported,
when the supported set is unknown (``None``/empty), or when the effort
isn't a recognized ladder level (custom providers may use bespoke names —
pass through rather than guess). Otherwise returns the **nearest weaker**
supported level, so a clamp never silently escalates cost; when nothing
weaker exists, the weakest supported level is returned (the caller asked
for *some* thinking and the provider's floor is the closest honest match).
The policy is monotonic: a stronger request never resolves to a weaker
wire level than a weaker request would.
"""
requested = str(effort or "").strip().lower()
if not requested or not supported:
return effort
supported_norm = [
str(level).strip().lower()
for level in supported
if str(level).strip().lower() in EFFORT_LADDER
]
if not supported_norm or requested in supported_norm:
return effort
if overrides:
mapped = overrides.get(requested)
if mapped in supported_norm:
return mapped
if requested not in EFFORT_LADDER:
return effort
# "none" disables reasoning — it is never a *degradation target* for an
# enabled ask (clamping "minimal" to "none" would silently switch
# thinking off). It still passes through verbatim when requested.
candidates = [level for level in supported_norm if level != "none"]
if not candidates:
return effort
requested_idx = EFFORT_LADDER.index(requested)
below = [
level for level in candidates
if EFFORT_LADDER.index(level) < requested_idx
]
if below:
return max(below, key=EFFORT_LADDER.index)
return min(candidates, key=EFFORT_LADDER.index)
def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]:
"""Extract the user's explicit effort from a reasoning config, or None.
Returns ``None`` when the config is absent, malformed, carries no effort,
or reasoning is explicitly disabled — callers should then omit the wire
field entirely so the server default applies (rule 2 above).
"""
if not isinstance(reasoning_config, dict):
return None
if reasoning_config.get("enabled") is False:
return None
effort = str(reasoning_config.get("effort") or "").strip().lower()
return effort or None
+31 -1
View File
@@ -434,6 +434,7 @@ class ManagedLlmStream(Iterator[Any]):
self._stream: Any = None
self._raw_stream_resource: Any = None
self._closed = False
self._runtime_lease: relay_runtime.RelayOperationLease | None = None
self._close_error: BaseException | None = None
self._callback_error: BaseException | None = None
self._logical: tuple[relay_runtime.RelayTurnContext, Any, str] | None = None
@@ -573,7 +574,12 @@ class ManagedLlmStream(Iterator[Any]):
self._callback_error = exc
raise
self._runtime_lease = runtime.acquire_operation_lease()
try:
loop = asyncio.new_event_loop()
except BaseException:
self._release_runtime_lease()
raise
self._loop = loop
self._relay_observes_chunks = True
try:
@@ -613,10 +619,14 @@ class ManagedLlmStream(Iterator[Any]):
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
operation_lease=self._runtime_lease,
)
self._logical = None
try:
loop.close()
finally:
self._loop = None
self._release_runtime_lease()
raise
def __iter__(self) -> "ManagedLlmStream":
@@ -662,6 +672,7 @@ class ManagedLlmStream(Iterator[Any]):
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
operation_lease=self._runtime_lease,
)
self._logical = None
self._close(logical_outcome="cancelled")
@@ -719,6 +730,7 @@ class ManagedLlmStream(Iterator[Any]):
self._stream = iter(pending)
self._raw_stream_resource = None
self._accept_chunk = None
try:
if loop is not None:
close = getattr(relay_stream, "aclose", None)
if callable(close):
@@ -741,14 +753,18 @@ class ManagedLlmStream(Iterator[Any]):
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
operation_lease=self._runtime_lease,
)
self._logical = None
finally:
self._release_runtime_lease()
def _close(self, *, logical_outcome: str) -> None:
if self._closed:
return
self._closed = True
self._prefetched_chunks.clear()
try:
loop = self._loop
self._loop = None
if loop is None:
@@ -778,6 +794,7 @@ class ManagedLlmStream(Iterator[Any]):
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
operation_lease=self._runtime_lease,
)
self._logical = None
return
@@ -799,9 +816,18 @@ class ManagedLlmStream(Iterator[Any]):
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
operation_lease=self._runtime_lease,
)
self._logical = None
loop.close()
finally:
self._release_runtime_lease()
def _release_runtime_lease(self) -> None:
lease = self._runtime_lease
self._runtime_lease = None
if lease is not None:
lease.release()
def __del__(self) -> None:
self._close(logical_outcome="cancelled")
@@ -941,6 +967,7 @@ def _complete_logical(
model_name: str | None = None,
provider_name: str | None = None,
response_model_name: str | None = None,
operation_lease: relay_runtime.RelayOperationLease | None = None,
) -> None:
if logical is None:
return
@@ -960,7 +987,10 @@ def _complete_logical(
output.update({"model": model_name, "provider": provider_name})
if response_model_name is not None:
output["response_model"] = response_model_name
lease.host.run_in_session(
callback = lease.host.run_in_session
if operation_lease is not None:
callback = operation_lease.run_in_session
callback(
lease.session,
relay_runtime.pop_relay_scope,
lease.host.relay,
+469 -18
View File
@@ -8,13 +8,21 @@ import contextvars
import importlib
import inspect
import logging
import os
import threading
import tomllib
import uuid
from concurrent.futures import TimeoutError as FuturesTimeoutError
from dataclasses import dataclass, field
from enum import Enum, auto
from pathlib import Path
from typing import Any, Callable
from hermes_constants import get_hermes_home
from hermes_cli.relay_plugin_cutover import (
RELAY_PLUGINS_CONFIG_ENV,
configured_legacy_relay_env_vars,
)
logger = logging.getLogger(__name__)
@@ -24,6 +32,7 @@ LOGICAL_LLM_SCOPE = "hermes.logical_llm_call"
RUNTIME_SCHEMA_KEY = "hermes.relay.schema_version"
RUNTIME_SCHEMA_VERSION = "hermes.relay.runtime.v1"
RUNTIME_INSTANCE_KEY = "hermes.relay.runtime_instance"
RELAY_PLUGINS_EXECUTION_CONSUMER = "hermes.nemo_relay.plugins"
_PROFILE_KEY_CACHE: dict[str, str] = {}
# Bound for native scope lifecycle operations (push/pop/flush) that gate
@@ -128,6 +137,20 @@ def pop_relay_scope(
return pop(handle, **kwargs)
class _RelayPluginConfigurationState(Enum):
"""Process-wide result shared by every currently hosted profile."""
UNINITIALIZED = auto()
DISABLED = auto()
ACTIVE = auto()
FOREIGN = auto()
FAILED = auto()
class _RelayPluginConfigurationLoadError(RuntimeError):
"""An explicitly selected Relay plugin configuration could not be loaded."""
@dataclass
class RelaySession:
"""One isolated Relay scope stack owned by a Hermes session."""
@@ -199,8 +222,232 @@ def _reset_segments_config_for_tests() -> None:
_SEGMENTS_CONFIG = None
class RelayOperationLease:
"""Keep process-wide Relay plugins alive across a deferred operation."""
def __init__(self, runtime: "RelayRuntime") -> None:
self._lock = threading.Lock()
self._runtime: RelayRuntime | None = runtime
def run_in_session(
self,
session: RelaySession,
callback: Callable[..., Any],
*args: Any,
**kwargs: Any,
) -> Any:
"""Run cleanup while this lease still owns the runtime lifetime."""
with self._lock:
runtime = self._runtime
if runtime is None:
raise RuntimeError("Hermes Relay operation lease is released")
return runtime._run_in_session_untracked(
session,
callback,
*args,
**kwargs,
)
def release(self) -> None:
"""Release this lease exactly once."""
with self._lock:
runtime = self._runtime
self._runtime = None
if runtime is not None:
runtime._end_operation()
class _ProcessRelayPluginConfiguration:
"""Own one Relay plugin configuration across profile-scoped hosts."""
def __init__(self) -> None:
self._lock = threading.RLock()
self._owners: set[int] = set()
self._state = _RelayPluginConfigurationState.UNINITIALIZED
self._active = False
self._relay: Any = None
self._activation: Any = None
def acquire(
self,
owner: Any,
relay: Any,
) -> _RelayPluginConfigurationState:
"""Join the process configuration, initializing it for the first host."""
owner_id = id(owner)
with self._lock:
if owner_id in self._owners:
return self._state
if self._owners:
self._owners.add(owner_id)
return self._state
if self._active and not self._clear_active():
logger.warning(
"Hermes Relay plugin cleanup is still pending; refusing to "
"replace the process-global configuration"
)
return self._remember(
owner_id,
_RelayPluginConfigurationState.FAILED,
)
try:
existing_report = relay.plugin.report()
except Exception:
logger.warning(
"Hermes could not determine whether a process-global Relay "
"plugin configuration is already active; refusing to replace it",
exc_info=True,
)
return self._remember(
owner_id,
_RelayPluginConfigurationState.FAILED,
)
if existing_report is not None:
logger.warning(
"A process-global Relay plugin configuration is already active "
"outside Hermes native ownership; leaving it unchanged and "
"disabling Hermes-managed Relay middleware for this process"
)
return self._remember(
owner_id,
_RelayPluginConfigurationState.FOREIGN,
)
try:
configured_inputs = _configured_plugin_inputs(relay)
if configured_inputs is None:
return self._remember(
owner_id,
_RelayPluginConfigurationState.DISABLED,
)
plugin_config, dynamic_plugins = configured_inputs
if dynamic_plugins:
try:
activation = _resolve_plugin_awaitable(
relay.plugin.initialize_with_dynamic_plugins(
plugin_config,
dynamic_plugins,
)
)
if activation is None:
raise RuntimeError(
"NeMo Relay dynamic plugin initialization "
"returned no activation handle"
)
self._activation = activation
except Exception as exc:
raise RuntimeError(
"Hermes Relay dynamic plugin activation failed"
) from exc
if self._activation is None:
# Hermes only enters Relay's initialization path after an
# explicit opt-in. Relay currently owns any subsequent ambient
# layering; a future discovery=False API can make this exact.
_resolve_plugin_awaitable(relay.plugin.initialize(plugin_config))
except Exception as exc:
self._activation = None
logger.warning(
"Hermes Relay plugin initialization failed: %s",
exc,
exc_info=True,
)
return self._remember(
owner_id,
_RelayPluginConfigurationState.FAILED,
)
self._active = True
self._relay = relay
state = self._remember(
owner_id,
_RelayPluginConfigurationState.ACTIVE,
)
logger.info(
"Relay plugins are active process-wide and apply to all profiles "
"hosted by this Hermes process."
)
return state
def _remember(
self,
owner_id: int,
state: _RelayPluginConfigurationState,
) -> _RelayPluginConfigurationState:
"""Retain one process decision for all concurrently hosted profiles."""
self._owners.add(owner_id)
self._state = state
return state
def release(self, owner: Any) -> None:
"""Release one host and clear Relay after the final host exits."""
owner_id = id(owner)
with self._lock:
if owner_id not in self._owners:
return
self._owners.remove(owner_id)
if self._owners:
return
if self._clear_active():
self._state = _RelayPluginConfigurationState.UNINITIALIZED
def reset_for_tests(self) -> None:
"""Clear process-global state left by directly constructed test hosts."""
with self._lock:
self._owners.clear()
if self._clear_active():
self._state = _RelayPluginConfigurationState.UNINITIALIZED
def retry_pending_cleanup(self) -> None:
"""Retry a failed final cleanup without disrupting live owners."""
with self._lock:
if not self._owners:
if self._clear_active():
self._state = _RelayPluginConfigurationState.UNINITIALIZED
def _clear_active(self) -> bool:
relay = self._relay
activation = self._activation
active = self._active
if not active or relay is None:
return True
try:
_flush_relay_subscribers(relay)
except Exception:
logger.warning(
"Hermes Relay plugin subscriber flush failed",
exc_info=True,
)
return False
try:
if activation is not None:
close = getattr(activation, "close", None)
if not callable(close):
raise RuntimeError(
"NeMo Relay dynamic plugin activation has no close method"
)
_resolve_plugin_awaitable(close())
else:
_clear_relay_plugins(relay)
except Exception:
logger.warning(
"Hermes Relay plugin configuration cleanup failed",
exc_info=True,
)
return False
self._active = False
self._relay = None
self._activation = None
return True
_PLUGIN_CONFIGURATION = _ProcessRelayPluginConfiguration()
atexit.register(_PLUGIN_CONFIGURATION.retry_pending_cleanup)
class RelayRuntime:
"""Own Relay session scopes independently of any exporter or plugin."""
"""Own Relay session scopes and optional process plugin configuration."""
def __init__(self, relay: Any = None, *, profile_key: str | None = None) -> None:
self.relay = relay or _load_nemo_relay()
@@ -210,8 +457,24 @@ class RelayRuntime:
self._sessions: dict[str, RelaySession] = {}
self._subagent_parents: dict[str, str] = {}
self._subagent_parent_handles: dict[str, Any] = {}
self._closing = False
self._shutdown_started = False
self._shutdown_complete = threading.Event()
self._operations_idle = threading.Event()
self._operations_idle.set()
self._active_operations = 0
self._execution_consumers_lock = threading.RLock()
self._execution_consumers: set[str] = set()
self._plugin_configuration_state = _PLUGIN_CONFIGURATION.acquire(
self,
self.relay,
)
self._plugin_configuration_registered = True
if (
self._plugin_configuration_state
is _RelayPluginConfigurationState.ACTIVE
):
self.retain_managed_execution(RELAY_PLUGINS_EXECUTION_CONSUMER)
self._shutdown_registered = True
atexit.register(self.shutdown)
@@ -244,6 +507,8 @@ class RelayRuntime:
if not session_id:
return None
with self._sessions_lock:
if self._closing:
return None
session = self._sessions.get(session_id)
if session is None:
parent_session_id = self._subagent_parents.get(session_id, "")
@@ -404,6 +669,8 @@ class RelayRuntime:
):
parent_handle = turn.handle
with self._sessions_lock:
if self._closing:
return None
self._subagent_parents[child_session_id] = parent_session_id
if parent_handle is not None:
self._subagent_parent_handles[child_session_id] = parent_handle
@@ -425,6 +692,8 @@ class RelayRuntime:
def get_session(self, session_id: str) -> RelaySession | None:
"""Return an active Hermes Relay session without creating one."""
with self._sessions_lock:
if self._closing:
return None
session = self._sessions.get(str(session_id or ""))
if session is None:
return None
@@ -458,6 +727,29 @@ class RelayRuntime:
span, never the agent. The abandoned daemon worker cannot block
process exit (tools.daemon_pool contract).
"""
self._begin_operation()
try:
return self._run_in_session_untracked(
session,
callback,
*args,
allow_closing=allow_closing,
timeout=timeout,
**kwargs,
)
finally:
self._end_operation()
def _run_in_session_untracked(
self,
session: RelaySession,
callback: Callable[..., Any],
*args: Any,
allow_closing: bool = False,
timeout: float | None = None,
**kwargs: Any,
) -> Any:
"""Run inside a session whose host-level lifetime is already held."""
with session.lock:
if session.closing and not allow_closing:
raise RuntimeError("Hermes Relay session is closing")
@@ -506,6 +798,8 @@ class RelayRuntime:
**kwargs: Any,
) -> Any:
"""Create and await an operation inside the session's saved context."""
self._begin_operation()
try:
with session.lock:
if session.closing and not allow_closing:
raise RuntimeError("Hermes Relay session is closing")
@@ -526,6 +820,27 @@ class RelayRuntime:
task = context.run(asyncio.create_task, invoke())
return await task
finally:
self._end_operation()
def _begin_operation(self) -> None:
"""Admit one Relay call while keeping process plugins alive."""
with self._sessions_lock:
if self._closing:
raise RuntimeError("Hermes Relay runtime is shutting down")
self._active_operations += 1
self._operations_idle.clear()
def _end_operation(self) -> None:
with self._sessions_lock:
self._active_operations -= 1
if self._active_operations == 0:
self._operations_idle.set()
def acquire_operation_lease(self) -> RelayOperationLease:
"""Retain plugin lifetime for work that outlives one Relay await."""
self._begin_operation()
return RelayOperationLease(self)
def emit_mark(
self,
@@ -586,6 +901,7 @@ class RelayRuntime:
allow_closing: bool = False,
failure_label: str = "scope close failed",
drain_limit: int = 32,
operation_already_held: bool = False,
) -> str | None:
"""Pop ``handle``, draining orphaned children in the same session context.
@@ -702,7 +1018,12 @@ class RelayRuntime:
error_holder["retry"] = retry_exc
try:
self.run_in_session(
run_in_session = (
self._run_in_session_untracked
if operation_already_held
else self.run_in_session
)
run_in_session(
session,
close_with_drain,
allow_closing=allow_closing,
@@ -721,6 +1042,17 @@ class RelayRuntime:
def close_session(self, event: dict[str, Any]) -> None:
"""Close one session scope and remove it from the core registry."""
try:
self._begin_operation()
except RuntimeError:
return
try:
self._close_session(event)
finally:
self._end_operation()
def _close_session(self, event: dict[str, Any]) -> None:
"""Close one session already admitted by the host lifecycle gate."""
session_id = _session_id(event)
with self._sessions_lock:
session = self._sessions.get(session_id)
@@ -741,23 +1073,13 @@ class RelayRuntime:
output={},
allow_closing=True,
failure_label="session scope close failed",
operation_already_held=True,
)
if failure:
failures.append(failure)
try:
try:
_scope_op_executor().submit(
self.relay.subscribers.flush
).result(timeout=_SCOPE_OP_TIMEOUT)
except RuntimeError:
# Interpreter shutdown: executor refuses new futures; flush
# on a bounded exit thread so a wedged pipeline cannot
# block process exit.
_run_bounded_on_exit_thread(
self.relay.subscribers.flush, _SCOPE_OP_TIMEOUT
)
except Exception as exc:
failures.append(f"subscriber flush failed: {exc}")
# Subscriber flushing is process-wide and may wait for publications
# owned by other sessions. Final plugin teardown flushes once after all
# tracked operations drain; doing it here can deadlock an asyncio loop.
with self._sessions_lock:
if self._sessions.get(session_id) is session:
self._sessions.pop(session_id, None)
@@ -771,17 +1093,64 @@ class RelayRuntime:
)
def shutdown(self) -> None:
"""Close all core-owned Relay session scopes."""
"""Close core scopes and release process plugin configuration."""
with self._sessions_lock:
if self._shutdown_started:
return
self._shutdown_started = True
self._closing = True
has_active_operations = self._active_operations > 0
if has_active_operations:
thread = threading.Thread(
target=self._finish_shutdown_after_operations,
name=f"hermes-nemo-relay-shutdown-{self.runtime_id[:8]}",
daemon=True,
)
try:
thread.start()
except Exception:
with self._sessions_lock:
self._shutdown_started = False
logger.warning(
"Hermes Relay deferred shutdown could not start",
exc_info=True,
)
return
self._finish_shutdown()
def _finish_shutdown_after_operations(self) -> None:
self._operations_idle.wait()
self._finish_shutdown()
def _finish_shutdown(self) -> None:
try:
with self._sessions_lock:
session_ids = list(self._sessions)
for session_id in session_ids:
self._safe(self.close_session, {"session_id": session_id})
self._safe(self._close_session, {"session_id": session_id})
if self._plugin_configuration_registered:
if (
self._plugin_configuration_state
is _RelayPluginConfigurationState.ACTIVE
):
self.release_managed_execution(
RELAY_PLUGINS_EXECUTION_CONSUMER
)
_PLUGIN_CONFIGURATION.release(self)
self._plugin_configuration_registered = False
if self._shutdown_registered:
try:
atexit.unregister(self.shutdown)
except Exception:
pass
self._shutdown_registered = False
except Exception:
with self._sessions_lock:
self._shutdown_started = False
logger.warning("Hermes Relay shutdown failed", exc_info=True)
return
with self._sessions_lock:
self._shutdown_complete.set()
@staticmethod
def _safe(callback: Callable[..., Any], *args: Any, **kwargs: Any) -> Any:
@@ -1611,6 +1980,87 @@ def _load_nemo_relay() -> Any:
return importlib.import_module("nemo_relay")
def _configured_plugin_inputs(
relay: Any,
) -> tuple[dict[str, Any], list[Any]] | None:
"""Load selected plugin inputs, or return ``None`` when none were selected."""
configured = os.environ.get(RELAY_PLUGINS_CONFIG_ENV, "").strip()
if not configured:
legacy_vars = configured_legacy_relay_env_vars(os.environ)
if legacy_vars:
logger.warning(
"Legacy NeMo Relay exporter variables are set but no %s was "
"provided. %s no longer activate Relay exporters; migrate the "
"exporter configuration to a Relay plugins.toml file.",
RELAY_PLUGINS_CONFIG_ENV,
", ".join(legacy_vars),
)
return None
config_path = Path(configured).expanduser()
try:
with config_path.open("rb") as config_file:
config = tomllib.load(config_file)
if "dynamic_plugins" in config:
raise ValueError(
"Hermes [[dynamic_plugins]] records are unsupported; use Relay "
"[[plugins.dynamic]] records"
)
dynamic_plugins: list[Any] = []
if "plugins" in config:
dynamic_plugins = relay.plugin.load_dynamic_plugin_activation_specs(
config_path
)
plugin_config = dict(config)
plugin_config.pop("plugins", None)
return plugin_config, dynamic_plugins
except Exception as exc:
raise _RelayPluginConfigurationLoadError(
"Hermes Relay plugin configuration could not be loaded from "
f"{config_path}; continuing without Relay plugins"
) from exc
def _flush_relay_subscribers(relay: Any) -> None:
"""Flush Relay without blocking an asyncio event-loop thread."""
_resolve_plugin_awaitable(relay.subscribers.flush_async())
def _clear_relay_plugins(relay: Any) -> None:
"""Clear Relay plugins without blocking an asyncio event-loop thread."""
_resolve_plugin_awaitable(relay.plugin.clear_async())
def _resolve_plugin_awaitable(value: Any) -> Any:
"""Resolve Relay's async plugin API from synchronous host construction."""
if not inspect.isawaitable(value):
return value
try:
asyncio.get_running_loop()
except RuntimeError:
return asyncio.run(value)
result: dict[str, Any] = {}
error: dict[str, BaseException] = {}
def _runner() -> None:
try:
result["value"] = asyncio.run(value)
except BaseException as exc: # pragma: no cover - re-raised below
error["exc"] = exc
thread = threading.Thread(
target=_runner,
name="hermes-nemo-relay-plugin-lifecycle",
daemon=True,
)
thread.start()
thread.join()
if "exc" in error:
raise error["exc"]
return result.get("value")
def _session_id(event: dict[str, Any]) -> str:
return str(event.get("session_id") or "")
@@ -1619,4 +2069,5 @@ def _reset_for_tests() -> None:
"""Reset all profile-scoped Relay hosts for isolated tests."""
SESSION_COORDINATOR._reset_active_turns_for_tests()
HOST_REGISTRY.shutdown_all()
_PLUGIN_CONFIGURATION.reset_for_tests()
_PROFILE_KEY_CACHE.clear()
+45 -14
View File
@@ -8,6 +8,7 @@ import json
import logging
import os
import re
import threading
from pathlib import Path
from typing import Any, Dict, Optional
@@ -24,6 +25,10 @@ logger = logging.getLogger(__name__)
_skill_commands: Dict[str, Dict[str, Any]] = {}
_skill_commands_platform: Optional[str] = None
_skill_commands_home: Optional[str] = None
# Guards the (map, platform-tag, home-tag) triple so publication and the
# freshness lookup always see a consistent snapshot. Scanning itself stays
# outside this lock.
_publish_lock = threading.Lock()
# Patterns for sanitizing skill names into clean hyphen-separated slugs.
_SKILL_INVALID_CHARS = re.compile(r"[^a-z0-9-]")
_SKILL_MULTI_HYPHEN = re.compile(r"-{2,}")
@@ -423,9 +428,15 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
Dict mapping "/skill-name" to {name, description, skill_md_path, skill_dir}.
"""
global _skill_commands, _skill_commands_platform, _skill_commands_home
_skill_commands_platform = _resolve_skill_commands_platform()
_skill_commands_home = _resolve_skill_commands_home()
_skill_commands = {}
platform = _resolve_skill_commands_platform()
home = _resolve_skill_commands_home()
# Build into a local map and publish once, at the end. Writing straight
# into the global made a scan's partial results visible to everything
# else in the process: a second, overlapping scan deduped against its own
# (empty) ``seen_names`` but collided against the first scan's already-
# published slugs, logging one bogus "already claimed" warning per skill —
# each naming the same skill as its own incumbent (#74574).
commands: Dict[str, Dict[str, Any]] = {}
try:
from tools.skills_tool import SKILLS_DIR, _parse_frontmatter, skill_matches_platform, skill_matches_environment, _get_disabled_skill_names
from agent.skill_utils import (
@@ -505,14 +516,14 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
# slug (e.g. "git_helper" vs "git-helper"). First-wins
# preserves local-before-external precedence.
cmd_key = f"/{cmd_name}"
if cmd_key in _skill_commands:
if cmd_key in commands:
logger.warning(
"Skill %r maps to slash command %s already claimed "
"by %r; keeping the first and skipping this one.",
name, cmd_key, _skill_commands[cmd_key]["name"],
name, cmd_key, commands[cmd_key]["name"],
)
continue
_skill_commands[cmd_key] = {
commands[cmd_key] = {
"name": name,
"description": description or f"Invoke the {name} skill",
"skill_md_path": str(skill_md),
@@ -522,7 +533,18 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
continue
except Exception:
pass
return _skill_commands
# Publish the finished map and the platform/home it was scanned for as
# ONE step. Bare assignments are not atomic together: a reader landing
# between them sees the NEW map still carrying the OLD platform tag, and
# if that stale tag happens to match its own platform it accepts the map
# without rescanning — serving another platform's disabled-skill view,
# exactly the leak #14536 closed. Only the publish/lookup pair is locked;
# the scan above (file I/O, deferred imports) stays outside it.
with _publish_lock:
_skill_commands = commands
_skill_commands_platform = platform
_skill_commands_home = home
return commands
def get_skill_commands() -> Dict[str, Dict[str, Any]]:
@@ -534,13 +556,22 @@ def get_skill_commands() -> Dict[str, Dict[str, Any]]:
active profile's Hermes home changes (e.g. Desktop switching profiles
mid-session) so each profile sees its own ``skills.external_dirs`` (#88023).
"""
if (
not _skill_commands
or _skill_commands_platform != _resolve_skill_commands_platform()
or _skill_commands_home != _resolve_skill_commands_home()
):
scan_skill_commands()
return _skill_commands
current_platform = _resolve_skill_commands_platform()
current_home = _resolve_skill_commands_home()
# Read the map and its tags under the same lock that publishes them, so
# the freshness decision is made against a consistent snapshot.
with _publish_lock:
commands = _skill_commands
is_fresh = (
bool(commands)
and _skill_commands_platform == current_platform
and _skill_commands_home == current_home
)
if is_fresh:
return commands
# Scan outside the lock — it does file I/O and deferred imports, and
# concurrent scans are already safe (each builds its own map).
return scan_skill_commands()
def reload_skills() -> Dict[str, Any]:
+44 -6
View File
@@ -33,10 +33,12 @@ from typing import Any, Dict, List, Optional
from agent.prompt_builder import (
DEFAULT_AGENT_IDENTITY,
EXECUTION_GUIDANCE_MODELS,
GOOGLE_MODEL_OPERATIONAL_GUIDANCE,
HERMES_AGENT_HELP_GUIDANCE,
KANBAN_GUIDANCE,
MEMORY_GUIDANCE,
USER_PROFILE_GUIDANCE,
OPENAI_MODEL_EXECUTION_GUIDANCE,
PARALLEL_TOOL_CALL_GUIDANCE,
PLATFORM_HINTS,
@@ -414,8 +416,21 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
# Tool-aware behavioral guidance: only inject when the tools are loaded
tool_guidance = []
# MEMORY_GUIDANCE instructs the model to save facts to the built-in
# MEMORY.md/USER.md stores. With both disabled in config no store is built,
# so the guidance would steer the model at a tool whose every call returns
# "Memory is not available". Defaults to True for the rare code paths that
# build an agent view without going through agent_init.
# When only the user profile store is enabled, the narrower
# USER_PROFILE_GUIDANCE is injected instead — the full block instructs the
# model to write notes to a MEMORY.md store that does not exist.
_mem_enabled = getattr(agent, "_memory_enabled", True)
_profile_enabled = getattr(agent, "_user_profile_enabled", True)
if "memory" in agent.valid_tool_names:
if _mem_enabled:
tool_guidance.append(MEMORY_GUIDANCE)
elif _profile_enabled:
tool_guidance.append(USER_PROFILE_GUIDANCE)
if "session_search" in agent.valid_tool_names:
tool_guidance.append(SESSION_SEARCH_GUIDANCE)
if "skill_manage" in agent.valid_tool_names:
@@ -477,12 +492,35 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
# paths, parallel tool calls, verify-before-edit, etc.)
if "gemini" in _model_lower or "gemma" in _model_lower:
stable_parts.append(GOOGLE_MODEL_OPERATIONAL_GUIDANCE)
# OpenAI GPT/Codex execution discipline (tool persistence,
# prerequisite checks, verification, anti-hallucination).
# Also applied to xAI Grok — same failure modes (claims completion
# without tool calls, suggests workarounds instead of using
# existing tools, replies with plans instead of executing).
if "gpt" in _model_lower or "codex" in _model_lower or "grok" in _model_lower:
# Execution-discipline guidance (tool persistence, mandatory tool use
# for arithmetic, external-write read-back, count reconciliation,
# literal preservation, verification-gated completion). Historically
# nested inside the tool-use-enforcement branch and fenced to
# gpt/codex/grok; now an independent gate so DeepSeek/Kimi/Qwen-class
# models receive it even when tool_use_enforcement is off. Controlled
# by config.yaml agent.execution_guidance:
# "auto" (default) — matches EXECUTION_GUIDANCE_MODELS
# true — always inject (all models)
# false — never inject
# list — custom model-name substrings to match
# Resolved once at session start keyed on the (fixed) model name, so
# the system prompt stays byte-stable for the life of the conversation.
if agent.valid_tool_names:
_exec_guidance = getattr(agent, "_execution_guidance", "auto")
_exec_inject = False
if _exec_guidance is True or (isinstance(_exec_guidance, str) and _exec_guidance.lower() in {"true", "always", "yes", "on"}):
_exec_inject = True
elif _exec_guidance is False or (isinstance(_exec_guidance, str) and _exec_guidance.lower() in {"false", "never", "no", "off"}):
_exec_inject = False
elif isinstance(_exec_guidance, list):
model_lower = (agent.model or "").lower()
_exec_inject = any(p.lower() in model_lower for p in _exec_guidance if isinstance(p, str))
else:
# "auto" or any unrecognised value — use hardcoded defaults
model_lower = (agent.model or "").lower()
_exec_inject = any(p in model_lower for p in EXECUTION_GUIDANCE_MODELS)
if _exec_inject:
stable_parts.append(OPENAI_MODEL_EXECUTION_GUIDANCE)
has_skills_tools = any(name in agent.valid_tool_names for name in ['skills_list', 'skill_view', 'skill_manage'])
+71 -1
View File
@@ -557,7 +557,11 @@ def make_tool_result_message(
The outer list itself is rebuilt rather than returned by identity, so
callers should compare by value, not by ``is``.
"""
wrapped = _maybe_wrap_untrusted(name, content)
# Order matters: detect provider-side elision on the RAW content and
# append the notice first, THEN wrap — so the notice lives inside the
# untrusted block next to the data it describes, appended exactly once
# at construction time (cache-safe).
wrapped = _maybe_wrap_untrusted(name, _maybe_append_elision_notice(name, content))
message = stamp_message_timestamp({
"role": "tool",
"name": name,
@@ -608,6 +612,70 @@ def _is_untrusted_tool(name: Optional[str]) -> bool:
return any(name.startswith(p) for p in _UNTRUSTED_TOOL_PREFIXES)
# --- Upstream-elision detection --------------------------------------------
#
# Some MCP servers elide data SERVER-SIDE and mark the elision inside the
# payload itself (e.g. Composio: '...13 more items' inside a JSON array,
# '"has_more": true', 'Complete response was large (N tokens). Full data
# saved to sandbox in /mnt/files/...', 'data_preview' envelopes). Because the
# result looks structurally complete, models treat the visible slice as the
# whole dataset and falsely claim completeness. When one of these markers is
# present, we append ONE compact notice at result-construction time — before
# the message enters history, never mutated later, so prompt caching is safe.
# Conservative patterns only: each one is an explicit provider-side "there is
# more data than what you can see" signal, not a generic truncation heuristic.
_UPSTREAM_ELISION_PATTERNS = (
re.compile(r"\.\.\.\s*\d+\s+more\s+items?", re.IGNORECASE),
re.compile(r'"has_more"\s*:\s*true', re.IGNORECASE),
re.compile(r"saved to sandbox", re.IGNORECASE),
re.compile(r"data_preview", re.IGNORECASE),
)
# Results smaller than this can't meaningfully hide an elided enumeration —
# skip the scan entirely so tiny results pay nothing.
_ELISION_SCAN_MIN_CHARS = 1_000
# Bound the regex scan: markers appear near the elided structure, which for
# the payload sizes that matter (20-50K) is always inside the first 64KB.
_ELISION_SCAN_MAX_CHARS = 65_536
_UPSTREAM_ELISION_NOTICE = (
'\n[hermes note: this result contains provider-side elision markers '
'(e.g. "...N more items" / has_more:true). The data shown is INCOMPLETE '
'— page/fetch the remainder before treating any enumeration as complete.]'
)
def _detect_upstream_elision(content: Any) -> bool:
"""True when a string tool result carries provider-side elision markers.
Cheap and safe by construction: non-string content is never scanned,
results under ``_ELISION_SCAN_MIN_CHARS`` short-circuit, and the regex
scan is capped at the first ``_ELISION_SCAN_MAX_CHARS`` chars.
"""
if not isinstance(content, str):
return False
if len(content) < _ELISION_SCAN_MIN_CHARS:
return False
window = content[:_ELISION_SCAN_MAX_CHARS]
return any(p.search(window) for p in _UPSTREAM_ELISION_PATTERNS)
def _maybe_append_elision_notice(name: str, content: Any) -> Any:
"""Append the incompleteness notice to untrusted string results that
embed upstream elision markers. Returns ``content`` unchanged otherwise.
Runs on the RAW result before untrusted-wrapping so the notice sits with
the data it describes, and only at result-construction time (cache-safe).
"""
if not _is_untrusted_tool(name):
return content
if _detect_upstream_elision(content):
return content + _UPSTREAM_ELISION_NOTICE
return content
def _tool_output_risk_metadata(name: str, content: Any) -> Optional[Dict[str, Any]]:
"""Classify textual attacker-controlled output without retaining a copy.
@@ -729,5 +797,7 @@ __all__ = [
"_extract_landed_file_mutation_paths",
"_extract_error_preview",
"_trajectory_normalize_msg",
"_detect_upstream_elision",
"_maybe_append_elision_notice",
"make_tool_result_message",
]
+106 -1
View File
@@ -48,12 +48,31 @@ from tools.thread_context import propagate_context_to_thread
from tools.tool_result_storage import (
maybe_persist_tool_result,
enforce_turn_budget,
extract_persisted_path,
)
from tools.budget_config import BudgetConfig, DEFAULT_BUDGET, budget_for_context_window
logger = logging.getLogger(__name__)
def _record_persisted_path_for_stub(agent, tool_call_id: str, function_result) -> None:
"""Tell the stall guards where a persisted result's full content lives.
When a large result is spilled to disk (<persisted-output> preview), a
later result-reference stub pointing at that first occurrence must carry
the spillover file path so the reference can't dangle. Best-effort: never
lets bookkeeping break tool execution.
"""
try:
if not isinstance(function_result, str):
return
path = extract_persisted_path(function_result)
if path:
agent._tool_guardrails.record_persisted_result(tool_call_id, path)
except Exception as exc:
logger.debug("persisted-path record for result stub failed: %s", exc)
def _ensure_file_checkpoint(
agent,
function_name: str,
@@ -87,7 +106,10 @@ def _budget_for_agent(agent) -> BudgetConfig:
"""
try:
ctx = getattr(getattr(agent, "context_compressor", None), "context_length", None)
return budget_for_context_window(int(ctx)) if ctx else DEFAULT_BUDGET
# budget_for_context_window(None) (rather than DEFAULT_BUDGET) so the
# config-driven MCP threshold override still applies when the context
# length isn't resolvable.
return budget_for_context_window(int(ctx) if ctx else None)
except Exception:
return DEFAULT_BUDGET
@@ -1730,6 +1752,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
function_args,
function_result,
failed=is_error,
tool_call_id=getattr(tc, "id", "") or "",
)
if is_error:
@@ -1764,6 +1787,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
env=get_active_env(effective_task_id),
config=_tool_budget,
) if not _is_multimodal_tool_result(function_result) else function_result
_record_persisted_path_for_stub(agent, tc.id, function_result)
subdir_hints = agent._subdirectory_hints.check_tool_call(name, args)
if subdir_hints:
@@ -2120,6 +2144,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
question=next_args.get("question", ""),
choices=next_args.get("choices"),
multi_select=next_args.get("multi_select", False),
questions=next_args.get("questions"),
callback=agent.clarify_callback,
)
function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware(
@@ -2177,6 +2202,57 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('read_preview', function_args, tool_duration, result=function_result)}")
elif function_name == "drive_preview":
def _execute(next_args: dict) -> Any:
from tools.drive_preview_tool import drive_preview_tool as _drive_preview_tool
return _drive_preview_tool(
action=next_args.get("action", ""),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
text=next_args.get("text"),
key=next_args.get("key"),
submit=next_args.get("submit"),
amount=next_args.get("amount"),
to=next_args.get("to"),
limit=next_args.get("max"),
callback=getattr(agent, "drive_preview_callback", None),
)
function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware(
agent,
function_name=function_name,
function_args=function_args,
effective_task_id=effective_task_id,
tool_call_id=getattr(tool_call, "id", "") or "",
execute=_execute,
scope_block=_ts_scope_block,
display_index=i,
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('drive_preview', function_args, tool_duration, result=function_result)}")
elif function_name == "annotate_preview":
def _execute(next_args: dict) -> Any:
from tools.annotate_preview_tool import annotate_preview_tool as _annotate_preview_tool
return _annotate_preview_tool(
action=next_args.get("action", "add"),
ref=next_args.get("ref"),
selector=next_args.get("selector"),
label=next_args.get("label"),
callback=getattr(agent, "drive_preview_callback", None),
)
function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware(
agent,
function_name=function_name,
function_args=function_args,
effective_task_id=effective_task_id,
tool_call_id=getattr(tool_call, "id", "") or "",
execute=_execute,
scope_block=_ts_scope_block,
display_index=i,
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('annotate_preview', function_args, tool_duration, result=function_result)}")
elif function_name == "read_window_below":
def _execute(next_args: dict) -> Any:
from tools.read_window_tool import read_window_below_tool as _read_window_below_tool
@@ -2196,6 +2272,33 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('read_window_below', function_args, tool_duration, result=function_result)}")
elif function_name == "tour":
def _execute(next_args: dict) -> Any:
from tools.tour_tool import tour_tool as _tour_tool
return _tour_tool(
action=next_args.get("action", ""),
surface=next_args.get("surface"),
selector=next_args.get("selector"),
title=next_args.get("title"),
text=next_args.get("text"),
side=next_args.get("side"),
steps=next_args.get("steps"),
step_index=next_args.get("step_index"),
callback=getattr(agent, "tour_callback", None),
)
function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware(
agent,
function_name=function_name,
function_args=function_args,
effective_task_id=effective_task_id,
tool_call_id=getattr(tool_call, "id", "") or "",
execute=_execute,
scope_block=_ts_scope_block,
display_index=i,
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('tour', function_args, tool_duration, result=function_result)}")
elif function_name == "setup_mcp":
def _execute(next_args: dict) -> Any:
from tools.setup_mcp_tool import setup_mcp_tool as _setup_mcp_tool
@@ -2541,6 +2644,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
function_args,
function_result,
failed=_is_error_result,
tool_call_id=getattr(tool_call, "id", "") or "",
)
result_preview = function_result if agent.verbose_logging else (
function_result[:200] if len(function_result) > 200 else function_result
@@ -2579,6 +2683,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
env=get_active_env(effective_task_id),
config=_tool_budget,
) if not _is_multimodal_tool_result(function_result) else function_result
_record_persisted_path_for_stub(agent, tool_call.id, function_result)
# Discover subdirectory context files from tool arguments
subdir_hints = agent._subdirectory_hints.check_tool_call(function_name, function_args)
+213
View File
@@ -59,6 +59,52 @@ MUTATING_TOOL_NAMES = frozenset(
}
)
# Tools that are legitimately re-invoked with identical arguments and may
# legitimately return an unchanged result while waiting on external progress —
# background-process management and job pollers. The identical-call loop
# notice (agent.stall_guards) never fires for these, so polling patterns like
# ``process(action="poll")`` or repeatedly checking a generation job stay
# unannotated.
STALL_GUARD_REPEATABLE_TOOLS = frozenset(
{
"process",
"bfl_flux3_get_result",
}
)
# Poller naming conventions (e.g. ``<vendor>_get_result``) used by generated /
# MCP tool surfaces. Matched as suffixes so vendor-prefixed pollers are exempt
# without enumerating every vendor.
_STALL_GUARD_REPEATABLE_SUFFIXES = (
"_get_result",
"_poll",
)
# The notice fires on the Nth consecutive identical call (same tool, same
# canonical args, same result). 3 tolerates one legitimate double-check while
# catching the observed re-issue loops (3x/4x identical calls in eval traces).
STALL_GUARD_IDENTICAL_CALL_THRESHOLD = 3
# Result-reference stubbing (agent.stall_guards): from the 2nd consecutive
# identical call whose FRESH result is byte-identical to the previous one,
# the duplicate payload is replaced in context by a short reference stub.
# Results under this size aren't worth stubbing (the stub itself plus the
# lost locality outweigh the savings), and error results are never stubbed
# (the model must see every fresh error verbatim).
IDENTICAL_RESULT_STUB_MIN_CHARS = 512
# How much of the canonical args JSON the stub carries so the model still
# knows WHAT the referenced call was even if context compression later
# evicts the referenced result (cheap dangling-reference mitigation).
_RESULT_STUB_ARGS_PREVIEW_CHARS = 120
def is_stall_guard_repeatable(tool_name: str) -> bool:
"""Whether a tool is exempt from the identical-call loop notice."""
if tool_name in STALL_GUARD_REPEATABLE_TOOLS:
return True
return tool_name.endswith(_STALL_GUARD_REPEATABLE_SUFFIXES)
@dataclass(frozen=True)
class ToolCallGuardrailConfig:
@@ -173,6 +219,21 @@ class LoopCapConfig:
)
@dataclass(frozen=True)
class IdenticalCallObservation:
"""Outcome of observing one completed tool call for the stall guards.
``notice`` is the identical-call loop-breaker notice (appended after the
result). ``stub`` is the result-reference replacement for a byte-identical
duplicate result (replaces the result content). Both may be set on the
same call (3rd+ identical call): the stub replaces the payload and the
notice is appended after it.
"""
notice: str | None = None
stub: str | None = None
@dataclass(frozen=True)
class ToolCallSignature:
"""Stable, non-reversible identity for a tool name plus canonical args."""
@@ -282,6 +343,26 @@ class ToolCallGuardrailController:
self._same_tool_failure_counts: dict[str, int] = {}
self._no_progress: dict[ToolCallSignature, tuple[str, int]] = {}
self._halt_decision: ToolGuardrailDecision | None = None
# Identical-call loop-breaker state (agent.stall_guards): tracks the
# CONSECUTIVE streak of identical (tool, canonical args) calls whose
# results were also identical. Any different call — or a different
# result — resets the streak, so legitimate re-reads after edits and
# varied polling are never flagged. Per-turn, like everything else here.
# NOTE: open PR #85352 (patrykkopycinski) tracks no-progress loops
# ACROSS turns via a detection window — a different mechanism from
# this per-turn consecutive streak. Coordinate future work there.
self._identical_streak_sig: ToolCallSignature | None = None
self._identical_streak_result_hash: str = ""
self._identical_streak_count: int = 0
# tool_call_id of the FIRST call in the current streak, so a
# result-reference stub can point at the message that carries the
# full payload.
self._identical_streak_first_call_id: str = ""
# tool_call_id -> spillover file path for results that were persisted
# out of context (persisted-output preview). Lets a reference stub
# carry the file path so the reference can't dangle when the first
# occurrence entered context as a preview.
self._persisted_result_paths: dict[str, str] = {}
# Per-turn runaway-loop cap counters. Reset every turn (this method
# runs at the start of each run_conversation), so the caps bound a
# single agent loop rather than accumulating across the session.
@@ -444,6 +525,138 @@ class ToolCallGuardrailController:
return False
return tool_name in self.config.idempotent_tools
def observe_identical_call(
self,
tool_name: str,
args: Mapping[str, Any] | None,
result: str | None,
) -> str | None:
"""Track consecutive identical calls; return a loop-breaker notice or None.
Back-compat wrapper around :meth:`observe_call` for callers that only
care about the loop-breaker notice.
"""
return self.observe_call(tool_name, args, result).notice
def observe_call(
self,
tool_name: str,
args: Mapping[str, Any] | None,
result: str | None,
*,
tool_call_id: str = "",
failed: bool = False,
) -> "IdenticalCallObservation":
"""Track consecutive identical calls; return notice + dedupe stub info.
Two independent outputs from the same consecutive-streak tracker:
- ``notice``: the compact loop-breaker notice, fired when the SAME
tool is called with identical canonical arguments AND returns an
identical result for the ``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th
(and every subsequent) consecutive time within the turn. Purely
observational — never blocks the call. Allowlisted pollers
(``is_stall_guard_repeatable``) are exempt from the NOTICE.
- ``stub``: a short reference replacement for the CURRENT result,
produced from the 2nd consecutive identical call whose fresh result
is byte-identical to the previous one. The tool still executed —
only the context representation is deduplicated, so polling
semantics are preserved (a changed result flows through whole and
resets the streak). Pollers are NOT exempt from stubbing: for a
poller, an identical result means nothing changed, which is exactly
when the stub saves the most context and loses nothing. Results
under ``IDENTICAL_RESULT_STUB_MIN_CHARS`` and failed/error results
are never stubbed, and only plain-string results are considered.
Any intervening different call or changed result resets the streak.
Callers substitute/append at tool RESULT construction time, which is
cache-safe: tool results are append-only and never mutate
already-sent context.
"""
is_plain_str = isinstance(result, str)
signature = ToolCallSignature.from_call(tool_name, _coerce_args(args))
result_hash = _result_hash(result) if is_plain_str else ""
if (
is_plain_str
and self._identical_streak_sig == signature
and self._identical_streak_result_hash == result_hash
):
self._identical_streak_count += 1
else:
# New streak (or non-string result, which never forms a streak —
# multimodal content lists pass through untouched).
self._identical_streak_sig = signature if is_plain_str else None
self._identical_streak_result_hash = result_hash
self._identical_streak_count = 1 if is_plain_str else 0
self._identical_streak_first_call_id = tool_call_id or ""
count = self._identical_streak_count
notice = None
if (
not is_stall_guard_repeatable(tool_name)
and count >= STALL_GUARD_IDENTICAL_CALL_THRESHOLD
):
ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}"
notice = (
f"[hermes note: this is the {ordinal} consecutive identical call to "
f"{tool_name} with identical arguments returning the same result. "
"Do not repeat it — change arguments, use a different tool, or "
"proceed with what you have.]"
)
stub = None
if (
is_plain_str
and count >= 2
and not failed
and len(result) >= IDENTICAL_RESULT_STUB_MIN_CHARS
):
stub = self._build_result_reference_stub(tool_name, args)
return IdenticalCallObservation(notice=notice, stub=stub)
def record_persisted_result(self, tool_call_id: str, file_path: str) -> None:
"""Remember the spillover path a persisted result was saved to.
When the first occurrence of a result entered context as a
persisted-output preview, a later reference stub must carry the
spillover file path so the reference can't dangle.
"""
if tool_call_id and file_path:
self._persisted_result_paths[tool_call_id] = file_path
def _build_result_reference_stub(
self, tool_name: str, args: Mapping[str, Any] | None
) -> str:
"""Build the reference stub replacing a byte-identical duplicate result.
Carries the tool name + a canonical-args preview so that even if
context compression later evicts the referenced result, the model
still knows WHAT the call was (cheap dangling-reference mitigation).
"""
try:
args_preview = canonical_tool_args(_coerce_args(args))
except TypeError:
args_preview = "{}"
if len(args_preview) > _RESULT_STUB_ARGS_PREVIEW_CHARS:
args_preview = args_preview[:_RESULT_STUB_ARGS_PREVIEW_CHARS] + "…"
first_id = self._identical_streak_first_call_id
ref = f" (tool_call_id {first_id})" if first_id else ""
stub = (
f"[hermes note: this result is byte-identical to the {tool_name} "
f"result earlier this turn{ref}. Refer to that result; it has not "
f"changed. Args: {args_preview}]"
)
spill_path = self._persisted_result_paths.get(first_id) if first_id else None
if spill_path:
stub += (
f"\n[The referenced result was persisted to: {spill_path} — "
"page through it with read_file if you need the full content.]"
)
return stub
def _check_loop_cap(
self,
tool_name: str,
+48 -16
View File
@@ -13,6 +13,15 @@ import json
from typing import Any, Dict
from agent.lmstudio_reasoning import resolve_lmstudio_effort
from agent.reasoning_effort import (
KIMI_K3_EFFORTS,
KIMI_K3_OVERRIDES,
OPENAI_COMPAT_WIRE_EFFORTS,
TOKENHUB_EFFORTS,
clamp_effort,
kimi_supported_efforts,
requested_effort,
)
from agent.moonshot_schema import is_moonshot_model, sanitize_moonshot_tools
from agent.prompt_builder import DEVELOPER_ROLE_MODELS
from agent.transports.base import ProviderTransport
@@ -84,15 +93,25 @@ def _add_prompt_cache_key(
def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None:
"""Return the model's wire-compatible reasoning config."""
"""Return the model's wire-compatible reasoning config.
Hermes' internal effort set extends the wire vocabulary with ``ultra``
(the /reasoning command documents none..xhigh|max|ultra). OpenAI-
compatible wires — OpenRouter chief among them — accept exactly
max|xhigh|high|medium|low|minimal|none and reject the extension with
HTTP 400 (#89503). Clamp against the declared wire vocabulary via the
shared policy in ``agent.reasoning_effort``; provider profiles with
narrower sets clamp again downstream.
"""
if not isinstance(reasoning_config, dict):
return reasoning_config
if (
"gpt-5.6" in (model or "").lower()
and str(reasoning_config.get("effort") or "").strip().lower() == "ultra"
):
effort = str(reasoning_config.get("effort") or "").strip().lower()
if not effort:
return reasoning_config
clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS)
if clamped != effort:
normalized = dict(reasoning_config)
normalized["effort"] = "max"
normalized["effort"] = clamped
return normalized
return reasoning_config
@@ -559,11 +578,22 @@ class ChatCompletionsTransport(ProviderTransport):
and reasoning_config.get("enabled") is False
)
if not _kimi_thinking_off:
_kimi_effort = "medium"
if reasoning_config and isinstance(reasoning_config, dict):
_e = (reasoning_config.get("effort") or "").strip().lower()
if _e in {"low", "medium", "high"}:
_kimi_effort = _e
# Kimi vocabularies are declared in agent.reasoning_effort:
# K3 = low/high/max (with the vendor-documented medium→high,
# xhigh→max rounding), K2-era = low/medium/high. Default when
# no effort was requested: K3's server default is high,
# K2-era's is medium.
_supported = kimi_supported_efforts(model)
_overrides = (
KIMI_K3_OVERRIDES if _supported is KIMI_K3_EFFORTS else None
)
_e = requested_effort(reasoning_config)
if _e is None:
_kimi_effort = (
"high" if _supported is KIMI_K3_EFFORTS else "medium"
)
else:
_kimi_effort = clamp_effort(_e, _supported, _overrides)
api_kwargs["reasoning_effort"] = _kimi_effort
# Tencent TokenHub: top-level reasoning_effort (unless thinking disabled)
@@ -574,11 +604,13 @@ class ChatCompletionsTransport(ProviderTransport):
and reasoning_config.get("enabled") is False
)
if not _tokenhub_thinking_off:
_tokenhub_effort = "high"
if reasoning_config and isinstance(reasoning_config, dict):
_e = (reasoning_config.get("effort") or "").strip().lower()
if _e in {"low", "medium", "high"}:
_tokenhub_effort = _e
# TokenHub accepts low/medium/high (declared in
# agent.reasoning_effort); default high when no effort was
# requested.
_e = requested_effort(reasoning_config)
_tokenhub_effort = (
"high" if _e is None else clamp_effort(_e, TOKENHUB_EFFORTS)
)
api_kwargs["reasoning_effort"] = _tokenhub_effort
# LM Studio: top-level reasoning_effort. Only emit when the model
+28 -20
View File
@@ -27,6 +27,13 @@ def _cache_scope_from_session_id(session_id: Optional[str]) -> str:
match = _CRON_SESSION_ID_RE.match(sid)
return match.group(1) if match else sid
from agent.reasoning_effort import (
ACTUAL_RELAY_EFFORTS,
XAI_GROK46_EFFORTS,
XAI_LEGACY_EFFORTS,
clamp_effort,
codex_supported_efforts,
)
from agent.transports.base import ProviderTransport
from agent.transports.types import NormalizedResponse, ToolCall
@@ -432,30 +439,31 @@ class ResponsesApiTransport(ProviderTransport):
elif reasoning_config.get("effort"):
reasoning_effort = reasoning_config["effort"]
_effort_clamp = {"minimal": "low"}
if "gpt-5.6" in (model or "").lower():
# Ultra is the Codex product tier; the Responses API wire value is max.
_effort_clamp["ultra"] = "max"
# Wire vocabularies are declared in agent.reasoning_effort; the shared
# clamp policy (nearest weaker supported level, never escalate,
# never invert the ladder) replaces the per-backend hand maps that
# repeatedly leaked internal levels like "ultra" to the wire
# (#89503 class) or clamped one rung below a model's real ceiling
# (#87279).
if params.get("is_xai_responses", False):
from agent.model_metadata import is_grok_46_family
# Grok 4.6 accepts xhigh as a wire value; older Grok models top
# out at high. max/ultra are Hermes ladder aliases for "this
# model's ceiling", so they clamp to the strongest level the
# model actually accepts — xhigh on grok-4.6, high elsewhere —
# never one rung below it (#87279).
if is_grok_46_family(model):
_effort_clamp.update({"max": "xhigh", "ultra": "xhigh"})
# Grok 4.6 accepts xhigh as a wire value; older Grok tops out
# at high.
_supported = (
XAI_GROK46_EFFORTS if is_grok_46_family(model)
else XAI_LEGACY_EFFORTS
)
elif (params.get("provider") or "").strip().lower() == "actual":
# Actual Computer relays to SGLang/vLLM backends:
# none/low/medium/high/max.
_supported = ACTUAL_RELAY_EFFORTS
else:
_effort_clamp["xhigh"] = "high"
_effort_clamp.update({"max": "high", "ultra": "high"})
if (params.get("provider") or "").strip().lower() == "actual":
# Actual Computer relays to SGLang/vLLM backends that accept only
# none/low/medium/high/max for reasoning effort — a forwarded
# xhigh/ultra fails with a wrapped HTTP 400 ("Expecting value:
# line 1 column 1"). Clamp Hermes' wider set to the supported one.
_effort_clamp.update({"xhigh": "high", "ultra": "max"})
reasoning_effort = _effort_clamp.get(reasoning_effort, reasoning_effort)
# OpenAI/Codex Responses backend — per-model vocabulary
# (live-verified: "max" is gpt-5.6-only, "minimal" always
# rejected). #68365 premise confirmed.
_supported = codex_supported_efforts(model)
reasoning_effort = clamp_effort(reasoning_effort, _supported)
response_tools = _responses_tools(tools)
+60
View File
@@ -605,6 +605,15 @@ def build_turn_context(
# NOTE: _turns_since_memory and _iters_since_skill are NOT reset here.
agent.iteration_budget = IterationBudget(agent.max_iterations)
# Wall-clock run budget: per-run_conversation clock. Only stamped when a
# budget is configured so the default path stays clock-free; the wrap-up
# latch resets each turn (one notice per run, not per session).
if getattr(agent, "run_budget_seconds", None):
agent._run_budget_started_at = time.time()
else:
agent._run_budget_started_at = None
agent._run_budget_wrapup_injected = False
# Log conversation turn start for debugging/observability.
_preview_text = summarize_user_message_for_log(user_message)
_msg_preview = (_preview_text[:80] + "...") if len(_preview_text) > 80 else _preview_text
@@ -1154,6 +1163,57 @@ def build_turn_context(
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
agent._mute_post_response = False
elif not agent.compression_enabled:
# Uncompressed session guard (#89297): when compression is explicitly
# disabled, sessions can grow past the model's context window across
# hundreds of messages with nothing to shrink them. The warning itself
# fires from the conversation loop's pre-API site, which reuses the
# unconditionally computed request estimate at zero marginal cost and
# covers both turn-start and mid-turn growth (every provider request
# passes through it). Here we only RE-ARM the dedup once the session
# is back under the window, so the guard can warn again after the
# user compacts (/compress with force=True works with compression
# disabled) and the context later regrows past the limit.
_ctx_len = getattr(
getattr(agent, "context_compressor", None), "context_length", None
)
if isinstance(_ctx_len, int) and _ctx_len > 0:
_raw_chars = 0
for _m in messages:
if not isinstance(_m, dict):
continue
_c = _m.get("content")
if isinstance(_c, str):
_raw_chars += len(_c)
elif _c:
# Non-string, non-empty content (multimodal part lists,
# dict payloads) defeats a char count — force the real
# estimate by treating it as over-gate. None/"" (routine
# assistant tool-call rows) contribute nothing.
_raw_chars = _ctx_len + 1
break
# Cheap gate: a session whose raw text is under ~1/4 of the
# window (4 chars/token upper bound) cannot be over it — skip
# the estimator. Non-string (multimodal) content defeats a char
# count, so any such message forces the real estimate.
if _raw_chars <= _ctx_len:
_clear_warn = getattr(
agent, "_clear_context_overflow_warn", None
)
if callable(_clear_warn):
_clear_warn()
else:
_uncompressed_tokens = estimate_request_tokens_rough(
messages,
system_prompt=active_system_prompt or "",
tools=agent.tools or None,
)
if _uncompressed_tokens <= _ctx_len:
_clear_warn = getattr(
agent, "_clear_context_overflow_warn", None
)
if callable(_clear_warn):
_clear_warn()
if _preflight_compressed:
# Compression rebuilt the list (tail messages are fresh compaction
+12
View File
@@ -132,6 +132,18 @@ def get_active_provider() -> Optional[VideoGenProvider]:
except Exception as exc:
logger.debug("Could not read video_gen.provider from config: %s", exc)
# The managed "Nous Subscription" selection is serviced by the FAL
# plugin through the managed fal-queue gateway (the plugin's resolver
# routes managed when the stored selection is "nous").
if configured:
try:
from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER
if configured.lower() == NOUS_MANAGED_PROVIDER:
configured = "fal"
except Exception: # pragma: no cover — helpers are in-repo
pass
with _lock:
snapshot = dict(_providers)
snapshot.update(_scoped_providers.get(hermes_home_key(), {}))
+16
View File
@@ -126,6 +126,22 @@ class WebSearchProvider(abc.ABC):
"""Return True if this provider implements :meth:`search`."""
return True
def is_keyless_available(self) -> bool:
"""Return True when this provider can serve calls WITHOUT credentials.
A separate, weaker tier than :meth:`is_available`: providers with a
public anonymous free tier (Exa / Parallel MCP endpoints) return
True here so the registry can fall back to them when NO provider is
configured or keyed — and only then. Keyless availability must never
make :meth:`is_available` return True, or the legacy preference walk
would route users with real credentials for a lower-priority backend
onto the free tier of a higher-priority one.
Like :meth:`is_available`, this must be cheap and must NOT make
network calls. Default: False.
"""
return False
def supports_extract(self) -> bool:
"""Return True if this provider implements :meth:`extract`.
+68
View File
@@ -166,6 +166,44 @@ _LEGACY_PREFERENCE = (
"ddgs",
)
# Keyless free-tier walk — strictly LAST-resort, tried only after the
# availability-filtered legacy walk finds nothing (i.e. the user has zero
# web credentials and no importable ddgs). All five vendors expose public
# anonymous free tiers (see plugins/web/keyless_mcp.py). Unpinned keyless
# traffic round-robins across the ring per request (the ring cursor lives
# in keyless_mcp; an explicit `hermes tools` pick bypasses this walk
# entirely, and rate-limited requests fail over to the next ring vendor).
# Disable the tier with ``web.keyless_fallback: false``.
_KEYLESS_PREFERENCE = (
"exa",
"parallel",
"tavily",
"firecrawl",
"keenable",
)
def _keyless_preference() -> tuple:
"""Return the keyless walk order for resolution.
Delegates the entry-vendor choice to the ring cursor in
:mod:`plugins.web.keyless_mcp` (round-robin per request, seeded by the
per-process random session id) so resolution and dispatch agree on
which vendor a fresh install starts at. The remaining vendors follow
in ring order as fallbacks for registration gaps.
"""
try:
from plugins.web.keyless_mcp import _KEYLESS_RING, _ring_cursor
start = _ring_cursor % len(_KEYLESS_RING)
return tuple(
_KEYLESS_RING[(start + i) % len(_KEYLESS_RING)]
for i in range(len(_KEYLESS_RING))
)
except Exception as exc: # noqa: BLE001 — ring optional in stripped envs
logger.debug("keyless ring order unavailable: %s", exc)
return _KEYLESS_PREFERENCE
def _resolve(configured: Optional[str], *, capability: str) -> Optional[WebSearchProvider]:
"""Resolve the active provider for a capability ("search" | "extract").
@@ -254,9 +292,39 @@ def _resolve(configured: Optional[str], *, capability: str) -> Optional[WebSearc
):
return provider
# 4. Keyless free-tier walk — the user has NO credentialed/importable
# backend at all. Fall back to providers that can serve anonymously
# (public MCP free tiers), unless disabled via
# ``web.keyless_fallback: false``. This tier never pre-empts a keyed
# setup: it is only reachable when the legacy walk found nothing.
if _keyless_tier_enabled():
for name in _keyless_preference():
provider = snapshot.get(name)
if provider is None or not _capable(provider):
continue
try:
if provider.is_keyless_available():
return provider
except Exception as exc: # noqa: BLE001 — buggy provider skipped
logger.debug(
"provider %s.is_keyless_available() raised %s", name, exc
)
return None
def _keyless_tier_enabled() -> bool:
"""Read ``web.keyless_fallback`` from config.yaml (default: enabled)."""
try:
from hermes_cli.config import load_config
web_cfg = load_config().get("web") or {}
return bool(web_cfg.get("keyless_fallback", True))
except Exception as exc: # noqa: BLE001 — config layer optional
logger.debug("keyless_fallback config read failed: %s", exc)
return True
def _disabled_web_plugin_for(configured: Optional[str] = None, *, capability: Optional[str] = None) -> Optional[str]:
"""Return the plugin key of a *disabled* bundled web plugin that would
have provided the configured backend, or None.
+1 -1
View File
@@ -32,7 +32,7 @@
"clsx": "2.1.1",
"katex": "0.16.47",
"lucide-react": "0.577.0",
"nanostores": "1.4.0",
"nanostores": "1.4.2",
"radix-ui": "1.6.7",
"react": "19.2.7",
"react-dom": "19.2.7",
+17 -1
View File
@@ -151,6 +151,12 @@ Notes:
- SVGs inherit `size-3.5` (`size-3` at `xs`). Don't re-set icon size.
- Polymorph with `asChild` when the button must render as a link/Slot.
## Badges — one component
`src/components/ui/badge.tsx`. Variants: `default` (tinted primary), `muted`,
`warn`, `destructive`, `outline`, `solid` (primary fill — icon-corner counts).
Sizes: `default`, `xs`, `overlay` (titlebar glyph counts).
## Form controls
- **`controlVariants`** (`src/components/ui/control.ts`) is the shared shape for
@@ -190,6 +196,15 @@ Notes:
- **Empty:** `EmptyState` for plain page bodies; `PanelEmpty` for overlay
master/detail empties with an icon and action. Don't hand-roll a third
centered empty.
- **Confirmation:** `ConfirmDialog` is the only way we ask "are you sure". It
opens focused on Confirm, so `Enter` confirms and `Esc` cancels, and it owns
the pending → done → close beat and the inline error — a call site passes an
async `onConfirm` and nothing else. A third way out (e.g. "Remove from
sidebar" beside "Delete worktree") goes in the one `secondaryAction` slot.
Never `window.confirm`: it's an unstyled blocking Chromium modal. A handler
that wants the answer inline instead of a mounted dialog calls `confirm()`
from `src/store/confirm.ts`, which renders this same primitive through the
single `ConfirmHost` at the shell — the way `notify()` backs notifications.
## Chat, tools & boot surfaces
@@ -315,7 +330,8 @@ The detailed state contract lives in the scoped
## Before you add something — checklist
- [ ] Reuse a primitive (`Button`, `SearchField`, `SegmentedControl`,
`ListRow`, `Loader`, `ErrorState`, `LogView`) instead of forking one?
`ListRow`, `Loader`, `ErrorState`, `LogView`, `ConfirmDialog`) instead of
forking one?
- [ ] Tokens (`--ui-*`, `shadow-nous`, `--stroke-nous`) — zero raw colors /
one-off shadows?
- [ ] No `className` overriding a primitive's padding / size / radius / chrome?
+3 -1
View File
@@ -182,7 +182,9 @@ Changing profiles or connection modes is a soft workspace switch, not another
cold boot. The shell and current management overlay remain mounted while
gateway-bound nanostores are wiped, query-backed data is invalidated, and the
new connection repopulates skeletons. This prevents rows or transcripts from
the previous gateway bleeding into the next one.
the previous gateway bleeding into the next one. Switching changes only the
foreground view and request route: it does not cancel turns or stop a backend,
and retained background sockets continue receiving events from running jobs.
### Verification
+86
View File
@@ -0,0 +1,86 @@
/**
* E2E batch clarify test — the multi-question clarify card must mount ONCE.
*
* Regression coverage for the duplicated-card bug: `tool.start` carries the
* model's tool_call_id while `clarify.request` carries a gateway-generated
* request_id. A batch payload has no top-level `question`, so the two rows
* only merge when the correlation key comes from the question list
* (`batchClarifyMatchValue` in lib/chat-messages/tool-parts.ts). Before that
* fix this exact flow rendered two identical interactive cards.
*
* The flow runs the real chain: composer → gateway → agent → clarify tool →
* clarify.request event → renderer, against the mock inference server.
*/
import { expect, test } from './test'
import { type MockBackendFixture, setupMockBackend, waitForAppReady } from './fixtures'
import { BATCH_CLARIFY_QUESTIONS, BATCH_CLARIFY_TRIGGER } from './mock-server'
let fixture: MockBackendFixture | null = null
test.beforeAll(async () => {
fixture = await setupMockBackend()
await waitForAppReady(fixture!, 120_000)
})
test.afterAll(async () => {
await fixture?.cleanup()
fixture = null
})
test.describe('batch clarify card', () => {
test('renders exactly one card and completes via per-question locks', async () => {
const page = fixture!.page
const composer = page.locator('[contenteditable="true"]').first()
await composer.waitFor({ state: 'visible', timeout: 10_000 })
await composer.click()
await composer.type(BATCH_CLARIFY_TRIGGER, { delay: 20 })
await page.keyboard.press('Enter')
// The live batch form marks itself with data-clarify-batch=<count>.
const batchCard = page.locator('form[data-clarify-batch]')
await batchCard.first().waitFor({ state: 'visible', timeout: 60_000 })
// THE regression assertion: one card, not two.
await expect(batchCard).toHaveCount(1)
await expect(batchCard).toHaveAttribute('data-clarify-batch', String(BATCH_CLARIFY_QUESTIONS.length))
// Both questions render inside the single card.
for (const entry of BATCH_CLARIFY_QUESTIONS) {
await expect(batchCard.getByText(entry.question)).toHaveCount(1)
}
// Each question text also appears exactly once in the whole transcript —
// catches a duplicate that mounts outside a form[data-clarify-batch].
for (const entry of BATCH_CLARIFY_QUESTIONS) {
await expect(page.getByText(entry.question)).toHaveCount(1)
}
// Answer both questions: stage picks locally (no server traffic yet).
const confirmButton = batchCard.locator('button[type="submit"]')
await expect(confirmButton).toContainText('Confirm and continue')
await expect(confirmButton).toBeDisabled()
await batchCard.getByRole('button', { name: /Coffee/ }).click()
await expect(confirmButton).toBeDisabled()
await batchCard.getByRole('button', { name: /Morning/ }).click()
await expect(confirmButton).toBeEnabled()
// ONE confirm submits the whole batch.
await confirmButton.click()
// The settled card lists both questions with their locked answers.
const settled = page.locator('[data-clarify-settled]')
await settled.waitFor({ state: 'visible', timeout: 30_000 })
await expect(settled.getByText(BATCH_CLARIFY_QUESTIONS[0].question)).toBeVisible()
await expect(settled.getByText('Coffee', { exact: true })).toBeVisible()
await expect(settled.getByText(BATCH_CLARIFY_QUESTIONS[1].question)).toBeVisible()
await expect(settled.getByText('Morning', { exact: true })).toBeVisible()
// And still no duplicate live card lingering after settle.
await expect(page.locator('form[data-clarify-batch]')).toHaveCount(0)
})
})
+19
View File
@@ -50,6 +50,25 @@ test.describe('dev-mode boot with mock backend', () => {
})
})
// A preload that throws never reaches contextBridge, so the renderer boots
// into "Desktop IPC bridge is unavailable" and every test below it dies on a
// 120s never-became-ready timeout instead. Checking the bridge by name makes
// that failure legible. The sandbox lets preload require only electron,
// events, timers and url — adding any other node builtin lands here.
test('the preload bridge reaches the renderer', async () => {
const bridge = await fixture!.page.evaluate(() => {
const desktop = (window as unknown as { hermesDesktop?: Record<string, unknown> }).hermesDesktop
return {
present: typeof desktop,
glassSupported: typeof desktop?.glassSupported,
translucencySupported: typeof desktop?.translucencySupported
}
})
expect(bridge).toEqual({ present: 'object', glassSupported: 'boolean', translucencySupported: 'boolean' })
})
test('backend boots and app becomes ready', async () => {
// This is the big one — wait for the full boot chain to complete:
// electron starts → hermes serve is spawned → WS connects → config
@@ -0,0 +1,123 @@
/**
* Context-menu edit verbs on real editables — the regressions jsdom cannot
* catch, exercised against the real renderer (real radix focus trap, real
* React unmount timing, real selection).
*
* The class under test: "Select all" from the app context menu must act on
* the FIELD the menu was opened on, never on the surrounding transcript.
* The first fix (focus-restore before dispatch) passed unit tests and still
* failed live because the radix trap steals focus back; the second fix runs
* selection renderer-side after the trap unmounts. These tests pin the
* observable outcome, not the mechanism.
*
* Menu items are addressed by accessible-name PREFIX (`/^Copy/`): the name
* includes the shortcut suffix ("Copy Ctrl+V" / "Copy ⌘V"), which is also
* host-dependent.
*/
import { type MockBackendFixture, setupMockBackend, waitForAppReady } from './fixtures'
import { expect, test } from './test'
let fixture: MockBackendFixture | null = null
test.beforeAll(async () => {
fixture = await setupMockBackend()
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
await fixture?.cleanup()
fixture = null
})
test('select all from the composer context menu selects the draft, not the chat', async () => {
const page = fixture!.page
const composer = page.locator('[data-slot="composer-rich-input"]').first()
// Put a message into the transcript so there is chat text a document-wide
// select-all WOULD grab — the bug this test exists to catch. Wait for the
// mock reply to COMPLETE: while the turn is busy the composer is in its
// steer shape and a typed draft does not land in it.
await composer.click()
await composer.pressSequentially('transcript anchor message')
await page.keyboard.press('Enter')
await page.waitForFunction(() => (document.body.textContent ?? '').includes('mock inference server'), undefined, {
timeout: 60_000
})
// Draft text in the composer, then right-click it.
await composer.click()
await composer.pressSequentially('draft under selection')
await composer.click({ button: 'right' })
const selectAll = page.getByRole('menuitem', { name: /^Select all/ })
await selectAll.waitFor({ state: 'visible', timeout: 10_000 })
await selectAll.click()
// The selection must live inside the composer and cover exactly the draft.
await expect
.poll(
() =>
page.evaluate(() => {
const selection = window.getSelection()
const editable = document.querySelector('[data-slot="composer-rich-input"]')
if (!selection || selection.rangeCount === 0 || !editable) {
return { inside: false, text: '' }
}
return {
inside: editable.contains(selection.getRangeAt(0).commonAncestorContainer),
text: selection.toString()
}
}),
{ timeout: 10_000 }
)
.toEqual({ inside: true, text: 'draft under selection' })
// Clear the draft so later tests start clean.
await page.keyboard.press('Delete')
})
test('cut, copy, and select all gray out in an empty composer', async () => {
const page = fixture!.page
const composer = page.locator('[data-slot="composer-rich-input"]').first()
await composer.click()
await composer.click({ button: 'right' })
const selectAll = page.getByRole('menuitem', { name: /^Select all/ })
await selectAll.waitFor({ state: 'visible', timeout: 10_000 })
await expect(selectAll).toHaveAttribute('data-disabled', /.*/)
await expect(page.getByRole('menuitem', { name: /^Cut/ })).toHaveAttribute('data-disabled', /.*/)
await expect(page.getByRole('menuitem', { name: /^Copy/ })).toHaveAttribute('data-disabled', /.*/)
await page.keyboard.press('Escape')
})
test('paste enables when the clipboard holds text', async () => {
const page = fixture!.page
const composer = page.locator('[data-slot="composer-rich-input"]').first()
// The empty-clipboard branch stays in the unit suite: the e2e app shares
// the SYSTEM clipboard, and writeText('') does not reliably clear it.
await page.evaluate(() =>
(
window as unknown as { hermesDesktop?: { writeClipboard?: (text: string) => Promise<boolean> } }
).hermesDesktop?.writeClipboard?.('clipboard payload')
)
await composer.click()
await composer.click({ button: 'right' })
const paste = page.getByRole('menuitem', { name: /^Paste/ })
await paste.waitFor({ state: 'visible', timeout: 10_000 })
// The clipboard probe is an async IPC — the item enables when it lands.
await expect.poll(() => paste.getAttribute('data-disabled'), { timeout: 10_000 }).toBeNull()
await page.keyboard.press('Escape')
})
+52
View File
@@ -339,6 +339,40 @@ const BLOCKING_CLARIFY_TURN: ScriptedTurn = {
toolCalls: [{ name: 'clarify', args: { question: BLOCKING_CLARIFY_QUESTION, choices: ['Yes', 'No'] } }],
}
/**
* A marker that makes the mock emit a blocking BATCH clarify tool call
* (multi-question form). Regression coverage for the duplicated-card bug:
* the tool.start row and the clarify.request row carry different ids and a
* batch payload has no top-level question, so the correlation key must come
* from the question list or the card mounts twice.
*/
export const BATCH_CLARIFY_TRIGGER = 'E2E_BATCH_CLARIFY_TRIGGER'
export const BATCH_CLARIFY_QUESTIONS = [
{ question: 'Pick a batch drink?', choices: ['Coffee', 'Tea'] },
{ question: 'Pick a batch time?', choices: ['Morning', 'Night'] },
]
const BATCH_CLARIFY_TURN: ScriptedTurn = {
text: '',
toolCalls: [{ name: 'clarify', args: { questions: BATCH_CLARIFY_QUESTIONS } }],
}
function includesBatchClarifyTrigger(value: unknown): boolean {
if (typeof value === 'string') {
return value.includes(BATCH_CLARIFY_TRIGGER)
}
if (Array.isArray(value)) {
return value.some(includesBatchClarifyTrigger)
}
if (value && typeof value === 'object') {
return Object.values(value).some(includesBatchClarifyTrigger)
}
return false
}
function includesBlockingClarifyTrigger(value: unknown): boolean {
if (typeof value === 'string') {
return value.includes(BLOCKING_CLARIFY_TRIGGER)
@@ -484,6 +518,24 @@ export function startMockServer(options: MockServerOptions = {}): Promise<MockSe
return
}
if (includesBatchClarifyTrigger(parsed.messages)) {
// Only the FIRST completion of the conversation scripts the batch
// clarify. The trigger text stays in message history, so once the
// answered tool result is present the turn falls through to the
// canned reply — otherwise the mock loops the quiz forever.
const hasToolResult = Array.isArray(parsed.messages)
&& parsed.messages.some((message: { role?: string }) => message?.role === 'tool')
if (!hasToolResult) {
if (stream) {
streamScriptedTurn(res, model, BATCH_CLARIFY_TURN)
} else {
nonStreamingScriptedTurn(res, model, BATCH_CLARIFY_TURN)
}
return
}
}
if (includesBlockingClarifyTrigger(parsed.messages)) {
if (stream) {
streamScriptedTurn(res, model, BLOCKING_CLARIFY_TURN)
@@ -3,8 +3,10 @@ import assert from 'node:assert/strict'
import { test } from 'vitest'
import {
isHostKeyChangedBootFailure,
isRetryableRemoteBootFailure,
shouldLatchBackendStartFailure,
shouldLatchHostKeyChangedFailure,
shouldLatchRemoteReauthFailure
} from './backend-start-failure'
@@ -81,3 +83,53 @@ test('retryable and reauth-latch are mutually exclusive for remote failures', ()
assert.equal(retry !== latch, true, `remote failure with reauth=${isReauth} must pick exactly one path`)
}
})
test('FIX host-key change: classified from the kind tag and from stringified ssh banners', () => {
// classifySshError tags the Error it built; errors that crossed an IPC or
// string boundary only keep the message. Both shapes must classify.
const tagged = Object.assign(new Error('SSH refused to connect.'), { kind: 'host-key-changed' })
assert.equal(isHostKeyChangedBootFailure(tagged), true)
assert.equal(
isHostKeyChangedBootFailure(new Error('@@@@ WARNING: REMOTE HOST IDENTIFICATION HAS CHANGED! @@@@')),
true
)
assert.equal(isHostKeyChangedBootFailure(new Error('Host key verification failed.')), true)
assert.equal(
isHostKeyChangedBootFailure(new Error('The host key for root@203.0.113.7 has CHANGED since you last connected.')),
true
)
assert.equal(isHostKeyChangedBootFailure(new Error('Connection refused')), false)
assert.equal(isHostKeyChangedBootFailure(null), false)
})
test('FIX host-key change: latches and is never auto-retried (157-failure loop, Aug 2026 bundle)', () => {
// SSH fails closed on a changed host key: every retry re-drives the same
// doomed boot until the user clears known_hosts. Terminal, like reauth.
const context = { attemptedRemote: true, isReauth: false, isHostKeyChanged: true }
assert.equal(shouldLatchHostKeyChangedFailure(context), true)
assert.equal(isRetryableRemoteBootFailure(context), false)
})
test('host-key latch never fires for local failures or ordinary remote faults', () => {
assert.equal(
shouldLatchHostKeyChangedFailure({ attemptedRemote: false, isReauth: false, isHostKeyChanged: true }),
false
)
assert.equal(
shouldLatchHostKeyChangedFailure({ attemptedRemote: true, isReauth: false, isHostKeyChanged: false }),
false
)
assert.equal(shouldLatchHostKeyChangedFailure({ attemptedRemote: true, isReauth: false }), false)
})
test('every remote failure picks exactly one path: retry, reauth latch, or host-key latch', () => {
for (const isReauth of [true, false]) {
for (const isHostKeyChanged of [true, false]) {
const retry = isRetryableRemoteBootFailure({ attemptedRemote: true, isReauth, isHostKeyChanged })
const reauth = shouldLatchRemoteReauthFailure({ attemptedRemote: true, isReauth })
const hostKey = shouldLatchHostKeyChangedFailure({ attemptedRemote: true, isReauth, isHostKeyChanged })
const picked = [retry, reauth, hostKey].filter(Boolean).length
assert.ok(picked >= 1, `remote failure reauth=${isReauth} hostKey=${isHostKeyChanged} fell through every path`)
}
}
})
+42 -3
View File
@@ -79,6 +79,44 @@ export interface RemoteBootRetryContext {
* never self-heal without the user signing in again.
*/
isReauth: boolean
/**
* True when SSH refused to connect because the host's key CHANGED
* (StrictHostKeyChecking fails closed). Retrying cannot succeed until the
* user verifies the change and removes the stale known_hosts entry, so this
* is terminal like a reauth rejection — not connectivity.
*/
isHostKeyChanged?: boolean
}
/**
* A host-key-change refusal is identifiable both by the `kind` tag
* classifySshError puts on the error and — for errors that crossed a
* stringifying boundary — by the stable phrases ssh/our own message carry.
* One user hit 157 consecutive boot-retry failures over 2.5h against a
* reinstalled VPS (Aug 2026 bundle) because this was classified as transient.
*/
export function isHostKeyChangedBootFailure(error: unknown): boolean {
if ((error as { kind?: string } | null | undefined)?.kind === 'host-key-changed') {
return true
}
const message = error instanceof Error ? error.message : String(error ?? '')
return /REMOTE HOST IDENTIFICATION HAS CHANGED|Host key verification failed|host key for .+ has CHANGED/i.test(
message
)
}
/**
* Whether a failed remote boot should latch (into `backendStartFailure`)
* because the host key changed. Same rationale as the reauth latch: the
* failure cannot self-heal, and an unlatched terminal failure makes every
* recovery surface re-drive the identical doomed boot. The latch is released
* by the existing reset/repair/apply-config paths once the user has run
* `ssh-keygen -R <host>`.
*/
export function shouldLatchHostKeyChangedFailure(context: RemoteBootRetryContext): boolean {
return context.attemptedRemote && context.isHostKeyChanged === true
}
/**
@@ -93,9 +131,10 @@ export interface RemoteBootRetryContext {
* only arms after a completed boot, so the app sat on "Desktop boot failed"
* until the user manually re-entered the same connection details (which just
* forced a fresh bootstrap). A missing capability differs from a transient
* failure: confirmed reauth rejections and local failures stay out of the
* retry path; everything else remote is connectivity and should retry.
* failure: confirmed reauth rejections, host-key changes, and local failures
* stay out of the retry path; everything else remote is connectivity and
* should retry.
*/
export function isRetryableRemoteBootFailure(context: RemoteBootRetryContext): boolean {
return context.attemptedRemote && !context.isReauth
return context.attemptedRemote && !context.isReauth && context.isHostKeyChanged !== true
}
@@ -455,10 +455,13 @@ test('apiRequestRegistryConnectionId extracts a genuinely non-local connection i
assert.equal(apiRequestRegistryConnectionId({ connectionId: ' gw-1 ', path: '/x' }), 'gw-1')
})
test('apiRequestRegistryConnectionId resolves null for the legacy/local routes', () => {
test('apiRequestRegistryConnectionId preserves an explicit local registry route', () => {
assert.equal(apiRequestRegistryConnectionId({ connectionId: 'local', path: '/x' }), 'local')
})
test('apiRequestRegistryConnectionId resolves null for unscoped legacy routes', () => {
assert.equal(apiRequestRegistryConnectionId({ path: '/api/cron/jobs' }), null)
assert.equal(apiRequestRegistryConnectionId({ connectionId: '', path: '/x' }), null)
assert.equal(apiRequestRegistryConnectionId({ connectionId: 'local', path: '/x' }), null)
assert.equal(apiRequestRegistryConnectionId({ connectionId: null, path: '/x' }), null)
assert.equal(apiRequestRegistryConnectionId(null), null)
assert.equal(apiRequestRegistryConnectionId(undefined), null)
@@ -715,6 +718,37 @@ test('resolveProfileApiRequest scopes complete safe families according to their
)
})
test('resolveProfileApiRequest routes action-status polls with the action-spawning routes', () => {
// /api/actions/{name}/status must land on the SAME backend as the endpoints
// that spawn actions (skills hub install/uninstall/update, mcp catalog
// install): _spawn_hermes_action registers the dynamic action name only in
// the spawning process. Splitting the pair 404s the poll with
// "Unknown action: skills-install-<slug>-<hash>".
assert.deepEqual(
resolveProfileApiRequest('iris', '/api/actions/skills-install-ascii-art-dd7bccf1/status?lines=200', {
requestMethod: 'GET'
}),
{
backendProfile: null,
requestPath: '/api/actions/skills-install-ascii-art-dd7bccf1/status?lines=200&profile=iris'
}
)
// The spawn side (hub install) and the poll side must agree on the backend.
assert.deepEqual(
resolveProfileApiRequest('iris', '/api/skills/hub/install', {
requestMethod: 'POST'
}),
{ backendProfile: null, requestPath: '/api/skills/hub/install?profile=iris' }
)
// MCP catalog installs spawn background actions too — same pairing rule.
assert.deepEqual(
resolveProfileApiRequest('iris', '/api/mcp/catalog/install', {
requestMethod: 'POST'
}),
{ backendProfile: null, requestPath: '/api/mcp/catalog/install?profile=iris' }
)
})
test('resolveProfileApiRequest preserves remote routing precedence', () => {
assert.deepEqual(
resolveProfileApiRequest('iris', '/api/memory/reset', {
+20 -5
View File
@@ -548,7 +548,11 @@ const LOCAL_PRIMARY_SCOPED_ROUTES = new Set([
'GET /api/skills/hub/search',
'GET /api/skills/hub/sources',
'POST /api/skills/hub/uninstall',
'POST /api/skills/hub/update'
'POST /api/skills/hub/update',
// Spawns a background action polled via /api/actions/{name}/status — must
// live on the SAME backend as that poll family (below), or the poll asks a
// backend that never registered the dynamic action name and 404s.
'POST /api/mcp/catalog/install'
])
function localPrimaryRequestScope(opts: ProfileRouteOptions): boolean | null {
@@ -572,6 +576,16 @@ function localPrimaryRequestScope(opts: ProfileRouteOptions): boolean | null {
return true
}
// Action-status polls MUST land on the same backend as the endpoints that
// spawned them: `_spawn_hermes_action` registers the (often dynamic, e.g.
// `skills-install-<slug>-<hash>`) action name only in the spawning
// process's memory. Every action-spawning route above scopes to the
// primary, so the poll family follows — a pooled-backend poll 404s with
// "Unknown action" even though the install itself succeeded (#89xxx).
if (pathname.startsWith('/api/actions/')) {
return true
}
// Every current /api/tools handler accepts `profile`; every /api/profiles
// handler either aggregates profiles or names its target in the path/body.
// These are the only whole families safe to route through the primary.
@@ -750,15 +764,16 @@ function pathWithProfileScope(path, profile) {
/**
* Registry connection a REST request is explicitly pinned to, or null for the
* legacy profile-routed path. `''`/`'local'` mean the local pool — callers
* only detour through the registry for a genuinely non-local connection, so
* single-source users keep the byte-identical v1 route.
* legacy profile-routed path. An explicit `local` id must stay registry-scoped:
* when the v1 route is remote, only the registry resolver can force the request
* back to this device. Single-source users omit the id and keep the
* byte-identical v1 route.
*/
function apiRequestRegistryConnectionId(request): null | string {
const raw = request && typeof request === 'object' ? (request as { connectionId?: unknown }).connectionId : ''
const id = String(raw ?? '').trim()
if (!id || id === 'local') {
if (!id) {
return null
}
@@ -28,7 +28,10 @@ import {
REGISTRY_VERSION,
rememberSshEnumeration,
removeConnection,
resolvedConnectionId,
resolveRegistryLocalRoute,
setConnectionLaunchMode,
setLastUsedConnection,
setPrimaryConnection,
shouldDeferLocalEnumeration,
shouldRetrySshInventory,
@@ -54,6 +57,49 @@ test('labelSlug kebab-cases and never returns empty for non-empty input', () =>
assert.equal(labelSlug('!!!'), 'connection')
})
test('resolvedConnectionId identifies local and migrated remote descriptors', () => {
const registry = migrateV1ToRegistry({
mode: 'local',
profiles: {
personal: { mode: 'remote', url: 'https://personal.example:9443/', authMode: 'token' },
work: { mode: 'ssh', host: 'work-host', user: 'root' }
}
})
const personal = registry.connections.find(connection => connection.kind === 'remote')
const work = registry.connections.find(connection => connection.kind === 'ssh')
assert.equal(resolvedConnectionId(registry, { mode: 'local' }), LOCAL_CONNECTION_ID)
assert.equal(
resolvedConnectionId(registry, {
baseUrl: 'https://personal.example:9443',
mode: 'remote',
remoteKind: 'url'
}),
personal?.id
)
assert.equal(
resolvedConnectionId(registry, {
baseUrl: 'http://127.0.0.1:49152',
mode: 'remote',
remoteHost: 'root@work-host',
remoteKind: 'ssh'
}),
work?.id
)
})
test('resolvedConnectionId does not guess an unregistered remote', () => {
assert.equal(
resolvedConnectionId(emptyRegistry(), {
baseUrl: 'https://unknown.example',
mode: 'remote',
remoteKind: 'url'
}),
null
)
})
test('agentHandle bare when unique, @name-device shape when duplicated', () => {
assert.equal(agentHandle('research', 'Homelab', false), 'research')
assert.equal(agentHandle('research', 'Homelab', true), 'research-homelab')
@@ -225,6 +271,28 @@ test('rememberSshEnumeration: live list wins, cache then seed default', () => {
})
})
test('rememberSshEnumeration: a bounced remote source keeps its last-known roster (4-bots-show-as-2)', () => {
// A VPS restart makes the remote source unreachable for a few polls. The
// last successful enumeration must keep painting so the roster does not
// silently drop that source's bots mid-outage.
assert.deepEqual(
rememberSshEnumeration({ profiles: null, error: 'unreachable' }, ['default', 'ceo', 'accounter'], 'remote'),
{ profiles: ['default', 'ceo', 'accounter'], error: 'unreachable' }
)
// Never-seen remote source: no seed — an unreachable URL is not evidence a
// backend exists there.
assert.deepEqual(rememberSshEnumeration({ profiles: null, error: 'unreachable' }, null, 'remote'), {
profiles: null,
error: 'unreachable'
})
// Local enumeration failures never reuse a cache (the local runtime answers
// authoritatively or not at all).
assert.deepEqual(rememberSshEnumeration({ profiles: null, error: 'boom' }, ['default'], 'local'), {
profiles: null,
error: 'boom'
})
})
test('shouldRetrySshInventory: first try, cooldown, then retry; cache never retries', () => {
assert.equal(shouldRetrySshInventory(false, null, 1_000), true)
assert.equal(shouldRetrySshInventory(false, 1_000, 30_000, 60_000), false)
@@ -644,6 +712,8 @@ test('normalizeRegistry degrades junk to a local-only registry', () => {
assert.equal(registry.version, REGISTRY_VERSION)
assert.equal(registry.primary, LOCAL_CONNECTION_ID)
assert.equal(registry.launchMode, 'primary')
assert.equal(registry.lastUsed, LOCAL_CONNECTION_ID)
assert.equal(registry.connections.length, 1)
assert.equal(registry.connections[0].kind, 'local')
}
@@ -675,6 +745,8 @@ test('normalizeRegistry round-trips a valid registry unchanged in shape', () =>
const input = {
version: 2,
primary: 'homelab',
launchMode: 'last-used',
lastUsed: 'homelab',
connections: [
{ id: 'local', kind: 'local', label: 'This device' },
{
@@ -700,6 +772,8 @@ test('normalizeRegistry round-trips a valid registry unchanged in shape', () =>
const registry = normalizeRegistry(input)
assert.equal(registry.primary, 'homelab')
assert.equal(registry.launchMode, 'last-used')
assert.equal(registry.lastUsed, 'homelab')
assert.equal(registry.connections.length, 4)
assert.deepEqual(
registry.connections.map(c => c.id),
@@ -709,6 +783,22 @@ test('normalizeRegistry round-trips a valid registry unchanged in shape', () =>
assert.equal(registry.connections[3].port, 2222)
})
test('normalizeRegistry falls back to Primary when the last-used source is missing', () => {
const registry = normalizeRegistry({
version: 2,
primary: 'homelab',
launchMode: 'last-used',
lastUsed: 'retired-host',
connections: [
{ id: 'local', kind: 'local', label: 'This device' },
{ id: 'homelab', kind: 'remote', label: 'Homelab', url: 'http://10.0.0.5:9119' }
]
})
assert.equal(registry.launchMode, 'last-used')
assert.equal(registry.lastUsed, 'homelab')
})
// --- v1 → v2 migration ---
test('migrate: v1 local-only config → local-only registry', () => {
@@ -793,17 +883,19 @@ test('migrate: duplicate host labels are suffixed, not dropped', () => {
// --- registry operations ---
test('removeConnection: local refuses, primary retargets to local', () => {
test('removeConnection: local refuses, primary and last-used retarget safely', () => {
let registry = emptyRegistry()
const entry = normalizeConnectionInput({ kind: 'remote', label: 'Homelab', url: 'http://10.0.0.5:9119' }, registry)
registry = upsertConnection(registry, entry)
registry = setPrimaryConnection(registry, entry.id)
registry = setLastUsedConnection(registry, entry.id)
assert.throws(() => removeConnection(registry, LOCAL_CONNECTION_ID), /cannot be removed/)
const after = removeConnection(registry, entry.id)
assert.equal(after.primary, LOCAL_CONNECTION_ID)
assert.equal(after.lastUsed, LOCAL_CONNECTION_ID)
assert.equal(after.connections.length, 1)
// Removing an unknown id is a no-op, not an error.
assert.equal(removeConnection(after, 'ghost'), after)
@@ -816,6 +908,17 @@ test('setPrimaryConnection validates the target id', () => {
assert.equal(setPrimaryConnection(registry, LOCAL_CONNECTION_ID).primary, LOCAL_CONNECTION_ID)
})
test('last-used source and launch mode validate their persisted values', () => {
let registry = emptyRegistry()
const entry = normalizeConnectionInput({ kind: 'remote', label: 'Homelab', url: 'http://10.0.0.5:9119' }, registry)
registry = upsertConnection(registry, entry)
assert.throws(() => setLastUsedConnection(registry, 'ghost'), /No connection/)
assert.equal(setLastUsedConnection(registry, entry.id).lastUsed, entry.id)
assert.equal(setConnectionLaunchMode(registry, 'last-used').launchMode, 'last-used')
assert.throws(() => setConnectionLaunchMode(registry, 'sometimes'), /Unknown connection launch mode/)
})
test('upsertConnection replaces by id and appends new ids', () => {
let registry = emptyRegistry()
const a = normalizeConnectionInput({ kind: 'remote', label: 'A', url: 'http://a:1' }, registry)
+118 -10
View File
@@ -76,6 +76,11 @@ export interface ConnectionRegistry {
version: typeof REGISTRY_VERSION
/** id of the connection that owns the window/primary backend. */
primary: string
/** Which saved source Sessions should restore when the app launches. */
launchMode: 'last-used' | 'primary'
/** Last source the Sessions workspace successfully opened. Additive in v2
* so registries written before multi-source switching still normalize. */
lastUsed: string
connections: RegistryConnection[]
}
@@ -178,6 +183,79 @@ export interface RegistryLocalRoute {
poolKey: string
}
export interface ResolvedConnectionDescriptor {
baseUrl?: string
mode?: 'local' | 'remote'
remoteHost?: string
remoteKind?: 'cloud' | 'ssh' | 'url'
}
/**
* Recover registry identity for a descriptor resolved through the legacy v1
* profile path. Registry-scoped routes already carry `connectionId`; this
* bridge keeps migrated per-profile remotes truthful until v1 is retired.
*/
export function resolvedConnectionId(
registry: ConnectionRegistry,
descriptor: ResolvedConnectionDescriptor
): null | string {
if (descriptor.mode === 'local') {
return registry.connections.find(connection => connection.kind === 'local')?.id ?? null
}
if (descriptor.mode !== 'remote') {
return null
}
if (descriptor.remoteKind === 'ssh') {
const remoteHost = String(descriptor.remoteHost || '')
.trim()
.toLowerCase()
if (!remoteHost) {
return null
}
return (
registry.connections.find(connection => {
if (connection.kind !== 'ssh') {
return false
}
const host = String(connection.host || '')
.trim()
.toLowerCase()
const target = connection.user ? `${String(connection.user).trim().toLowerCase()}@${host}` : host
return target === remoteHost
})?.id ?? null
)
}
let baseUrl = ''
try {
baseUrl = normalizeRemoteBaseUrl(descriptor.baseUrl)
} catch {
return null
}
return (
registry.connections.find(connection => {
if (connection.kind !== 'cloud' && connection.kind !== 'remote') {
return false
}
try {
return normalizeRemoteBaseUrl(connection.url) === baseUrl
} catch {
return false
}
})?.id ?? null
)
}
/**
* How the registry's 'local' entry resolves a backend for `profile`.
*
@@ -261,10 +339,15 @@ export interface RosterAgent {
}
/**
* SSH roster enumeration skips undialed sources (connect-on-demand). Reuse the
* last successful profile list so Bot Mode does not go empty the moment the
* window switches back to local. Never-seen SSH sources still get a `default`
* seed so the device is clickable.
* Roster enumeration skips undialed sources (connect-on-demand) and reports
* unreachable ones with `profiles: null`. Reuse the last successful profile
* list so Bot Mode does not go empty (or drop to a partial roster) the moment
* a source is briefly unreachable — SSH tunnels drop on sleep/wake, and a
* remote gateway bounce (VPS restart) otherwise erased its bots from the
* roster until the next successful enumeration ("my 4 bots show as 2", Aug
* 2026 bundle). Never-seen SSH sources still get a `default` seed so the
* device is clickable; never-seen remote sources stay empty (no seed) since
* an unreachable URL is not evidence a backend exists there.
*/
export function rememberSshEnumeration(
enumeration: Pick<ConnectionAgents, 'error' | 'profiles'>,
@@ -275,7 +358,7 @@ export function rememberSshEnumeration(
return enumeration
}
if (kind !== 'ssh') {
if (kind === 'local') {
return enumeration
}
@@ -283,7 +366,7 @@ export function rememberSshEnumeration(
return { profiles: cached, error: enumeration.error }
}
if (enumeration.error === 'connect-on-demand') {
if (kind === 'ssh' && enumeration.error === 'connect-on-demand') {
return { profiles: ['default'], error: 'connect-on-demand' }
}
@@ -818,11 +901,15 @@ export function normalizeRegistry(raw: unknown): ConnectionRegistry {
connections.unshift(localEntry())
}
const primary = String(parsed.primary || '').trim()
const storedPrimary = String(parsed.primary || '').trim()
const primary = connections.some(c => c.id === storedPrimary) ? storedPrimary : LOCAL_CONNECTION_ID
const storedLastUsed = String(parsed.lastUsed || '').trim()
return {
version: REGISTRY_VERSION,
primary: connections.some(c => c.id === primary) ? primary : LOCAL_CONNECTION_ID,
primary,
launchMode: parsed.launchMode === 'last-used' ? 'last-used' : 'primary',
lastUsed: connections.some(c => c.id === storedLastUsed) ? storedLastUsed : primary,
connections
}
}
@@ -967,7 +1054,7 @@ export function migrateV1ToRegistry(v1: unknown): ConnectionRegistry {
}
}
return { version: REGISTRY_VERSION, primary, connections }
return { version: REGISTRY_VERSION, primary, launchMode: 'primary', lastUsed: primary, connections }
}
/** Insert or replace by id. Input must already be normalized/validated. */
@@ -994,9 +1081,12 @@ export function removeConnection(registry: ConnectionRegistry, id: string): Conn
throw new Error('The local connection cannot be removed.')
}
const primary = registry.primary === id ? LOCAL_CONNECTION_ID : registry.primary
return {
...registry,
primary: registry.primary === id ? LOCAL_CONNECTION_ID : registry.primary,
primary,
lastUsed: registry.lastUsed === id ? primary : registry.lastUsed,
connections: registry.connections.filter(c => c.id !== id)
}
}
@@ -1009,3 +1099,21 @@ export function setPrimaryConnection(registry: ConnectionRegistry, id: string):
return { ...registry, primary: id }
}
/** Remember the last source the Sessions workspace opened successfully. */
export function setLastUsedConnection(registry: ConnectionRegistry, id: string): ConnectionRegistry {
if (!registry.connections.some(c => c.id === id)) {
throw new Error(`No connection with id "${id}".`)
}
return { ...registry, lastUsed: id }
}
/** Choose whether launch restores the explicit primary or the last-used source. */
export function setConnectionLaunchMode(registry: ConnectionRegistry, launchMode: string): ConnectionRegistry {
if (launchMode !== 'last-used' && launchMode !== 'primary') {
throw new Error(`Unknown connection launch mode "${String(launchMode)}".`)
}
return { ...registry, launchMode }
}
+197
View File
@@ -0,0 +1,197 @@
// IPC surface for local filesystem operations the renderer's project/file
// surfaces use: directory reads, reveal/open in the OS file manager, plugin
// roots + git installs, rename/write/trash. Extracted from main.ts; path
// hardening, HERMES_HOME resolution, and the git binary stay injected.
import fs from 'node:fs'
import path from 'node:path'
import { ipcMain, shell } from 'electron'
import { installDesktopPluginFromGit, probePluginRepo } from './desktop-plugin-install'
import { readDirForIpc } from './fs-read-dir'
import { gitRootForIpc } from './git-root'
export interface FsIpcDeps {
hermesHome: string
readActiveDesktopProfile: () => null | string
expandUserPath: (value: string) => string
resolveRequestedPathForIpc: (value: string, options: { purpose: string }) => string
directoryExists: (value: string) => boolean
resolveGitBinary: () => string
}
export function registerFsIpc({
hermesHome,
readActiveDesktopProfile,
expandUserPath,
resolveRequestedPathForIpc,
directoryExists,
resolveGitBinary
}: FsIpcDeps) {
ipcMain.handle('hermes:fs:readDir', async (_event, dirPath) => readDirForIpc(dirPath))
ipcMain.handle('hermes:fs:gitRoot', async (_event, startPath) => gitRootForIpc(startPath))
// Reveal a path in the OS file manager (Finder / Explorer / Files).
ipcMain.handle('hermes:fs:reveal', async (_event, targetPath) => {
const target = String(targetPath || '').trim()
if (!target) {
return false
}
try {
shell.showItemInFolder(target)
return true
} catch {
return false
}
})
// Open a DIRECTORY in the OS file manager, creating it first if needed. Unlike
// `reveal` (which selects an existing item and silently no-ops on a missing
// path — the "Open plugins folder" Windows bug), this is for the plugins door,
// which often doesn't exist on first use. `shell.openPath` returns '' on
// success or an error string; both mkdir + openPath failures are surfaced.
ipcMain.handle('hermes:fs:openDir', async (_event, dirPath) => {
const dir = String(dirPath || '').trim()
if (!dir) {
return { ok: false, error: 'no path' }
}
try {
await fs.promises.mkdir(dir, { recursive: true })
const error = await shell.openPath(path.normalize(dir))
return error ? { ok: false, error } : { ok: true }
} catch (error) {
return { ok: false, error: error instanceof Error ? error.message : String(error) }
}
})
// The LOCAL Desktop runtime-plugin root: `<HERMES_HOME>/desktop-plugins`,
// resolved from the main-process HERMES_HOME (see resolveHermesHome) — NOT from
// the connected backend. A remote backend reports its own `hermes_home` over
// the gateway, which is a path on the REMOTE box; deriving the plugin dir from
// it yields `undefined/desktop-plugins` (or a non-existent remote path) and the
// on-disk plugin door silently breaks (#66899). Electron owns this resolution
// so it stays valid in every connection mode. Created on demand, like openDir.
async function localPluginsRoot(dirName: string): Promise<string> {
// Profile-aware: a named Desktop profile gets its own plugin root under
// profiles/<name>/, matching the profile-scoped hermes_home the backend
// reported before this resolver existed. 'default'/unset pins the global root.
const profile = readActiveDesktopProfile()
const base = profile && profile !== 'default' ? path.join(hermesHome, 'profiles', profile) : hermesHome
const dir = path.join(base, dirName)
try {
await fs.promises.mkdir(dir, { recursive: true })
} catch {
// Best-effort create; return the path regardless so the reveal action can
// still surface a real openPath error and the scanner can retry later.
}
return dir
}
ipcMain.handle('hermes:fs:desktopPluginsRoot', async () => localPluginsRoot('desktop-plugins'))
// The LOCAL agent-plugin root (`<HERMES_HOME>/plugins`), same Electron-local
// resolution as above. This is the desktop half of a UNIFIED plugin package:
// an agent plugin may ship `desktop/plugin.js` alongside its Python code (the
// same shape as `dashboard/manifest.json`), and the renderer's disk door scans
// this root for it — one installable folder serving both SDKs.
ipcMain.handle('hermes:fs:agentPluginsRoot', async () => localPluginsRoot('plugins'))
ipcMain.handle('hermes:plugin:probe', async (_event, payload) => {
const identifier = String(payload?.identifier || payload?.repo || '').trim()
if (!identifier) {
return { ok: false, error: 'identifier is required', agent: false, desktop: false, warnings: [] }
}
return probePluginRepo(resolveGitBinary(), identifier)
})
ipcMain.handle('hermes:plugin:installDesktop', async (_event, payload) => {
const identifier = String(payload?.identifier || payload?.repo || '').trim()
if (!identifier) {
return { ok: false, error: 'identifier is required' }
}
const desktopPluginsRoot = await localPluginsRoot('desktop-plugins')
return installDesktopPluginFromGit(resolveGitBinary(), identifier, desktopPluginsRoot, Boolean(payload?.force))
})
// Rename a file/folder in place. The renderer passes the existing path + a new
// base name; the destination is resolved in the SAME parent dir so a rename can
// never move the item elsewhere or traverse out. Rejects on a name collision.
ipcMain.handle('hermes:fs:rename', async (_event, targetPath, newName) => {
const src = String(targetPath || '').trim()
const name = String(newName || '').trim()
if (!src || !name || name === '.' || name === '..' || name.includes('/') || name.includes('\\')) {
throw new Error('Invalid rename')
}
const dst = path.join(path.dirname(src), name)
if (dst === src) {
return { path: dst }
}
if (fs.existsSync(dst)) {
throw new Error(`"${name}" already exists`)
}
await fs.promises.rename(src, dst)
return { path: dst }
})
// Write a small UTF-8 text file (e.g. a project's IDEA.md at creation). The path
// is hardened (resolveRequestedPathForIpc) and the parent must already exist —
// this never creates directory trees or escapes the allowed roots, and content
// is size-capped so it can't be abused as a bulk-write primitive.
ipcMain.handle('hermes:fs:writeText', async (_event, filePath, content) => {
const raw = String(filePath || '').trim()
if (!raw) {
throw new Error('Invalid path')
}
const text = String(content ?? '')
if (text.length > 1_000_000) {
throw new Error('Content too large')
}
const resolved = resolveRequestedPathForIpc(expandUserPath(raw), { purpose: 'Write text file' })
if (!directoryExists(path.dirname(resolved))) {
throw new Error('Parent directory does not exist')
}
await fs.promises.writeFile(resolved, text, 'utf8')
return { path: resolved }
})
// Move a file/folder to the OS trash (recoverable) — the VS Code "Delete"
// default. `shell.trashItem` routes to Finder/Explorer/Files trash per platform.
ipcMain.handle('hermes:fs:trash', async (_event, targetPath) => {
const target = String(targetPath || '').trim()
if (!target) {
throw new Error('Invalid delete')
}
await shell.trashItem(target)
return true
})
}
+120
View File
@@ -0,0 +1,120 @@
// IPC surface for git-driven features: worktree management ("Start work"),
// the composer coding rail's repo status, the Codex-style review pane, and
// repo-first project discovery. Extracted from main.ts; the git/gh binary
// resolvers stay injected because main.ts also uses them for self-update and
// plugin installs.
import { ipcMain } from 'electron'
import { scanGitRepos } from './git-repo-scan'
import {
fileDiffVsHead,
repoStatus,
reviewCommit,
reviewCommitContext,
reviewCreatePr,
reviewDiff,
reviewFetchPrComment,
reviewList,
reviewPrList,
reviewPush,
reviewRevert,
reviewRevParse,
reviewShipInfo,
reviewStage,
reviewUnstage
} from './git-review-ops'
import {
addWorktree,
listBaseBranches,
listBranches,
listWorktrees,
removeWorktree,
switchBranch
} from './git-worktree-ops'
export interface GitIpcDeps {
resolveGitBinary: () => string
resolveGhBinary: () => string
}
export function registerGitIpc({ resolveGitBinary, resolveGhBinary }: GitIpcDeps) {
// Git-driven worktree management ("Start work" flow). Errors surface to the
// renderer as rejected promises so it can toast a friendly message.
ipcMain.handle('hermes:git:worktreeList', async (_event, repoPath) => listWorktrees(repoPath, resolveGitBinary()))
ipcMain.handle('hermes:git:worktreeAdd', async (_event, repoPath, options) =>
addWorktree(repoPath, options || {}, resolveGitBinary())
)
ipcMain.handle('hermes:git:worktreeRemove', async (_event, repoPath, worktreePath, options) =>
removeWorktree(repoPath, worktreePath, options || {}, resolveGitBinary())
)
ipcMain.handle('hermes:git:branchSwitch', async (_event, repoPath, branch) =>
switchBranch(repoPath, branch, resolveGitBinary())
)
ipcMain.handle('hermes:git:branchList', async (_event, repoPath) => listBranches(repoPath, resolveGitBinary()))
ipcMain.handle('hermes:git:baseBranchList', async (_event, repoPath) =>
listBaseBranches(repoPath, resolveGitBinary())
)
// Compact repo status (branch, ahead/behind, change counts + files) for the
// composer coding rail. Returns null on a non-repo / remote backend so the rail
// hides cleanly rather than erroring.
ipcMain.handle('hermes:git:repoStatus', async (_event, repoPath) => repoStatus(repoPath, resolveGitBinary()))
// Codex-style review pane: list changed files for a scope, fetch one file's
// unified diff, and stage / unstage / revert. Reads return empty on failure;
// mutations reject so the renderer can toast.
ipcMain.handle('hermes:git:review:list', async (_event, repoPath, scope, baseRef) =>
reviewList(repoPath, scope, baseRef, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:diff', async (_event, repoPath, filePath, scope, baseRef, staged) =>
reviewDiff(repoPath, filePath, scope, baseRef, staged, resolveGitBinary())
)
// Working-tree-vs-HEAD diff for one file (the preview's "show the diff" view).
ipcMain.handle('hermes:git:fileDiff', async (_event, repoPath, filePath) =>
fileDiffVsHead(repoPath, filePath, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:stage', async (_event, repoPath, filePath) =>
reviewStage(repoPath, filePath ?? null, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:unstage', async (_event, repoPath, filePath) =>
reviewUnstage(repoPath, filePath ?? null, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:revert', async (_event, repoPath, filePath) =>
reviewRevert(repoPath, filePath ?? null, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:revParse', async (_event, repoPath, ref) =>
reviewRevParse(repoPath, ref, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:commit', async (_event, repoPath, message, push) =>
reviewCommit(repoPath, message, Boolean(push), resolveGitBinary())
)
ipcMain.handle('hermes:git:review:commitContext', async (_event, repoPath) =>
reviewCommitContext(repoPath, resolveGitBinary())
)
ipcMain.handle('hermes:git:review:push', async (_event, repoPath) => reviewPush(repoPath, resolveGitBinary()))
ipcMain.handle('hermes:git:review:shipInfo', async (_event, repoPath) => reviewShipInfo(repoPath, resolveGhBinary()))
ipcMain.handle('hermes:git:review:prList', async (_event, repoPath, branches, numbers) =>
reviewPrList(repoPath, resolveGhBinary(), branches, numbers)
)
ipcMain.handle('hermes:git:review:fetchPrComment', async (_event, repoPath, url) =>
reviewFetchPrComment(repoPath, resolveGhBinary(), url)
)
ipcMain.handle('hermes:git:review:createPr', async (_event, repoPath) =>
reviewCreatePr(repoPath, resolveGitBinary(), resolveGhBinary())
)
// Repo-first project discovery: scan bounded roots for git repos (pure fs walk,
// no native addon). Never throws to the renderer — failures yield an empty list.
ipcMain.handle('hermes:git:scanRepos', async (_event, roots, options) => {
try {
return await scanGitRepos(roots || [], options || {})
} catch {
return []
}
})
}
+10 -18
View File
@@ -3,31 +3,23 @@ import os from 'node:os'
import path from 'node:path'
import { fileURLToPath } from 'node:url'
// Relative, not `@hermes/shared`: the electron bundle is built by esbuild with
// no tsconfig path resolution (see scripts/bundle-electron-main.mjs), so a bare
// specifier would typecheck and then fail to bundle.
import {
clampDataUrlReadMaxMb,
DATA_URL_READ_DEFAULT_MAX_MB,
DATA_URL_READ_MAX_MAX_MB,
DATA_URL_READ_MIN_MAX_MB
} from '../../shared/src/data-url-read-max'
const DEFAULT_FETCH_TIMEOUT_MS = 15_000
// Default / floor / ceiling for Desktop's data-URL file load (composer attach,
// image preview, etc.). The whole file is base64-buffered in main, so this is
// a memory guard — not a model limit. Settings → Chat takes a free-form MB
// value; 16 MB ships as default. The ceiling is only a typo guard (very large
// values can OOM / crash the app).
const DATA_URL_READ_DEFAULT_MAX_MB = 16
const DATA_URL_READ_MIN_MAX_MB = 1
const DATA_URL_READ_MAX_MAX_MB = 4096
// Remote file.attach sends one base64 JSON-RPC frame. Cap the dedicated attach
// reader so the payload still fits uvicorn's raised ws_max_size (384 MiB)
// after base64 + framing. Preview stays on the Settings-configurable path.
const ATTACHMENT_UPLOAD_DEFAULT_MAX_BYTES = 256 * 1024 * 1024
const TEXT_PREVIEW_SOURCE_MAX_BYTES = 64 * 1024 * 1024
function clampDataUrlReadMaxMb(value) {
const parsed = Number(value)
if (!Number.isFinite(parsed)) {
return DATA_URL_READ_DEFAULT_MAX_MB
}
return Math.min(DATA_URL_READ_MAX_MAX_MB, Math.max(DATA_URL_READ_MIN_MAX_MB, Math.round(parsed)))
}
function dataUrlReadMaxBytesFromMb(maxMb) {
return clampDataUrlReadMaxMb(maxMb) * 1024 * 1024
}
+192
View File
@@ -0,0 +1,192 @@
// IPC surface for HUD mode (the chrome-free floating chat band). Extracted
// from main.ts; the HUD window handle and session-id latch stay injected
// because main.ts owns the window lifecycle and the close broadcast reads the
// latch when handing the session back to the app window.
import { type BrowserWindow, ipcMain } from 'electron'
import { hudFrostFor, type TranslucencyState } from './translucency'
export interface HudIpcDeps {
isMac: boolean
isWindows: boolean
glassSupported: boolean
/** Main's authoritative translucency state (Settings → Appearance). */
getTranslucencyState: () => TranslucencyState
getHudWindow: () => BrowserWindow | null
openHudWindow: (sessionId: null | string, profile: null | string) => void
closeHudWindow: () => void
setHudSessionId: (sessionId: null | string) => void
}
export function registerHudIpc({
isMac,
isWindows,
glassSupported,
getTranslucencyState,
getHudWindow,
openHudWindow,
closeHudWindow,
setHudSessionId
}: HudIpcDeps) {
// Whether the band currently covers the window below the bar. The renderer
// is the only party that can know this (it measures the transcript), and it
// is half of the frost decision — the other half is the user's setting,
// which main owns. Latched so a Settings change can re-decide without
// waiting for the HUD to report again.
let bandShowing = false
let applied: null | string = null
let appliedTo: BrowserWindow | null = null
// Real frosted glass behind the band — the thing CSS backdrop-filter cannot do,
// because Chromium composites a transparent window's page against nothing and
// the desktop is not in its backdrop root. The material IS the window's content
// view, so it frosts the whole rectangle; the HUD's layout leaves no dead
// margins for that reason, and it only turns on while the band is showing
// (idle HUD mode must be the bar and nothing else).
//
// Diffed before issuing: `setVibrancy` carries a 150ms animation that restarts
// if re-issued, so a repeated call would keep the material from ever settling
// (the same churn the chat windows' native-diff contract exists to prevent).
//
// The diff is keyed to the WINDOW as well as the value. A HUD respawn (the
// profile switch in openHudWindow destroys and rebuilds it) hands back a
// fresh window carrying no material, and a latch that only remembered the
// value would recognise its own last answer and skip — leaving the new HUD
// unfrosted until something else happened to change the signature.
const applyHudFrost = () => {
const hudWindow = getHudWindow()
if (!hudWindow || hudWindow.isDestroyed()) {
applied = null
appliedTo = null
return
}
const frost = hudFrostFor(getTranslucencyState(), bandShowing)
const signature = `${frost.vibrancy ?? 'off'}:${frost.backgroundMaterial}`
if (applied === signature && appliedTo === hudWindow) {
return
}
applied = signature
appliedTo = hudWindow
if (isMac && typeof hudWindow.setVibrancy === 'function') {
hudWindow.setVibrancy(frost.vibrancy)
}
if (isWindows && glassSupported && typeof hudWindow.setBackgroundMaterial === 'function') {
hudWindow.setBackgroundMaterial(frost.backgroundMaterial)
}
}
ipcMain.handle('hermes:hud:open', async (_event, request) => {
openHudWindow(
typeof request?.sessionId === 'string' ? request.sessionId : null,
typeof request?.profile === 'string' ? request.profile : null
)
return { ok: true }
})
ipcMain.handle('hermes:hud:frost', (_event, showing) => {
bandShowing = Boolean(showing)
applyHudFrost()
return { ok: true }
})
// Let clicks fall through the HUD wherever it isn't really there. An
// always-on-top window eats every click inside its rectangle, and most of that
// rectangle is a faded-out band over whatever the user is actually working in.
// `forward` keeps mousemove flowing so the renderer can re-arm when the cursor
// reaches the bar.
ipcMain.on('hermes:hud:ignore-mouse', (_event, ignore) => {
const hudWindow = getHudWindow()
if (hudWindow && !hudWindow.isDestroyed()) {
hudWindow.setIgnoreMouseEvents(Boolean(ignore), { forward: true })
}
})
ipcMain.on('hermes:hud:move-by', (event, delta) => {
const hudWindow = getHudWindow()
if (!hudWindow || hudWindow.isDestroyed() || event.sender !== hudWindow.webContents) {
return
}
const dx = Number(delta?.x)
const dy = Number(delta?.y)
const width = Number(delta?.width)
const height = Number(delta?.height)
if (!Number.isFinite(dx) || !Number.isFinite(dy) || !Number.isFinite(width) || !Number.isFinite(height)) {
return
}
const [x, y] = hudWindow.getPosition()
// setBounds — NOT setPosition: on Windows, a transparent frameless window
// silently grows ~1px per setPosition call (worse at >100% DPI). The renderer
// snapshots outerWidth/outerHeight when the composer drag arms and re-pins
// to that size on every moveBy (same pattern as the pet overlay drag).
hudWindow.setBounds({
x: Math.round(x + dx),
y: Math.round(y + dy),
width: Math.round(width),
height: Math.round(height)
})
})
// Resize from the HUD's corner handle. The window is created non-resizable
// (see spawnHudWindow — a transparent frameless window must not expose a
// system resize hot-zone, or dragging grows it), which on Windows/Linux also
// blocks programmatic setBounds sizing — so briefly flip resizable on while
// the size actually changes, exactly like the pet overlay's wheel-scale does.
ipcMain.on('hermes:hud:set-bounds', (event, bounds) => {
const hudWindow = getHudWindow()
if (!hudWindow || hudWindow.isDestroyed() || event.sender !== hudWindow.webContents || !bounds) {
return
}
const win = hudWindow
const width = Math.max(380, Math.round(Number(bounds.width)))
const height = Math.max(160, Math.round(Number(bounds.height)))
const [curW, curH] = win.getSize()
const resizing = width !== curW || height !== curH
if (resizing && !win.isResizable()) {
win.setResizable(true)
}
win.setBounds({ x: Math.round(Number(bounds.x)), y: Math.round(Number(bounds.y)), width, height })
if (resizing) {
win.setResizable(false)
}
})
// The HUD renderer reporting which session it is on, so the close broadcast
// can hand it back to the app window (see hudSessionId).
ipcMain.on('hermes:hud:session', (event, sessionId) => {
const hudWindow = getHudWindow()
if (hudWindow && !hudWindow.isDestroyed() && event.sender === hudWindow.webContents) {
setHudSessionId(typeof sessionId === 'string' && sessionId ? sessionId : null)
}
})
ipcMain.handle('hermes:hud:close', async () => {
closeHudWindow()
return { ok: true }
})
// Main re-applies the frost when the translucency SETTING changes, since the
// band's own report only fires when the band itself moves.
return { applyHudFrost }
}
@@ -1,81 +0,0 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { imageContextMenuItems } from './image-context-menu'
function createActions() {
const calls = {
copyImageAt: [],
openImage: [],
copyImageAddress: [],
saveImage: []
}
return {
calls,
actions: {
copyImageAt: (x, y) => calls.copyImageAt.push([x, y]),
openImage: url => calls.openImage.push(url),
copyImageAddress: url => calls.copyImageAddress.push(url),
saveImage: url => calls.saveImage.push(url)
}
}
}
test('keeps Copy Image available when Chromium omits a large image srcURL', () => {
const { actions, calls } = createActions()
const items = imageContextMenuItems(
{ mediaType: 'image', hasImageContents: true, srcURL: '', x: 100, y: 120 },
actions
)
assert.deepEqual(
items.map(item => item.label),
['Copy Image']
)
items[0].click()
assert.deepEqual(calls.copyImageAt, [[100, 120]])
})
test('keeps URL-dependent image actions when srcURL is available', () => {
const { actions, calls } = createActions()
const url = 'https://example.com/image.png'
const items = imageContextMenuItems({ mediaType: 'image', hasImageContents: true, srcURL: url, x: 5, y: 8 }, actions)
assert.deepEqual(
items.map(item => item.label),
['Open Image', 'Copy Image', 'Copy Image Address', 'Save Image As...']
)
items[0].click()
items[1].click()
items[2].click()
items[3].click()
assert.deepEqual(calls.openImage, [url])
assert.deepEqual(calls.copyImageAt, [[5, 8]])
assert.deepEqual(calls.copyImageAddress, [url])
assert.deepEqual(calls.saveImage, [url])
})
test('does not add image actions for a non-image target', () => {
const { actions } = createActions()
assert.deepEqual(
imageContextMenuItems({ mediaType: 'none', hasImageContents: false, srcURL: '', x: 0, y: 0 }, actions),
[]
)
})
test('does not offer Copy Image when the target has no decoded image contents', () => {
const { actions } = createActions()
assert.deepEqual(
imageContextMenuItems({ mediaType: 'image', hasImageContents: false, srcURL: '', x: 0, y: 0 }, actions),
[]
)
})
@@ -1,40 +0,0 @@
export function imageContextMenuItems(params, actions) {
if (params.mediaType !== 'image' || !params.hasImageContents) {
return []
}
const items = []
const srcURL = params.srcURL || ''
if (srcURL) {
items.push({
label: 'Open Image',
click: () => {
if (!srcURL.startsWith('data:')) {
actions.openImage(srcURL)
}
},
enabled: !srcURL.startsWith('data:')
})
}
items.push({
label: 'Copy Image',
click: () => actions.copyImageAt(params.x, params.y)
})
if (srcURL) {
items.push(
{
label: 'Copy Image Address',
click: () => actions.copyImageAddress(srcURL)
},
{
label: 'Save Image As...',
click: () => actions.saveImage(srcURL)
}
)
}
return items
}
File diff suppressed because it is too large Load Diff
@@ -13,6 +13,7 @@ import { test } from 'vitest'
import {
oauthGuardMayHardFail,
oauthSessionIsLive,
resolveGatedDownloadAuth,
resolveJsonBody,
resolveOauthRestAuth,
resolveReadinessProbeAuth
@@ -130,3 +131,20 @@ test('oauthGuardMayHardFail keeps the strict guard when the list is unusable', (
assert.equal(oauthGuardMayHardFail('nonsense' as any), true)
assert.equal(oauthGuardMayHardFail([{ supportsPassword: true }]), true)
})
// --- 6. gated download auth (guards the Files-panel 401 on cookieless native) ---
test('resolveGatedDownloadAuth matches oauth REST: bearer first, then cookie', () => {
assert.deepEqual(resolveGatedDownloadAuth('oauth', 'native-at'), { kind: 'bearer', token: 'native-at' })
assert.deepEqual(resolveGatedDownloadAuth('oauth', null), { kind: 'cookie' })
assert.deepEqual(resolveGatedDownloadAuth('oauth', ''), { kind: 'cookie' })
})
test('resolveGatedDownloadAuth uses the session token for token and local modes', () => {
assert.deepEqual(resolveGatedDownloadAuth('token', 'native-at', 'session-token'), {
kind: 'token',
token: 'session-token'
})
assert.deepEqual(resolveGatedDownloadAuth('local', null, 'sess'), { kind: 'token', token: 'sess' })
assert.deepEqual(resolveGatedDownloadAuth(undefined, null, null), { kind: 'token', token: null })
})
+29 -2
View File
@@ -2,7 +2,7 @@
* native-auth-decisions.ts
*
* Pure decision helpers extracted from main.ts for the RFC 8252 native-app
* auth flow. These encode three choices that were each the site of a real
* auth flow. These encode six choices that were each the site of a real
* runtime bug — invisible to the mocked flow tests because the tests never
* exercised the real main.ts internals. Keeping them pure + unit-tested here
* prevents silent regressions:
@@ -31,7 +31,12 @@
* can satisfy neither the native-bearer nor the OAuth-partition-cookie
* check by design, so the pre-flight guard must not hard-fail it.
*
* All five are trivial once named; the value is the test that pins the
* 6. resolveGatedDownloadAuth — file save/read must present the SAME
* credentials as oauth REST. `saveGatewayFile` used to always ride the
* OAuth cookie partition, so a cookieless native (or native-password)
* session could list files via `hermes:api` and still 401 on Download.
*
* All six are trivial once named; the value is the test that pins the
* contract so the god-file call sites can't drift back to the buggy shape.
*/
@@ -106,6 +111,28 @@ export function resolveReadinessProbeAuth(
return { kind: 'public' }
}
export type GatedDownloadAuth = OauthRestAuth | { kind: 'token'; token: string | null }
/**
* Decide how a gated file download authenticates.
*
* Must match oauth REST (`resolveOauthRestAuth`): native bearer when present,
* else the OAuth cookie partition. Token/local connections keep the static
* session-token header. A cookie-only download against a cookieless native
* session is the #88987 401 — Files panel listing works, Download does not.
*/
export function resolveGatedDownloadAuth(
authMode: string | null | undefined,
nativeAccessToken?: string | null,
connectionToken?: string | null
): GatedDownloadAuth {
if (authMode === 'oauth') {
return resolveOauthRestAuth(nativeAccessToken)
}
return { kind: 'token', token: connectionToken ?? null }
}
export interface AdvertisedAuthProvider {
name?: string
supportsPassword?: boolean
+151
View File
@@ -0,0 +1,151 @@
// IPC surface for the pop-out pet overlay (mascot window). Extracted from
// main.ts; window handles stay injected because main.ts owns their lifecycle.
import { type BrowserWindow, ipcMain } from 'electron'
export interface PetOverlayIpcDeps {
getMainWindow: () => BrowserWindow | null
getPetOverlayWindow: () => BrowserWindow | null
openPetOverlay: (bounds: unknown) => void
closePetOverlay: () => void
}
export function registerPetOverlayIpc({
getMainWindow,
getPetOverlayWindow,
openPetOverlay,
closePetOverlay
}: PetOverlayIpcDeps) {
// `request` is `{ bounds, screen }`. A fresh pop-out passes viewport-space
// bounds (screen=false): convert to screen space by adding the main window's
// content origin so the pet lands where it sat in-window. A remembered/dragged
// spot passes screen-space bounds (screen=true) and is used as-is. We return the
// resolved screen bounds so the renderer can persist exactly where it opened.
ipcMain.handle('hermes:pet-overlay:open', async (_event, request) => {
const bounds = request && request.bounds ? request.bounds : request
const isScreen = Boolean(request && request.screen)
const mainWindow = getMainWindow()
let screenBounds = bounds
try {
if (bounds && !isScreen && mainWindow && !mainWindow.isDestroyed()) {
const content = mainWindow.getContentBounds()
screenBounds = {
x: content.x + (bounds.x || 0),
y: content.y + (bounds.y || 0),
width: bounds.width,
height: bounds.height
}
}
} catch {
// Fall back to raw bounds if the window geometry is unavailable.
}
openPetOverlay(screenBounds)
return { ok: true, bounds: screenBounds }
})
ipcMain.handle('hermes:pet-overlay:close', async () => {
closePetOverlay()
return { ok: true }
})
// Drag/resize: the overlay reports new absolute screen bounds (it already knows
// the pointer's screen coords). Drag keeps the size constant; the wheel-to-scale
// gesture grows/shrinks it so the sprite is never cropped by the window edge.
// The window is created non-resizable (no stray edge-drag on the transparent
// frameless panel), which on Windows/Linux also blocks programmatic setBounds
// sizing — so briefly flip resizable on whenever the size actually changes.
ipcMain.on('hermes:pet-overlay:set-bounds', (_event, bounds) => {
const petOverlayWindow = getPetOverlayWindow()
if (!petOverlayWindow || petOverlayWindow.isDestroyed() || !bounds) {
return
}
const win = petOverlayWindow
const width = Math.max(80, Math.round(bounds.width))
const height = Math.max(80, Math.round(bounds.height))
const [curW, curH] = win.getSize()
const resizing = width !== curW || height !== curH
if (resizing && !win.isResizable()) {
win.setResizable(true)
}
win.setBounds({ x: Math.round(bounds.x), y: Math.round(bounds.y), width, height })
if (resizing) {
win.setResizable(false)
}
})
// Click-through: the overlay window is a full rectangle but only the pet pixels
// should be interactive. The renderer toggles this as the cursor enters/leaves
// the sprite so transparent margins pass clicks to whatever is behind.
ipcMain.on('hermes:pet-overlay:ignore-mouse', (_event, ignore) => {
const petOverlayWindow = getPetOverlayWindow()
if (petOverlayWindow && !petOverlayWindow.isDestroyed()) {
petOverlayWindow.setIgnoreMouseEvents(Boolean(ignore), { forward: true })
}
})
// The overlay is a non-activating panel (focusable:false) so it never steals
// the app's cmd/alt-tab anchor from the main window. But the pop-up composer
// needs the keyboard, so the renderer asks us to flip it focusable + focus it
// while the composer is open, then back to non-activating when it closes.
ipcMain.on('hermes:pet-overlay:set-focusable', (_event, focusable) => {
const petOverlayWindow = getPetOverlayWindow()
if (!petOverlayWindow || petOverlayWindow.isDestroyed()) {
return
}
petOverlayWindow.setFocusable(Boolean(focusable))
if (focusable) {
petOverlayWindow.focus()
}
})
// Main renderer → overlay: forward the latest pet state for the overlay to render.
ipcMain.on('hermes:pet-overlay:state', (_event, payload) => {
const petOverlayWindow = getPetOverlayWindow()
if (petOverlayWindow && !petOverlayWindow.isDestroyed()) {
petOverlayWindow.webContents.send('hermes:pet-overlay:state', payload)
}
})
// Overlay → main renderer: control messages (pop back in, composer submit).
ipcMain.on('hermes:pet-overlay:control', (_event, payload) => {
const mainWindow = getMainWindow()
if (!mainWindow || mainWindow.isDestroyed()) {
return
}
// Double-click toggles the app window: hide it away if it's up front, bring it
// back if it's minimized/buried. Pure window control — nothing for the
// renderer to do, so don't forward it.
if (payload && payload.type === 'toggle-app') {
if (mainWindow.isMinimized() || !mainWindow.isVisible()) {
mainWindow.show()
mainWindow.focus()
} else {
mainWindow.minimize()
}
return
}
// The mail icon means "take me to the app": raise the main window (it may be
// minimized or buried) before the renderer navigates to the latest thread.
if (payload && payload.type === 'open-app') {
if (mainWindow.isMinimized()) {
mainWindow.restore()
}
mainWindow.show()
mainWindow.focus()
}
mainWindow.webContents.send('hermes:pet-overlay:control', payload)
})
}
+33 -3
View File
@@ -1,6 +1,17 @@
import { contextBridge, ipcRenderer, webUtils } from 'electron'
import { contextBridge, ipcRenderer, webFrame, webUtils } from 'electron'
// Which translucency the OS can back. Asked synchronously because the renderer
// needs it before its first paint, and answered by main because deciding it
// needs `os.release()` — a sandboxed preload may only require electron, events,
// timers and url, so importing node:os here throws before contextBridge runs
// and takes the ENTIRE bridge down with it (window.hermesDesktop undefined =>
// "Desktop IPC bridge is unavailable"). No reply means no glass, which degrades
// to an ordinary opaque window rather than a page thinned over nothing.
const translucencySupport = ipcRenderer.sendSync('hermes:translucency:support')
contextBridge.exposeInMainWorld('hermesDesktop', {
glassSupported: translucencySupport?.glass === true,
translucencySupported: translucencySupport?.translucency === true,
getConnection: profile => ipcRenderer.invoke('hermes:connection', profile),
// Registry-scoped backend resolution: { connectionId, profile } → descriptor.
getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload),
@@ -64,7 +75,10 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
setIgnoreMouse: ignore => ipcRenderer.send('hermes:hud:ignore-mouse', ignore),
moveBy: delta => ipcRenderer.send('hermes:hud:move-by', delta),
setBounds: bounds => ipcRenderer.send('hermes:hud:set-bounds', bounds),
setVibrancy: on => ipcRenderer.invoke('hermes:hud:vibrancy', on),
// Whether the band covers the window below the bar. Main pairs it with the
// user's translucency setting to decide the native frost (macOS vibrancy /
// Windows 11 DWM backdrop) — see hudFrostFor.
setFrost: showing => ipcRenderer.invoke('hermes:hud:frost', showing),
// The HUD tells main which session it is on; main hands that back to the
// app window when the HUD closes, so the app can re-home onto it.
setSession: sessionId => ipcRenderer.send('hermes:hud:session', sessionId),
@@ -137,9 +151,12 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
save: payload => ipcRenderer.invoke('hermes:connections:save', payload),
remove: id => ipcRenderer.invoke('hermes:connections:remove', id),
setPrimary: id => ipcRenderer.invoke('hermes:connections:set-primary', id),
setLaunchMode: mode => ipcRenderer.invoke('hermes:connections:set-launch-mode', mode),
setLastUsed: id => ipcRenderer.invoke('hermes:connections:set-last-used', id),
test: id => ipcRenderer.invoke('hermes:connections:test', id),
// Fan out `hermes update` to every eligible registered connection.
updateAll: () => ipcRenderer.invoke('hermes:connections:update-all'),
// Optional excludeIds skips rows the caller updates through another path.
updateAll: options => ipcRenderer.invoke('hermes:connections:update-all', options),
// Registry lifecycle push (main → renderer): a connection was removed or
// materially edited, so secondaries scoped to it must be disposed (and,
// for edits, re-dialed at the new target).
@@ -185,6 +202,16 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
readClipboard: () => ipcRenderer.invoke('hermes:readClipboard'),
saveGatewayFile: payload => ipcRenderer.invoke('hermes:saveGatewayFile', payload),
saveImageFromUrl: url => ipcRenderer.invoke('hermes:saveImageFromUrl', url),
contextMenuEdit: command => ipcRenderer.invoke('hermes:context-menu:edit', command),
contextMenuCopyImage: () => ipcRenderer.invoke('hermes:context-menu:copy-image'),
contextMenuSpellcheck: action => ipcRenderer.invoke('hermes:context-menu:spellcheck', action),
contextMenuGuestAddWord: payload => ipcRenderer.invoke('hermes:context-menu:guest-add-word', payload),
onContextMenuSpellcheck: callback => {
const listener = (_event, payload) => callback(payload)
ipcRenderer.on('hermes:context-menu-spellcheck', listener)
return () => ipcRenderer.removeListener('hermes:context-menu-spellcheck', listener)
},
saveImageBuffer: (data, ext) => ipcRenderer.invoke('hermes:saveImageBuffer', { data, ext }),
saveClipboardImage: () => ipcRenderer.invoke('hermes:saveClipboardImage'),
getPathForFile: file => {
@@ -218,6 +245,9 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
zoom: {
// Current zoom of this window, as { level, percent }.
get: () => ipcRenderer.invoke('hermes:zoom:get'),
// Synchronous zoom factor (1 = 100%). Coordinate math needs it in the
// same tick as the event it converts, so no IPC round-trip here.
factor: () => webFrame.getZoomFactor(),
setPercent: percent => ipcRenderer.send('hermes:zoom:set-percent', percent),
// Fires on every zoom change, including the Ctrl/Cmd +/-/0 shortcuts,
// so the settings UI can stay in sync with the keyboard.
+13 -20
View File
@@ -28,6 +28,7 @@
import crypto from 'node:crypto'
import { parseRemoteProfileListing } from './connection-registry'
import { assertBootstrapNotSuperseded } from './ssh-connection'
const LOCKFILE_SCHEMA_VERSION = 2
// Bumped when the desktop<->dashboard reuse contract changes in a way that makes
@@ -558,7 +559,7 @@ async function scrapeReadyPort(ssh, logPath, { timeoutMs = DEFAULT_READY_TIMEOUT
const remoteLog = expandRemotePath(logPath)
while (Date.now() < deadline) {
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
if (isAlive && !(await isAlive())) {
const err: any = new Error('Remote dashboard process exited before announcing its port.')
@@ -696,14 +697,6 @@ async function cancelForwardSafe(deps, localPort, remotePort) {
}
}
function assertNotAborted(signal) {
if (signal?.aborted) {
const error: any = new Error('SSH bootstrap was cancelled.')
error.kind = 'superseded'
throw error
}
}
function isForwardBindCollision(error) {
return /address already in use|cannot listen to port|bind.*failed/i.test(String(error?.message || error || ''))
}
@@ -769,7 +762,7 @@ async function connect(deps) {
const log = msg => rememberLog(`[ssh-lifecycle] ${msg}`)
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
const platform = await probeRemotePlatform(ssh)
log(`remote platform ${platform.os}/${platform.arch}`)
const hermesPath = await locateHermes(ssh, remoteHermesPath)
@@ -810,7 +803,7 @@ async function connect(deps) {
lock.hermesHome === hermesHome
if (reusable) {
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
const localPort = await openForward(deps, lock.port)
try {
@@ -827,7 +820,7 @@ async function connect(deps) {
}
if (reuseClassification === 'authenticated-stale') {
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
await cancelForwardSafe(deps, localPort, lock.port)
await cleanupStale(ssh, ownershipId, lock)
} else if (reuseClassification === 'authenticated-ok') {
@@ -840,7 +833,7 @@ async function connect(deps) {
'reused remote dashboard'
)
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
log(`reusing remote dashboard pid=${lock.pid} port=${lock.port}`)
return {
@@ -868,12 +861,12 @@ async function connect(deps) {
throw error
}
} else {
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
await cleanupStale(ssh, ownershipId, lock, pidAlive)
}
}
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
const spawnToken = mintToken()
const { pid, spawnNonce, logPath, tokenFilePath } = await spawnRemoteDashboard(ssh, {
@@ -914,21 +907,21 @@ async function connect(deps) {
isAlive: () => remotePidAlive(ssh, pid),
signal
})
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
log(`remote dashboard bound port ${remotePort}`)
localPort = await openForward(deps, remotePort)
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
const baseUrl = `http://127.0.0.1:${localPort}`
await waitForHermes(baseUrl, spawnToken)
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
const token = await adoptOwnedServedToken(adoptServedToken, baseUrl, spawnToken, ssh, pid, 'remote dashboard')
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
const tokenFingerprint = fingerprintToken(token)
await writeLockfile(ssh, ownershipId, { ...ownedSpawn, port: remotePort, tokenFingerprint })
assertNotAborted(signal)
assertBootstrapNotSuperseded(signal)
return {
baseUrl,
@@ -3,6 +3,7 @@ import assert from 'node:assert/strict'
import { test } from 'vitest'
import {
buildInstanceWindowUrl,
buildSessionWindowUrl,
chatWindowWebPreferences,
createSessionWindowRegistry,
@@ -88,6 +89,19 @@ test('buildSessionWindowUrl adds the watch flag for spectator windows, before th
assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1#/abc')
})
test('buildInstanceWindowUrl marks a full peer without selecting a specialized renderer', () => {
const url = buildInstanceWindowUrl({ devServer: 'http://localhost:5173/' })
assert.equal(url, 'http://localhost:5173/?peer=1')
assert.ok(!url.includes('win='))
})
test('buildInstanceWindowUrl marks a packaged full peer', () => {
const url = buildInstanceWindowUrl({ rendererIndexPath: '/opt/app/index.html' })
assert.match(url, /^file:\/\/.*index\.html\?peer=1$/)
})
test('instanceWindowBounds cascades a new window off its source bounds', () => {
const bounds = instanceWindowBounds({ x: 100, y: 120, width: 1400, height: 900 }, { width: 1, height: 1 })
+18
View File
@@ -77,6 +77,23 @@ function buildSessionWindowUrl(sessionId: string, { devServer, rendererIndexPath
return `${pathToFileURL(rendererIndexPath).toString()}${query}${route}`
}
// Full peer windows render the ordinary app shell, so they deliberately do
// not use the `win` query parameter that selects a specialized renderer. The
// separate marker lets the renderer distinguish a peer from the one primary
// app window: app-launch source restoration belongs to the primary only, while
// a peer keeps the already-running backend it joined during boot.
function buildInstanceWindowUrl({ devServer, rendererIndexPath }: any = {}) {
const query = '?peer=1'
if (devServer) {
const base = devServer.endsWith('/') ? devServer.slice(0, -1) : devServer
return `${base}/${query}`
}
return `${pathToFileURL(rendererIndexPath).toString()}${query}`
}
// Full "instance" windows (⌘⇧N / the "New Window" command) open a complete app
// peer, not a compact chat. Cascade each one off its source window's bounds so a
// new window doesn't land exactly on top of the one it was spawned from. Pure so
@@ -160,6 +177,7 @@ function createSessionWindowRegistry() {
}
export {
buildInstanceWindowUrl,
buildSessionWindowUrl,
chatWindowWebPreferences,
createSessionWindowRegistry,
+12
View File
@@ -988,7 +988,19 @@ function createSshProbeConnection(config, options: any = {}) {
return new SshConnection(config, { ...options, mux: false })
}
// Bootstrap loops poll a remote for readiness; a newer attempt aborts the
// signal so the stale one stops polling and unwinds. `superseded` tells the
// caller this was replaced, not that it failed.
function assertBootstrapNotSuperseded(signal) {
if (signal?.aborted) {
const error: any = new Error('SSH bootstrap was cancelled.')
error.kind = 'superseded'
throw error
}
}
export {
assertBootstrapNotSuperseded,
baseSshOptions,
buildControlArgs,
buildExecArgs,
+377
View File
@@ -0,0 +1,377 @@
// The embedded terminal's PTY host: shell resolution, env scrubbing, session
// registry, and the hermes:terminal:* IPC surface. Extracted from main.ts; the
// factory owns the session map and returns the dispose helpers main.ts needs
// for SSH teardown. findOnPath / logging / connection routing stay injected.
import { execFile } from 'node:child_process'
import crypto from 'node:crypto'
import fs from 'node:fs'
import path from 'node:path'
import { app, ipcMain } from 'electron'
import nodePty from 'node-pty'
import { resolveTerminalConnection } from './connection-apply'
import { ensureSpawnHelperExecutable } from './spawn-helper-perms'
import { buildInteractiveSshArgs } from './ssh-connection'
import { buildWindowsInteractiveCommand } from './windows-remote-lifecycle'
export interface TerminalIpcDeps {
isWindows: boolean
findOnPath: (command: string) => null | string
rememberLog: (line: string) => void
activeSshTerminalTarget: () => unknown
ensureBackend: () => Promise<unknown>
getSshConnectionState: (scope: string) => undefined | { remotePlatform?: string }
}
export interface TerminalIpcApi {
disposeTerminalSession: (id: string) => boolean
disposeTerminalSessionsForSshScope: (scope: string) => void
disposeAllTerminalSessions: () => void
}
export function registerTerminalIpc({
isWindows,
findOnPath,
rememberLog,
activeSshTerminalTarget,
ensureBackend,
getSshConnectionState
}: TerminalIpcDeps): TerminalIpcApi {
const terminalSessions = new Map()
function isExecutableFile(filePath) {
if (!filePath || !path.isAbsolute(filePath)) {
return false
}
try {
fs.accessSync(filePath, fs.constants.X_OK)
return true
} catch {
return false
}
}
function posixShellSpec(shellPath) {
const shellName = path.basename(shellPath)
const interactiveArgs = shellName.includes('zsh') || shellName.includes('bash') ? ['-il'] : ['-i']
return { args: interactiveArgs, command: shellPath, name: shellName }
}
// Windows PowerShell 5.1 ships at a fixed System32 path on every Windows box;
// prefer it only after PowerShell 7+ (`pwsh`).
function windowsPowerShellPath() {
const systemRoot = process.env.SystemRoot || process.env.windir || 'C:\\Windows'
const builtin = path.join(systemRoot, 'System32', 'WindowsPowerShell', 'v1.0', 'powershell.exe')
return isExecutableFile(builtin) ? builtin : findOnPath('powershell.exe')
}
// Map a resolved shell path to its spawn spec, picking interactive flags by
// family: PowerShell drops its logo banner (so the prompt sits flush like the
// POSIX shells), cmd needs nothing, and everything else (zsh/bash/fish/sh…)
// gets POSIX interactive-login flags.
function shellSpecFor(shellPath) {
const name = path.basename(shellPath).toLowerCase()
if (name.startsWith('pwsh') || name.startsWith('powershell')) {
return { args: ['-NoLogo'], command: shellPath, name }
}
if (name.startsWith('cmd')) {
return { args: [], command: shellPath, name }
}
return posixShellSpec(shellPath)
}
// Best installed Windows shell: PowerShell 7+ (`pwsh`), then Windows PowerShell
// 5.1, then comspec/cmd.exe as the universal fallback.
function windowsShellSpec() {
const command =
findOnPath('pwsh.exe') || findOnPath('pwsh') || windowsPowerShellPath() || process.env.COMSPEC || 'cmd.exe'
return shellSpecFor(command)
}
// Resolve the interactive shell for the embedded terminal: an explicit user
// override wins, otherwise auto-detect the best one installed for the platform.
function terminalShellCommand() {
// HERMES_DESKTOP_SHELL is the cross-platform escape hatch (a path or a bare
// name on PATH); $SHELL is honored on POSIX, where it's the user's canonical
// choice, but ignored on Windows, where it's usually a stray MSYS/Git path
// node-pty can't spawn natively.
const override = (process.env.HERMES_DESKTOP_SHELL || (isWindows ? '' : process.env.SHELL) || '').trim()
if (override) {
const resolved = isExecutableFile(override) ? override : findOnPath(override)
if (resolved) {
return shellSpecFor(resolved)
}
}
if (isWindows) {
return windowsShellSpec()
}
const shellPath = ['/bin/zsh', '/bin/bash', '/bin/sh'].find(candidate => isExecutableFile(candidate))
return posixShellSpec(shellPath || '/bin/sh')
}
function safeTerminalCwd(cwd) {
const candidate = path.resolve(String(cwd || app.getPath('home')))
try {
const stat = fs.statSync(candidate)
return stat.isDirectory() ? candidate : path.dirname(candidate)
} catch {
return app.getPath('home')
}
}
function terminalShellEnv() {
const env = { ...process.env }
// Electron is commonly launched through `npm run dev`; do not leak npm's
// managed prefix into a user's interactive shell (nvm/proto warn loudly).
for (const key of Object.keys(env)) {
if (key === 'npm_config_prefix' || key.startsWith('npm_config_') || key.startsWith('npm_package_')) {
delete env[key]
}
}
// Strip color/theme-detection vars that ride along when Electron is launched
// from a non-tty agent shell (Cursor's runner sets NO_COLOR/FORCE_COLOR=0
// /TERM=dumb; some terminals set COLORFGBG which would flip Hermes' TUI into
// light-mode). Our PTY is a real xterm-compat terminal — force truecolor.
delete env.NO_COLOR
delete env.FORCE_COLOR
delete env.COLORFGBG
env.COLORTERM = 'truecolor'
env.LC_CTYPE = env.LC_CTYPE || 'UTF-8'
env.TERM = 'xterm-256color'
env.TERM_PROGRAM = 'Hermes'
env.TERM_PROGRAM_VERSION = app.getVersion()
// Let a hermes/--tui launched in this pane know it's embedded in the desktop
// GUI (build_environment_hints surfaces this). Distinct from HERMES_DESKTOP,
// which marks the agent *backend* and gates cron/gateway behavior.
env.HERMES_DESKTOP_TERMINAL = '1'
return env
}
function terminalChannel(id, suffix) {
return `hermes:terminal:${id}:${suffix}`
}
// Best-effort read of a live PTY child's current working directory so a
// reopened tab can restart the shell where the user last `cd`'d, instead of the
// tab's original launch dir. Shell-agnostic (no prompt/OSC config needed) on
// POSIX; Windows has no cheap per-process cwd query without a native module, so
// it returns null and the caller falls back to the launch cwd.
function readProcessCwd(pid) {
return new Promise(resolve => {
if (!Number.isInteger(pid) || pid <= 0) {
resolve(null)
return
}
if (process.platform === 'linux') {
fs.promises
.readlink(`/proc/${pid}/cwd`)
.then(target => resolve(target || null))
.catch(() => resolve(null))
return
}
if (process.platform === 'darwin') {
// lsof ships with macOS; -Fn emits the cwd fd's path on an `n<path>` line.
execFile('lsof', ['-a', '-p', String(pid), '-d', 'cwd', '-Fn'], { timeout: 2000 }, (err, stdout) => {
if (err) {
resolve(null)
return
}
const line = String(stdout || '')
.split('\n')
.find(entry => entry.startsWith('n'))
resolve(line ? line.slice(1) : null)
})
return
}
resolve(null)
})
}
function disposeTerminalSession(id: string) {
const sessionInfo = terminalSessions.get(id)
if (!sessionInfo) {
return false
}
terminalSessions.delete(id)
try {
sessionInfo.pty.kill()
} catch {
// Process may already be gone.
}
return true
}
// SSH teardown: close every pane whose PTY rode the disconnected tunnel.
function disposeTerminalSessionsForSshScope(scope: string) {
for (const [id, info] of [...terminalSessions.entries()]) {
if (info.sshScope === scope) {
disposeTerminalSession(id)
}
}
}
// App shutdown: kill every open PTY before environment teardown.
function disposeAllTerminalSessions() {
for (const id of [...terminalSessions.keys()]) {
disposeTerminalSession(id)
}
}
// node-pty's published tarball ships the POSIX `spawn-helper` without an exec
// bit; the dev flow resolves node-pty straight from node_modules (nothing
// chmods it there), so the first terminal spawn dies with `posix_spawnp
// failed`. Restore the bit once, lazily, right before the first spawn. Packaged
// builds already stage an executable copy, so this is a no-op there.
let _spawnHelperEnsured = false
function ensureNodePtySpawnHelper() {
if (_spawnHelperEnsured || isWindows) {
return
}
_spawnHelperEnsured = true
try {
const nodePtyRoot = path.dirname(require.resolve('node-pty/package.json'))
const { fixed, errors } = ensureSpawnHelperExecutable(nodePtyRoot)
for (const helperPath of fixed) {
rememberLog(`[terminal] restored +x on node-pty spawn-helper: ${helperPath}`)
}
for (const failure of errors) {
rememberLog(`[terminal] could not chmod spawn-helper ${failure.path}: ${failure.error}`)
}
} catch (error) {
rememberLog(
`[terminal] spawn-helper exec check skipped: ${error instanceof Error ? error.message : String(error)}`
)
}
}
ipcMain.handle('hermes:terminal:start', async (event, payload = {}) => {
ensureNodePtySpawnHelper()
const id = crypto.randomUUID()
const { args, command, name } = terminalShellCommand()
const cwd = safeTerminalCwd(payload?.cwd)
const cols = Math.max(2, Number.parseInt(String(payload?.cols || 80), 10) || 80)
const rows = Math.max(2, Number.parseInt(String(payload?.rows || 24), 10) || 24)
const sshTarget = await resolveTerminalConnection(activeSshTerminalTarget, ensureBackend)
const remote = Boolean(sshTarget)
const remoteState = remote ? getSshConnectionState(sshTarget.scope) : null
const remoteCommand =
remoteState?.remotePlatform === 'Windows'
? buildWindowsInteractiveCommand(String(payload?.cwd || '').trim())
: undefined
const ptyProcess = remote
? nodePty.spawn(
process.platform === 'win32'
? path.join(process.env.SystemRoot || 'C:\\Windows', 'System32', 'OpenSSH', 'ssh.exe')
: 'ssh',
buildInteractiveSshArgs(sshTarget.ssh, String(payload?.cwd || '').trim(), undefined, remoteCommand),
{ cols, cwd: app.getPath('home'), env: terminalShellEnv(), name: 'xterm-256color', rows }
)
: nodePty.spawn(command, args, { cols, cwd, env: terminalShellEnv(), name: 'xterm-256color', rows })
terminalSessions.set(id, {
pty: ptyProcess,
webContentsId: event.sender.id,
...(remote ? { sshScope: sshTarget.scope, remoteCwd: String(payload?.cwd || '') } : {})
})
const send = (suffix, payload) => {
if (event.sender.isDestroyed()) {
return
}
event.sender.send(terminalChannel(id, suffix), payload)
}
ptyProcess.onData(data => send('data', data))
ptyProcess.onExit(({ exitCode, signal }) => {
terminalSessions.delete(id)
send('exit', { code: exitCode, signal: signal || null })
})
event.sender.once('destroyed', () => disposeTerminalSession(id))
return { cwd: remote ? null : cwd, id, shell: remote ? 'ssh' : name }
})
ipcMain.handle('hermes:terminal:write', (_event, id, data) => {
const sessionInfo = terminalSessions.get(String(id || ''))
if (!sessionInfo) {
return false
}
sessionInfo.pty.write(String(data || ''))
return true
})
ipcMain.handle('hermes:terminal:resize', (_event, id, size = {}) => {
const sessionInfo = terminalSessions.get(String(id || ''))
if (!sessionInfo) {
return false
}
const cols = Math.max(2, Number.parseInt(String(size?.cols || 80), 10) || 80)
const rows = Math.max(2, Number.parseInt(String(size?.rows || 24), 10) || 24)
sessionInfo.pty.resize(cols, rows)
return true
})
ipcMain.handle('hermes:terminal:cwd', async (_event, id) => {
const sessionInfo = terminalSessions.get(String(id || ''))
if (!sessionInfo) {
return null
}
return sessionInfo.sshScope !== undefined ? null : readProcessCwd(sessionInfo.pty.pid)
})
ipcMain.handle('hermes:terminal:dispose', (_event, id) => disposeTerminalSession(String(id || '')))
return { disposeTerminalSession, disposeTerminalSessionsForSshScope, disposeAllTerminalSessions }
}
+349 -12
View File
@@ -11,26 +11,39 @@
import { describe, expect, it } from 'vitest'
import {
backgroundMaterialFor,
clampIntensity,
DEFAULT_GLASS_MATERIAL,
DEFAULT_GLASS_SCOPE,
defaultTranslucencyState,
defaultTranslucencyValues,
GLASS_MATERIALS,
GLASS_SCOPES,
glassActive,
type GlassMaterial,
glassMaterialForPicker,
glassMaterialsFor,
glassSupportedOn,
glassSurfaceKeep,
hudFrostFor,
normalizeBook,
normalizeMaterial,
normalizeMode,
normalizeScope,
normalizeState,
resolveTranslucency,
setTranslucencyValues,
TRANSLUCENCY_CURVE,
TRANSLUCENCY_MAX,
TRANSLUCENCY_MIN,
TRANSLUCENCY_OPACITY_FLOOR,
type TranslucencyState,
translucencySupportedOn,
vibrancyFor,
windowBackingOptions,
windowOpacityFor
windowOpacityFor,
WINDOWS_BACKGROUND_MATERIALS,
WINDOWS_GLASS_MIN_BUILD
} from './translucency'
/** The linear ramp the curve replaced. Endpoints must still agree with it. */
@@ -38,13 +51,15 @@ const legacyOpacity = (intensity: number) => 1 - (intensity / 100) * 0.7
const clear = (intensity: number): TranslucencyState => ({
intensity,
fade: 0,
mode: 'clear',
material: DEFAULT_GLASS_MATERIAL,
scope: DEFAULT_GLASS_SCOPE
})
const glass = (intensity: number, material: GlassMaterial = DEFAULT_GLASS_MATERIAL): TranslucencyState => ({
const glass = (intensity: number, material: GlassMaterial = DEFAULT_GLASS_MATERIAL, fade = 0): TranslucencyState => ({
intensity,
fade,
mode: 'glass',
material,
scope: DEFAULT_GLASS_SCOPE
@@ -78,7 +93,7 @@ describe('clampIntensity', () => {
})
describe('normalizeMode', () => {
it('accepts glass on macOS only — there is no vibrancy to ride elsewhere', () => {
it('accepts glass only on a platform that has a native material', () => {
expect(normalizeMode('glass', true)).toBe('glass')
expect(normalizeMode('glass', false)).toBe('clear')
})
@@ -90,7 +105,7 @@ describe('normalizeMode', () => {
// Glass is pre-selected so the better half of the feature is the one you
// find, which is free because the intensity still starts at 0.
it('pre-selects glass on macOS when nothing is recorded', () => {
it('pre-selects glass when the platform supports it and nothing is recorded', () => {
expect(normalizeMode(undefined, true)).toBe('glass')
expect(normalizeMode('acrylic', true)).toBe('glass')
expect(normalizeMode(42, true)).toBe('glass')
@@ -163,11 +178,23 @@ describe('windowOpacityFor', () => {
expect(windowOpacityFor(clear(240))).toBe(windowOpacityFor(clear(TRANSLUCENCY_MAX)))
})
it('never fades the native window in glass mode — the renderer paints that effect', () => {
// The tint is painted by the renderer, so the intensity lever must never
// reach setOpacity under glass — that separation is what keeps text sharp.
it('ignores the intensity lever entirely in glass mode', () => {
expect(windowOpacityFor(glass(0))).toBe(1)
expect(windowOpacityFor(glass(60))).toBe(1)
expect(windowOpacityFor(glass(100))).toBe(1)
})
it('fades a glass window only through its own lever, on the ramp clear uses', () => {
expect(windowOpacityFor(glass(60, DEFAULT_GLASS_MATERIAL, 0))).toBe(1)
expect(windowOpacityFor(glass(60, DEFAULT_GLASS_MATERIAL, 40))).toBe(windowOpacityFor(clear(40)))
expect(windowOpacityFor(glass(100, DEFAULT_GLASS_MATERIAL, 100))).toBe(windowOpacityFor(clear(100)))
})
it('leaves fade inert under clear, where the intensity lever already is the opacity', () => {
expect(windowOpacityFor({ ...clear(40), fade: 100 })).toBe(windowOpacityFor(clear(40)))
})
})
describe('glassSurfaceKeep', () => {
@@ -224,10 +251,172 @@ describe('vibrancyFor', () => {
})
})
// The HUD is a transparent window, so its frost has no opaque page to hide
// behind: every state that isn't "frost wanted" has to resolve to no material
// at all, or the band leaves a grey slab hanging over another app.
describe('hudFrostFor', () => {
it('wears the chosen frost on both platforms while the band is showing', () => {
expect(hudFrostFor(glass(60, 'header'), true)).toEqual({ vibrancy: 'header', backgroundMaterial: 'mica' })
expect(hudFrostFor(glass(60, 'under-window'), true)).toEqual({
vibrancy: 'under-window',
backgroundMaterial: 'acrylic'
})
})
// The material is the whole window rectangle and nothing on the page can
// clip it, so a hidden band must mean no frost — this is the veto that keeps
// idle HUD mode the bar and nothing else.
it('is off whenever the band is not covering the window', () => {
expect(hudFrostFor(glass(60, 'header'), false)).toEqual({ vibrancy: null, backgroundMaterial: 'none' })
})
// ...and the setting is the other veto: Glass off, or the tint at zero,
// means the HUD never frosts however engaged the band is.
it('is off whenever glass itself is off', () => {
expect(hudFrostFor(clear(60), true)).toEqual({ vibrancy: null, backgroundMaterial: 'none' })
expect(hudFrostFor(glass(0, 'header'), true)).toEqual({ vibrancy: null, backgroundMaterial: 'none' })
})
// Unlike a chat window, which keeps 'sidebar' under its titlebar band in
// every non-glass state. Pinning this is what stops someone "fixing" the
// null into a resting material and painting the slab back.
it('resolves off to no material at all, not to a resting one', () => {
expect(hudFrostFor(clear(60), true).vibrancy).toBeNull()
expect(vibrancyFor(clear(60))).toBe('sidebar')
})
// The tint is painted by the renderer, exactly as it is for a chat window —
// dragging it must not re-issue setVibrancy, whose 150ms animation restarts
// on every call and never lets the material settle.
it('does not move any native property as the tint slider is dragged', () => {
for (let intensity = 1; intensity <= 100; intensity += 1) {
expect(hudFrostFor(glass(intensity, 'popover'), true)).toEqual({
vibrancy: 'popover',
backgroundMaterial: 'tabbed'
})
}
})
})
describe('glassSupportedOn', () => {
it('is on for macOS regardless of kernel version', () => {
expect(glassSupportedOn('darwin')).toBe(true)
expect(glassSupportedOn('darwin', '24.6.0')).toBe(true)
})
it('is on for Windows 11 22H2 and newer, off for everything older', () => {
expect(glassSupportedOn('win32', `10.0.${WINDOWS_GLASS_MIN_BUILD}`)).toBe(true)
expect(glassSupportedOn('win32', `10.0.${WINDOWS_GLASS_MIN_BUILD}.1`)).toBe(true)
expect(glassSupportedOn('win32', `10.0.${WINDOWS_GLASS_MIN_BUILD - 1}`)).toBe(false)
expect(glassSupportedOn('win32', '10.0.19045')).toBe(false)
expect(glassSupportedOn('win32', '10.0')).toBe(false)
expect(glassSupportedOn('win32', '')).toBe(false)
})
it('is off on Linux — Electron has no first-party desktop material there', () => {
expect(glassSupportedOn('linux', '6.8.0')).toBe(false)
})
})
describe('backgroundMaterialFor', () => {
it('is none while glass is off so DWM does not keep drawing under the backing', () => {
expect(backgroundMaterialFor(glass(0, 'header'))).toBe('none')
expect(backgroundMaterialFor(clear(60))).toBe('none')
})
it('maps the sheer → heavy frost ladder onto acrylic / tabbed / mica', () => {
expect(backgroundMaterialFor(glass(60, 'under-window'))).toBe('acrylic')
expect(backgroundMaterialFor(glass(60, 'popover'))).toBe('tabbed')
expect(backgroundMaterialFor(glass(60, 'titlebar'))).toBe('mica')
})
// Windows 11 has three system materials for four rungs, so the two heaviest
// land on mica. The mapping stays total — a saved 'header' still resolves —
// and the picker drops the duplicate instead (see glassMaterialsFor).
it('collapses Glare onto mica with Bright', () => {
expect(backgroundMaterialFor(glass(60, 'header'))).toBe('mica')
expect(backgroundMaterialFor(glass(60, 'header'))).toBe(backgroundMaterialFor(glass(60, 'titlebar')))
})
it('resolves every shipped rung to a real system material', () => {
for (const material of GLASS_MATERIALS) {
expect(WINDOWS_BACKGROUND_MATERIALS, material).toContain(backgroundMaterialFor(glass(60, material)))
}
})
})
describe('translucencySupportedOn', () => {
it('covers the two platforms where setOpacity or a native material exists', () => {
expect(translucencySupportedOn('darwin')).toBe(true)
expect(translucencySupportedOn('win32')).toBe(true)
})
// Electron documents setOpacity as doing nothing on Linux, and there is no
// material either — so the setting has no working half to offer there.
it('is off on Linux, where neither mode does anything', () => {
expect(translucencySupportedOn('linux')).toBe(false)
expect(translucencySupportedOn('freebsd')).toBe(false)
})
// Win10 loses glass but keeps clear, so the row must survive there.
it('stays on for a Windows build too old for glass', () => {
const oldWindows = `10.0.${WINDOWS_GLASS_MIN_BUILD - 1}`
expect(glassSupportedOn('win32', oldWindows)).toBe(false)
expect(translucencySupportedOn('win32')).toBe(true)
})
})
describe('the frost rungs a platform offers', () => {
it('offers the whole ladder on macOS', () => {
expect(glassMaterialsFor(false)).toEqual(GLASS_MATERIALS)
})
// The census rule, now enforced on Windows too: no two options in the picker
// may composite to the same thing. Bright and Glare are both mica.
it('never offers two rungs that render the same Windows backdrop', () => {
const backdrops = glassMaterialsFor(true).map(material => backgroundMaterialFor(glass(60, material)))
expect(new Set(backdrops).size).toBe(backdrops.length)
expect(glassMaterialsFor(true).length).toBeLessThan(GLASS_MATERIALS.length)
})
it('keeps every distinct Windows backdrop reachable from the picker', () => {
const backdrops = new Set(glassMaterialsFor(true).map(material => backgroundMaterialFor(glass(60, material))))
const reachable = new Set(GLASS_MATERIALS.map(material => backgroundMaterialFor(glass(60, material))))
expect(backdrops).toEqual(reachable)
})
// Settings synced from a Mac carry a rung Windows has no button for. The
// picker highlights the button that renders the same backdrop rather than
// showing nothing selected — and does NOT rewrite what the Mac saved.
it('folds a dropped rung onto the button that looks the same', () => {
expect(glassMaterialForPicker('header', true)).toBe('titlebar')
expect(glassMaterialsFor(true)).toContain(glassMaterialForPicker('header', true))
expect(backgroundMaterialFor(glass(60, glassMaterialForPicker('header', true)))).toBe(
backgroundMaterialFor(glass(60, 'header'))
)
})
it('leaves every rung alone on macOS and every offered rung alone on Windows', () => {
for (const material of GLASS_MATERIALS) {
expect(glassMaterialForPicker(material, false)).toBe(material)
}
for (const material of glassMaterialsFor(true)) {
expect(glassMaterialForPicker(material, true)).toBe(material)
}
})
})
describe('normalizeState', () => {
it('parses a modern payload', () => {
expect(normalizeState({ intensity: 40, mode: 'glass', material: 'header', scope: 'sidebar' }, true)).toEqual({
expect(
normalizeState({ intensity: 40, fade: 15, mode: 'glass', material: 'header', scope: 'sidebar' }, true)
).toEqual({
intensity: 40,
fade: 15,
mode: 'glass',
material: 'header',
scope: 'sidebar'
@@ -239,19 +428,28 @@ describe('normalizeState', () => {
it('keeps a legacy intensity-only payload on clear', () => {
expect(normalizeState({ intensity: 70 }, true)).toEqual({
intensity: 70,
fade: 0,
mode: 'clear',
material: DEFAULT_GLASS_MATERIAL,
scope: DEFAULT_GLASS_SCOPE
})
})
it('survives junk payloads', () => {
const base = { intensity: 0, material: DEFAULT_GLASS_MATERIAL, scope: DEFAULT_GLASS_SCOPE }
// Fade arrived after glass shipped, so a profile written by the older build
// has no key for it and must come back unfaded rather than undefined.
it('defaults a payload written before fade existed to no fade', () => {
expect(normalizeState({ intensity: 60, mode: 'glass' }, true).fade).toBe(0)
})
// A fresh macOS profile lands on glass at zero intensity: selected, but off.
it('survives junk payloads', () => {
const base = { intensity: 0, fade: 0, material: DEFAULT_GLASS_MATERIAL, scope: DEFAULT_GLASS_SCOPE }
// A fresh glass-capable profile lands on glass at zero intensity: selected, but off.
expect(normalizeState(null, true)).toEqual({ ...base, mode: 'glass' })
expect(normalizeState('nope', true)).toEqual({ ...base, mode: 'glass' })
expect(normalizeState({ intensity: 'x', material: 'nope', mode: 'glass', scope: 'nope' }, false)).toEqual({
expect(
normalizeState({ intensity: 'x', fade: 'x', material: 'nope', mode: 'glass', scope: 'nope' }, false)
).toEqual({
...base,
mode: 'clear'
})
@@ -266,8 +464,8 @@ describe('glassActive', () => {
})
})
// The default must be selected-but-off: a fresh macOS profile shows Glass in
// the picker while the window itself is untouched until the lever moves.
// The default must be selected-but-off: a fresh glass-capable profile shows
// Glass in the picker while the window itself is untouched until the lever moves.
describe('a fresh profile', () => {
const fresh = normalizeState(null, true)
@@ -326,12 +524,22 @@ describe('what an update actually changes natively', () => {
expect(nativeDiff(clear(40), clear(41))).toEqual({ backing: false, material: false, opacity: true })
})
// The one glass drag that reaches main, and it costs what a clear drag costs.
it('is only the opacity while dragging fade under glass', () => {
expect(nativeDiff(glass(60, DEFAULT_GLASS_MATERIAL, 40), glass(60, DEFAULT_GLASS_MATERIAL, 41))).toEqual({
backing: false,
material: false,
opacity: true
})
})
it('is the material alone when the frost level changes', () => {
expect(nativeDiff(glass(60, 'under-window'), glass(60, 'header'))).toEqual({
backing: false,
material: true,
opacity: false
})
expect(backgroundMaterialFor(glass(60, 'under-window'))).not.toBe(backgroundMaterialFor(glass(60, 'header')))
})
// Crossing zero flips glass on/off, which is exactly when the backing has to
@@ -344,4 +552,133 @@ describe('what an update actually changes natively', () => {
it('is everything when switching between the two modes', () => {
expect(nativeDiff(clear(60), glass(60))).toEqual({ backing: true, material: true, opacity: true })
})
it('leaves a window alone when glass is selected but off', () => {
// The light default carries one point of fade. Someone who dragged the
// tint to zero asked for an opaque window, and that point must not follow
// them there — off has to mean exactly 1, not 0.9999.
expect(windowOpacityFor({ ...glass(0), fade: 1 })).toBe(1)
expect(windowOpacityFor({ ...glass(0), fade: 40 })).toBe(1)
})
it('still fades a window whose glass is actually on', () => {
expect(windowOpacityFor({ ...glass(66), fade: 40 })).toBeLessThan(1)
})
})
/**
* The shipped defaults, per platform. These are the numbers a fresh profile
* gets before anyone opens Settings, so they are the ones most people will
* ever see — and they differ by platform because the lever means different
* things behind macOS vibrancy and Windows acrylic.
*/
describe('the defaults a fresh profile lands on', () => {
const mac = (appearance: 'dark' | 'light') => defaultTranslucencyValues(appearance, false)
const win = (appearance: 'dark' | 'light') => defaultTranslucencyValues(appearance, true)
it('ships glass on, not a lever resting at zero', () => {
for (const values of [mac('light'), mac('dark'), win('light'), win('dark')]) {
expect(values.intensity).toBeGreaterThan(0)
expect(glassActive({ ...values, mode: 'glass' })).toBe(true)
}
for (const appearance of ['light', 'dark'] as const) {
expect(defaultTranslucencyState(appearance, true, false).mode).toBe('glass')
expect(defaultTranslucencyState(appearance, true, true).mode).toBe('glass')
}
})
it('falls back to clear where no native material exists', () => {
expect(defaultTranslucencyState('dark', false, false).mode).toBe('clear')
})
it('tints light more heavily than dark, on both platforms', () => {
// A dark field already separates from what is behind it; a bright one
// needs real thinning before the desktop reads as a layer underneath.
expect(mac('light').intensity).toBeGreaterThan(mac('dark').intensity)
expect(win('light').intensity).toBeGreaterThan(win('dark').intensity)
})
it('asks far less of Windows, which composites its own tint in DWM', () => {
expect(win('light').intensity).toBeLessThan(mac('light').intensity)
expect(win('dark').intensity).toBeLessThan(mac('dark').intensity)
})
it('never fades a Windows window — setOpacity dims the composited backdrop', () => {
expect(win('light').fade).toBe(0)
expect(win('dark').fade).toBe(0)
})
it('defaults each platform onto a frost that platform can actually render', () => {
for (const appearance of ['light', 'dark'] as const) {
expect(glassMaterialsFor(true)).toContain(win(appearance).material)
expect(glassMaterialsFor(false)).toContain(mac(appearance).material)
}
})
it('opens the whole window, not just the sidebar rail', () => {
for (const values of [mac('light'), mac('dark'), win('light'), win('dark')]) {
expect(values.scope).toBe('window')
}
})
})
/**
* The per-appearance ladder: appearance slot → base → platform default, per
* key. This is what makes tuning light mode stay in light mode while an
* untouched dark keeps inheriting.
*/
describe('resolving the book for the painted appearance', () => {
const empty = normalizeBook(null, true)
it('falls all the way through to the platform default', () => {
expect(resolveTranslucency(empty, 'dark', false).intensity).toBe(defaultTranslucencyValues('dark', false).intensity)
expect(resolveTranslucency(empty, 'dark', true).intensity).toBe(defaultTranslucencyValues('dark', true).intensity)
})
it('scopes an edit to the appearance it was made in', () => {
const book = setTranslucencyValues(empty, 'light', { intensity: 90 })
expect(resolveTranslucency(book, 'light', false).intensity).toBe(90)
expect(resolveTranslucency(book, 'dark', false).intensity).toBe(defaultTranslucencyValues('dark', false).intensity)
})
it('carries a v1 state into BOTH appearances via base', () => {
// Someone who tuned a window before appearances were split keeps exactly
// what was on screen, in either appearance, until they edit one of them.
const migrated = normalizeBook({ intensity: 40, mode: 'glass' }, true)
expect(migrated.base.intensity).toBe(40)
expect(resolveTranslucency(migrated, 'light', false).intensity).toBe(40)
expect(resolveTranslucency(migrated, 'dark', false).intensity).toBe(40)
})
it('lets an appearance override base without disturbing the other', () => {
const tuned = setTranslucencyValues(normalizeBook({ intensity: 40, mode: 'glass' }, true), 'dark', {
intensity: 10
})
expect(resolveTranslucency(tuned, 'dark', false).intensity).toBe(10)
expect(resolveTranslucency(tuned, 'light', false).intensity).toBe(40)
})
it('inherits per KEY, not per appearance', () => {
// Editing only the tint in dark must leave dark's material still tracking
// base — a partial edit is not a full snapshot of the appearance.
const book = setTranslucencyValues(normalizeBook({ material: 'popover', mode: 'glass' }, true), 'dark', {
intensity: 33
})
const resolved = resolveTranslucency(book, 'dark', false)
expect(resolved.intensity).toBe(33)
expect(resolved.material).toBe('popover')
})
it('keeps mode global — clear vs glass is about the window, not the palette', () => {
const book = setTranslucencyValues({ ...empty, mode: 'clear' }, 'light', { intensity: 50 })
expect(resolveTranslucency(book, 'light', false).mode).toBe('clear')
expect(resolveTranslucency(book, 'dark', false).mode).toBe('clear')
})
})
+15 -1
View File
@@ -14,25 +14,39 @@
import { glassActive, type TranslucencyState } from '../../shared/src/translucency'
export {
backgroundMaterialFor,
clampIntensity,
DEFAULT_GLASS_MATERIAL,
DEFAULT_GLASS_SCOPE,
defaultTranslucencyState,
defaultTranslucencyValues,
GLASS_MATERIALS,
GLASS_SCOPES,
glassActive,
type GlassMaterial,
glassMaterialForPicker,
glassMaterialsFor,
glassSupportedOn,
glassSurfaceKeep,
hudFrostFor,
normalizeBook,
normalizeMaterial,
normalizeMode,
normalizeScope,
normalizeState,
resolveTranslucency,
setTranslucencyValues,
TRANSLUCENCY_CURVE,
TRANSLUCENCY_MAX,
TRANSLUCENCY_MIN,
TRANSLUCENCY_OPACITY_FLOOR,
type TranslucencyState,
translucencySupportedOn,
vibrancyFor,
windowOpacityFor
windowOpacityFor,
WINDOWS_BACKGROUND_MATERIALS,
WINDOWS_GLASS_MIN_BUILD,
type WindowsBackgroundMaterial
} from '../../shared/src/translucency'
/**
@@ -5,9 +5,9 @@
// 1. buildPathExtCandidates() — PATHEXT extensions must be tried BEFORE the
// empty extension, or an extensionless Git-Bash `hermes` shim shadows
// the real hermes.cmd/hermes.exe.
// 2. chooseUpdaterArgs() — must gate on haveRealInstall (any real-install
// signal), not just the hermes.exe console-script shim, or healthy
// installs get forced into a destructive --repair.
// 2. chooseUpdaterArgs() — must distinguish a runnable updater from stale
// install provenance. The bootstrap marker can outlive the venv, and a
// partial venv cannot run the updater; those states require --repair.
// 3. resolveVenvHermesCommand() — must probe the venv python via
// canImportHermesCli() before trusting it, or a broken venv gets
// re-selected forever instead of falling through to bootstrap.
@@ -45,17 +45,43 @@ test('buildPathExtCandidates: non-Windows only tries the bare name', () => {
assert.deepEqual(buildPathExtCandidates(undefined, false), [''])
})
test('chooseUpdaterArgs: gentle --update when a real-install signal is present', () => {
assert.deepEqual(chooseUpdaterArgs(true, 'main'), ['--update', '--branch', 'main'])
test('chooseUpdaterArgs: gentle --update when both updater runtime files exist', () => {
assert.deepEqual(chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: true, hasVenvPython: true }, 'main'), [
'--update',
'--branch',
'main'
])
})
test('chooseUpdaterArgs: destructive --repair only when NO real-install signal is present', () => {
assert.deepEqual(chooseUpdaterArgs(false, 'main'), ['--repair', '--branch', 'main'])
test('chooseUpdaterArgs: marker-only install uses --repair when the venv is gone', () => {
assert.deepEqual(
chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: false, hasVenvPython: false }, 'main'),
['--repair', '--branch', 'main']
)
})
test('chooseUpdaterArgs: passes the branch through unchanged in both cases', () => {
assert.deepEqual(chooseUpdaterArgs(true, 'release/1.2'), ['--update', '--branch', 'release/1.2'])
assert.deepEqual(chooseUpdaterArgs(false, 'release/1.2'), ['--repair', '--branch', 'release/1.2'])
test('chooseUpdaterArgs: partial updater runtimes use --repair', () => {
assert.deepEqual(chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: false, hasVenvPython: true }, 'main'), [
'--repair',
'--branch',
'main'
])
assert.deepEqual(chooseUpdaterArgs({ hasBootstrapMarker: true, hasVenvHermes: true, hasVenvPython: false }, 'main'), [
'--repair',
'--branch',
'main'
])
})
test('chooseUpdaterArgs: passes the branch through unchanged in both modes', () => {
assert.deepEqual(
chooseUpdaterArgs({ hasBootstrapMarker: false, hasVenvHermes: true, hasVenvPython: true }, 'release/1.2'),
['--update', '--branch', 'release/1.2']
)
assert.deepEqual(
chooseUpdaterArgs({ hasBootstrapMarker: false, hasVenvHermes: false, hasVenvPython: false }, 'release/1.2'),
['--repair', '--branch', 'release/1.2']
)
})
function makeDeps(overrides: Partial<Parameters<typeof resolveVenvHermesCommand>[2]> = {}) {
+21 -19
View File
@@ -11,12 +11,11 @@
* hermes.cmd/hermes.exe; the shim then failed the --version probe and
* the desktop fell through to a spurious bootstrap/repair. The fix:
* PATHEXT extensions first, empty extension LAST.
* 2. chooseUpdaterArgs() — handOffWindowsBootstrapRecovery() chose
* --update vs the destructive --repair by checking ONLY
* venv\Scripts\hermes.exe (the console-script shim, written at the END
* of venv setup and absent in interrupted states), so it escalated to a
* full venv recreate even on healthy installs. The fix: gate on ANY
* real-install signal, not just the shim.
* 2. chooseUpdaterArgs() — handOffWindowsBootstrapRecovery() must separate
* install provenance from updater viability. A bootstrap-complete marker
* can outlive a deleted venv, while the updater needs BOTH the venv Python
* and Hermes launcher. Marker-only or partial runtimes must use --repair;
* only a runnable pair can use --update.
* 3. resolveVenvHermesCommand() — unwrapWindowsVenvHermesCommand() returned
* the venv python with NO runtime probe (bypassing the caller's
* --version check too), so a venv broken mid-update (e.g. missing
@@ -61,23 +60,26 @@ export function buildPathExtCandidates(pathext: string | undefined, isWindows: b
}
/**
* Choose the Windows bootstrap-recovery updater invocation: the gentle
* in-place --update when ANY real-install signal is present, the
* destructive --repair (full venv recreate) otherwise.
* Choose the Windows bootstrap-recovery invocation. The gentle in-place
* updater can only start when both pieces of its runtime contract exist: the
* venv Python interpreter and the Hermes launcher that drives `hermes update`.
* A bootstrap-complete marker proves install provenance, not current runtime
* usability, and may remain after the venv is removed or quarantined.
*
* haveRealInstall must be computed by the caller from ALL real-install
* signals (venv python interpreter, venv hermes shim, bootstrap-complete
* marker) — gating on just the hermes.exe console-script shim alone is the
* regression this function's callers must avoid: that shim is written at
* the END of venv setup and is absent in exactly the interrupted/quarantined
* states this recovery exists to heal.
*
* @param {boolean} haveRealInstall
* @param {BootstrapRecoverySignals} signals
* @param {string} branch
* @returns {string[]} updater argv, e.g. ['--update', '--branch', 'main'].
*/
export function chooseUpdaterArgs(haveRealInstall: boolean, branch: string): string[] {
return haveRealInstall ? ['--update', '--branch', branch] : ['--repair', '--branch', branch]
export interface BootstrapRecoverySignals {
hasBootstrapMarker: boolean
hasVenvHermes: boolean
hasVenvPython: boolean
}
export function chooseUpdaterArgs(signals: BootstrapRecoverySignals, branch: string): string[] {
const canRunUpdater = signals.hasVenvHermes && signals.hasVenvPython
return canRunUpdater ? ['--update', '--branch', branch] : ['--repair', '--branch', branch]
}
/**
@@ -1,6 +1,6 @@
import crypto from 'node:crypto'
import { redactSecrets, SSH_ERROR } from './ssh-connection'
import { assertBootstrapNotSuperseded, redactSecrets, SSH_ERROR } from './ssh-connection'
const LOCKFILE_SCHEMA_VERSION = 2
const PROTOCOL_VERSION = 1
@@ -162,14 +162,6 @@ function reusableWindowsLock(lock, state, profile, reuseToken, runtime) {
)
}
function assertCurrent(signal) {
if (signal?.aborted) {
const error: any = new Error('SSH bootstrap was cancelled.')
error.kind = 'superseded'
throw error
}
}
async function processState(ssh, runtime, lock) {
return helper(ssh, runtime, 'process-state', [
String(lock.pid),
@@ -215,7 +207,7 @@ async function waitReady(ssh, runtime, ownershipId, lock, timeoutMs, signal) {
const deadline = Date.now() + timeoutMs
while (Date.now() < deadline) {
assertCurrent(signal)
assertBootstrapNotSuperseded(signal)
let state
try {
@@ -286,7 +278,7 @@ async function connectWindowsRemote(deps) {
readyTimeoutMs = 45_000
} = deps
assertCurrent(signal)
assertBootstrapNotSuperseded(signal)
const runtime = await probeWindowsRemote(ssh, remoteHermesPath)
const inspection = await helper(ssh, runtime, 'inspect', [runtime.hermesPath])
@@ -356,7 +348,7 @@ async function connectWindowsRemote(deps) {
await helper(ssh, runtime, 'remove-lock', [ownershipId])
}
assertCurrent(signal)
assertBootstrapNotSuperseded(signal)
const token = crypto.randomBytes(32).toString('hex')
const spawnNonce = crypto.randomBytes(8).toString('hex')
await helper(ssh, runtime, 'upload-token', [ownershipId, spawnNonce], token)
@@ -405,7 +397,7 @@ async function connectWindowsRemote(deps) {
await forward(localPort, remotePort)
const baseUrl = `http://127.0.0.1:${localPort}`
await waitForHermes(baseUrl, token)
assertCurrent(signal)
assertBootstrapNotSuperseded(signal)
await helper(ssh, runtime, 'write-lock', [ownershipId], JSON.stringify({ ...owned, port: remotePort }))
return {
+3 -2
View File
@@ -113,13 +113,14 @@
"@xterm/addon-web-links": "0.12.0",
"@xterm/addon-webgl": "0.19.0",
"@xterm/xterm": "6.0.0",
"blobatar": "0.2.0",
"blobatar": "2.0.0",
"class-variance-authority": "0.7.1",
"clsx": "2.1.1",
"cmdk": "1.1.1",
"d3-force": "3.0.0",
"dnd-core": "14.0.1",
"dompurify": "3.4.13",
"driver.js": "1.8.0",
"emojibase-data": "16.0.3",
"fflate": "0.8.3",
"frimousse": "0.3.0",
@@ -129,7 +130,7 @@
"katex": "0.16.47",
"mermaid": "11.16.1",
"motion": "12.42.2",
"nanostores": "1.4.0",
"nanostores": "1.4.2",
"node-pty": "1.1.0",
"radix-ui": "1.6.7",
"react": "19.2.7",
+46 -22
View File
@@ -437,7 +437,7 @@ const GET_WINDOWS_VERSION = '9.3.0'
export function stageGetWindowsInto(
srcRoot,
destRoot,
{ platform = process.platform, arch = process.arch, rebuild } = {}
{ platform = process.platform, arch = process.arch, install } = {}
) {
// The STAGED_WINDOWS_JS rewrite mirrors this exact version's export surface.
// A version bump must fail the build here until the rewrite is re-verified —
@@ -495,6 +495,7 @@ export function stageGetWindowsInto(
)
: []
let bindingDirs = scanBindingDirs()
let installAttempted = false
if (bindingDirs.length === 0 && arch === 'arm64') {
// get-windows 9.3.0 publishes win32 prebuilds for ia32/x64 only.
// The staged windows.js deliberately fails soft when binding/ is absent,
@@ -503,24 +504,25 @@ export function stageGetWindowsInto(
'[stage-native-deps] get-windows has no win32-arm64 prebuilt binding; ' +
'staging the fail-soft JS surface without native window enumeration.'
)
} else if (bindingDirs.length === 0 && typeof rebuild === 'function') {
} else if (bindingDirs.length === 0 && typeof install === 'function') {
// A plain `npm install` won't re-run an install script for a package
// that is already on disk, so every checkout that installed while
// get-windows was missing from allowScripts stays bricked even after
// the allowlist is fixed. `npm rebuild` re-runs it.
// the allowlist is fixed. Invoke node-pre-gyp directly: npm treats this
// optional dependency's failed lifecycle as non-fatal and can report a
// successful rebuild without producing the Windows binding.
console.log(
'[stage-native-deps] get-windows has no win32 binding; running `npm rebuild get-windows`...'
'[stage-native-deps] get-windows has no win32 binding; running its native installer...'
)
rebuild()
installAttempted = true
install()
bindingDirs = scanBindingDirs()
}
if (bindingDirs.length === 0 && arch !== 'arm64') {
throw new Error(
`[stage-native-deps] get-windows has no win32-${arch} prebuilt binding under lib/binding. ` +
'Recover from the checkout root with:\n' +
' npm install-scripts approve get-windows\n' +
' npm rebuild get-windows'
)
const reason = installAttempted
? `native installer completed without producing a win32-${arch} binding under lib/binding`
: `has no win32-${arch} prebuilt binding under lib/binding`
throw new Error(`[stage-native-deps] get-windows ${reason}`)
}
for (const dir of bindingDirs) {
const dest = join(destRoot, 'lib', 'binding', dir)
@@ -542,15 +544,35 @@ export function stageGetWindowsInto(
return destRoot
}
function rebuildGetWindowsViaNpm() {
const result = spawnSync('npm', ['rebuild', 'get-windows'], {
cwd: resolve(projectRoot, '..', '..'),
stdio: 'inherit',
// npm resolves to npm.cmd on Windows, which needs a shell.
shell: process.platform === 'win32'
export function installGetWindowsNativeBinding(
srcRoot,
{ resolveInstaller, spawn = spawnSync } = {}
) {
let installerPath
try {
const resolveNodePreGyp =
resolveInstaller ??
(() =>
require.resolve('@mapbox/node-pre-gyp/bin/node-pre-gyp', {
paths: [srcRoot]
}))
installerPath = resolveNodePreGyp()
} catch (error) {
const detail = error instanceof Error ? error.message : String(error)
throw new Error(`[stage-native-deps] cannot resolve get-windows native installer: ${detail}`)
}
const result = spawn(process.execPath, [installerPath, 'install', '--fallback-to-build'], {
cwd: srcRoot,
stdio: 'inherit'
})
if (result.error) {
throw new Error(
`[stage-native-deps] get-windows native installer could not start: ${result.error.message}`
)
}
if (result.status !== 0) {
console.warn(`[stage-native-deps] npm rebuild get-windows exited with ${result.status}`)
throw new Error(`[stage-native-deps] get-windows native installer exited with ${result.status}`)
}
}
@@ -585,10 +607,12 @@ export function stageGetWindows(
}
// Only a win32 host can produce the win32 binding, so a cross-platform pack
// has nothing to gain from the rebuild.
const rebuild =
platform === 'win32' && process.platform === 'win32' ? rebuildGetWindowsViaNpm : undefined
return stageGetWindowsInto(srcRoot, destRoot, { platform, arch, rebuild })
// has nothing to gain from the native installer.
const install =
platform === 'win32' && process.platform === 'win32'
? () => installGetWindowsNativeBinding(srcRoot)
: undefined
return stageGetWindowsInto(srcRoot, destRoot, { platform, arch, install })
}
// Allow direct CLI invocation: node scripts/stage-native-deps.mjs [platform] [arch]
@@ -6,6 +6,7 @@ import { pathToFileURL } from 'node:url'
import { test } from 'vitest'
import {
installGetWindowsNativeBinding,
stageGetWindows,
stageGetWindowsInto,
stageNodePtyInto,
@@ -460,7 +461,7 @@ test('win32-arm64 staging omits incompatible bindings and keeps the fail-soft JS
}
})
test('win32 staging self-heals through the rebuild hook when the binding is missing', () => {
test('win32 staging self-heals through the native installer when the binding is missing', () => {
const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-'))
try {
const srcRoot = join(tmp, 'get-windows')
@@ -471,7 +472,7 @@ test('win32 staging self-heals through the rebuild hook when the binding is miss
makeFakeGetWindows(srcRoot, { bindings: [] })
let calls = 0
const rebuild = () => {
const install = () => {
calls += 1
makeFakeNode(
join(srcRoot, 'lib', 'binding', 'napi-9-win32-unknown-x64', 'node-get-windows.node'),
@@ -479,7 +480,7 @@ test('win32 staging self-heals through the rebuild hook when the binding is miss
)
}
stageGetWindowsInto(srcRoot, destRoot, { platform: 'win32', arch: 'x64', rebuild })
stageGetWindowsInto(srcRoot, destRoot, { platform: 'win32', arch: 'x64', install })
assert.equal(calls, 1)
assert.ok(
@@ -490,7 +491,7 @@ test('win32 staging self-heals through the rebuild hook when the binding is miss
}
})
test('win32 staging reports the recovery steps when the rebuild hook produces nothing', () => {
test('win32 staging rejects a successful installer that produces no binding', () => {
const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-'))
try {
const srcRoot = join(tmp, 'get-windows')
@@ -503,15 +504,69 @@ test('win32 staging reports the recovery steps when the rebuild hook produces no
stageGetWindowsInto(srcRoot, destRoot, {
platform: 'win32',
arch: 'x64',
rebuild: () => {}
install: () => {}
}),
/npm rebuild get-windows/
(error) => {
assert.match(error.message, /installer completed without producing a win32-x64 binding/)
assert.doesNotMatch(error.message, /npm rebuild/)
return true
}
)
} finally {
fs.rmSync(tmp, { recursive: true, force: true })
}
})
test('get-windows native install invokes node-pre-gyp directly from the package root', () => {
const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-'))
try {
const srcRoot = join(tmp, 'get-windows')
const installer = join(
srcRoot,
'node_modules',
'@mapbox',
'node-pre-gyp',
'bin',
'node-pre-gyp'
)
fs.mkdirSync(path.dirname(installer), { recursive: true })
fs.writeFileSync(
join(srcRoot, 'node_modules', '@mapbox', 'node-pre-gyp', 'package.json'),
JSON.stringify({ name: '@mapbox/node-pre-gyp', version: '1.0.11' })
)
fs.writeFileSync(installer, '')
const calls = []
installGetWindowsNativeBinding(srcRoot, {
spawn: (command, args, options) => {
calls.push({ command, args, options })
return { status: 0 }
}
})
assert.deepEqual(calls, [
{
command: process.execPath,
args: [fs.realpathSync(installer), 'install', '--fallback-to-build'],
options: { cwd: srcRoot, stdio: 'inherit' }
}
])
} finally {
fs.rmSync(tmp, { recursive: true, force: true })
}
})
test('get-windows native install surfaces node-pre-gyp failure', () => {
assert.throws(
() =>
installGetWindowsNativeBinding('C:\\fake\\get-windows', {
resolveInstaller: () => 'C:\\fake\\node-pre-gyp',
spawn: () => ({ status: 1 })
}),
/native installer exited with 1/
)
})
test('staging refuses a get-windows version the lib/windows.js rewrite was not verified against', () => {
const tmp = fs.mkdtempSync(join(os.tmpdir(), 'hermes-stage-'))
try {
+154
View File
@@ -0,0 +1,154 @@
import { JsonRpcGatewayClient } from '@hermes/shared'
import type { HermesApiRequest } from '@/global'
// Desktop startup fires a burst of read-only data calls (config, profiles,
// model info/options, cron) the moment the backend passes readiness. On a
// profile-heavy or remote install these can each take tens of seconds — e.g.
// /api/profiles runs list_profiles(), which does a recursive skill-tree walk
// per profile — so the 15s default (DEFAULT_FETCH_TIMEOUT_MS in hardening.ts)
// times out a backend that is alive-but-busy, surfacing as a spurious
// "Timed out connecting to Hermes backend" that hangs the UI (#48504).
//
// Give the boot burst a generous per-call timeout instead of raising the
// global default: interactive/runtime calls and the liveness poll (/api/status)
// keep the short default so a genuinely-dead backend is still detected fast.
export const STARTUP_REQUEST_TIMEOUT_MS = 60_000
const DEFAULT_GATEWAY_REQUEST_TIMEOUT_MS = 30_000
// prompt.submit is effectively fire-and-forget: turn completion is signaled by
// stream / message.complete events, NOT by the RPC return. A long turn (MoA
// presets running references + aggregator in series, deep reasoning, large tool
// chains) can legitimately take minutes to ACK, so bounding the ack by the
// generic 30s default surfaces a false "request timed out" toast while the turn
// is still running and will succeed (issue #55024). Match the backend's
// agent-turn ceiling (agent.gateway_timeout = 1800s) so the ack timeout only
// ever fires when the turn itself would have been abandoned server-side.
export const PROMPT_SUBMIT_REQUEST_TIMEOUT_MS = 1_800_000
export class HermesGateway extends JsonRpcGatewayClient {
constructor() {
super({
closedErrorMessage: 'Hermes gateway connection closed',
connectErrorMessage: 'Could not connect to Hermes gateway',
createRequestId: nextId => nextId,
notConnectedErrorMessage: 'Hermes gateway is not connected',
requestTimeoutMs: DEFAULT_GATEWAY_REQUEST_TIMEOUT_MS
})
}
}
// Profile that profile-scoped REST settings (config/env/skills/tools/model/…)
// should target. Mirrors $activeGatewayProfile, pushed in from the store via
// setApiRequestProfile so this module needs no store import (avoids a cycle).
// Electron main consumes request.profile as request scope. Local calls whose
// REST handlers accept profile reuse the primary dashboard via ?profile=;
// unscoped handlers retain a profile backend. Remote overrides still route to
// their owning backend. Null → primary, so single-profile users are unaffected.
let _apiProfile: null | string = null
export function setApiRequestProfile(profile: null | string): void {
_apiProfile = profile || null
}
export function profileScoped(profile?: null | string): { profile?: string } {
const selected = profile === undefined ? _apiProfile : profile
return selected ? { profile: selected } : {}
}
/** Profile that profile-scoped REST/WS calls should target (null → primary).
* Read-only twin of setApiRequestProfile for modules (e.g. voice playback)
* that build their own connection URLs and must stay on the same backend. */
export function getApiRequestProfile(): null | string {
return _apiProfile
}
// Registry connection serving the active gateway (null → the local pool).
// Pushed from store/gateway's setActive — the single seam BOTH
// ensureGatewayProfile and ensureGatewayAgent funnel through — so WS calls
// that dial their own backend (pluginSocket) resolve it through the SAME
// source of truth those paths maintain for $connection. That makes the plugin
// socket follow registry-agent activations too, not just profile switches.
// Same no-store-import contract as _apiProfile (avoids a cycle).
let _apiConnectionId: null | string = null
export function setApiRequestConnection(connectionId: null | string): void {
_apiConnectionId = connectionId || null
}
// Registry connection scope for a REST request. A registered remote gateway
// owns its own state.db — cron jobs and their run sessions live THERE — so
// requests for gateway-owned data must carry the connection id for the main
// process to route them to that host (hermes:api's registry branch). Null
// resolves to no tag, keeping single-source users byte-identical; explicit
// 'local' must remain tagged when the legacy primary points elsewhere.
export function connectionScoped(): { connectionId?: string } {
return _apiConnectionId ? { connectionId: _apiConnectionId } : {}
}
/** Send a REST request to the renderer's active registry source. Request-level
* routing may override the active source for an explicitly-owned resource.
*
* Helpers under `api/` go through here rather than calling the preload bridge
* directly, so the connection tag cannot be forgotten on a new one — with one
* exception. A capabilityScoped() helper must NOT: that scope says "the local
* pool" by omitting `connectionId` entirely, and an absent key cannot override
* the ambient tag spread underneath it, so a 'local' pin would silently route
* to whatever remote gateway happened to be active. Those helpers call the
* bridge directly and own their routing end to end. */
export function hermesApi<T>(request: HermesApiRequest): Promise<T> {
return window.hermesDesktop.api<T>({ ...connectionScoped(), ...request })
}
// ── Capability scope: (connection, profile) routing for the Capabilities
// surface (skills / toolsets / MCP / hub / env / toolset config) ────────────
//
// A profile is not a machine-global name — it belongs to ONE gateway. The
// Capabilities surface can be pointed at any (connection, profile) pair
// (SkillsView's scope selector, Bot Mode's fixedProfile/fixedConnection), so
// its REST helpers accept either the legacy string form or an explicit scope
// object:
//
// - `undefined` / string → the legacy profile path, PLUS the active registry
// connection tag (connectionScoped, same contract the cron helpers adopted
// in #87882). Without the tag, a window activated onto a registered remote
// gateway read the LOCAL pool's skills/tools/MCP — the wrong machine.
// - `{ connectionId, profile }` → explicit pin. `''`/`'local'` connection
// ids mean the local pool and deliberately DROP the ambient connection
// tag, so a local-profile pick made while a remote gateway is active still
// routes to the local machine.
export type ProfileScope = null | string | { connectionId?: null | string; profile?: null | string }
export function capabilityScoped(scope?: ProfileScope): { connectionId?: string; profile?: string } {
if (scope && typeof scope === 'object') {
const profile = (scope.profile ?? '').trim()
const connectionId = (scope.connectionId ?? '').trim()
return {
...(profile ? { profile } : {}),
...(connectionId && connectionId !== 'local' ? { connectionId } : {})
}
}
return { ...profileScoped(scope), ...connectionScoped() }
}
/** Stable cache-key for a capability scope: `profile` for the local/legacy
* path, `connectionId::profile` for an explicit remote pin. Mirrors
* normalizeProfileKey for plain strings so existing keys stay byte-identical. */
export function profileScopeKey(scope?: ProfileScope): string {
if (scope && typeof scope === 'object') {
const profile = (scope.profile ?? '').trim() || 'default'
const connectionId = (scope.connectionId ?? '').trim()
return connectionId && connectionId !== 'local' ? `${connectionId}::${profile}` : profile
}
return (scope ?? '').trim() || 'default'
}
/** Registry connection id that connection-scoped WS calls should target
* (null → the local pool). Read-only twin of setApiRequestConnection. */
export function getApiRequestConnection(): null | string {
return _apiConnectionId
}
+239
View File
@@ -0,0 +1,239 @@
import type {
ConfigSchemaResponse,
CustomEndpointsResponse,
CustomEndpointUpdate,
CustomEndpointValidationResponse,
EnvVarInfo,
HermesConfig,
HermesConfigRecord,
LogsResponse,
OAuthPollResponse,
OAuthProvidersResponse,
OAuthStartResponse,
OAuthSubmitResponse,
StatusResponse
} from '@/types/hermes'
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped, STARTUP_REQUEST_TIMEOUT_MS } from './client'
export function getStatus(): Promise<StatusResponse> {
return hermesApi<StatusResponse>({
...profileScoped(),
path: '/api/status'
})
}
export function getLogs(params: {
component?: string
file?: string
level?: string
lines?: number
search?: string
}): Promise<LogsResponse> {
const query = new URLSearchParams()
if (params.file) {
query.set('file', params.file)
}
if (typeof params.lines === 'number') {
query.set('lines', String(params.lines))
}
if (params.level && params.level !== 'ALL') {
query.set('level', params.level)
}
if (params.component && params.component !== 'all') {
query.set('component', params.component)
}
if (params.search) {
query.set('search', params.search)
}
const suffix = query.toString()
return hermesApi<LogsResponse>({
...profileScoped(),
path: suffix ? `/api/logs?${suffix}` : '/api/logs'
})
}
export function getHermesConfig(profile?: string): Promise<HermesConfig> {
return hermesApi<HermesConfig>({
...profileScoped(profile),
path: '/api/config',
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export function getHermesConfigRecord(profile?: ProfileScope): Promise<HermesConfigRecord> {
return window.hermesDesktop.api<HermesConfigRecord>({
...capabilityScoped(profile),
path: '/api/config'
})
}
export function getHermesConfigDefaults(): Promise<HermesConfigRecord> {
return hermesApi<HermesConfigRecord>({
...profileScoped(),
path: '/api/config/defaults',
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export function getHermesConfigSchema(profile?: null | string): Promise<ConfigSchemaResponse> {
return hermesApi<ConfigSchemaResponse>({
...profileScoped(profile),
path: '/api/config/schema'
})
}
export function saveHermesConfig(config: HermesConfigRecord, profile?: null | string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(profile),
path: '/api/config',
method: 'PUT',
body: { config }
})
}
export function getEnvVars(profile?: null | string): Promise<Record<string, EnvVarInfo>> {
return hermesApi<Record<string, EnvVarInfo>>({
...profileScoped(profile),
path: '/api/env'
})
}
export function setEnvVar(key: string, value: string, profile?: ProfileScope): Promise<{ ok: boolean }> {
return window.hermesDesktop.api<{ ok: boolean }>({
...capabilityScoped(profile),
path: '/api/env',
method: 'PUT',
body: { key, value }
})
}
export function deleteEnvVar(key: string, profile?: ProfileScope): Promise<{ ok: boolean }> {
return window.hermesDesktop.api<{ ok: boolean }>({
...capabilityScoped(profile),
path: '/api/env',
method: 'DELETE',
body: { key }
})
}
export function revealEnvVar(key: string, profile?: ProfileScope): Promise<{ key: string; value: string }> {
return window.hermesDesktop.api<{ key: string; value: string }>({
...capabilityScoped(profile),
path: '/api/env/reveal',
method: 'POST',
body: { key }
})
}
export function validateProviderCredential(
key: string,
value: string,
apiKey?: string
): Promise<{ ok: boolean; reachable: boolean; message: string; models?: string[] }> {
return hermesApi<{ ok: boolean; reachable: boolean; message: string; models?: string[] }>({
...profileScoped(),
path: '/api/providers/validate',
method: 'POST',
body: { key, value, api_key: apiKey ?? '' }
})
}
export function getCustomEndpoints(): Promise<CustomEndpointsResponse> {
return hermesApi<CustomEndpointsResponse>({
...profileScoped(),
path: '/api/providers/custom-endpoints'
})
}
export function saveCustomEndpoint(endpoint: CustomEndpointUpdate): Promise<CustomEndpointsResponse> {
return hermesApi<CustomEndpointsResponse>({
...profileScoped(),
path: '/api/providers/custom-endpoints',
method: 'POST',
body: endpoint
})
}
export function validateCustomEndpoint(endpoint: CustomEndpointUpdate): Promise<CustomEndpointValidationResponse> {
return hermesApi<CustomEndpointValidationResponse>({
path: '/api/providers/custom-endpoints/validate',
method: 'POST',
body: endpoint
})
}
export function activateCustomEndpoint(id: string): Promise<{ ok: boolean; provider: string; model: string }> {
return hermesApi<{ ok: boolean; provider: string; model: string }>({
...profileScoped(),
path: `/api/providers/custom-endpoints/${encodeURIComponent(id)}/activate`,
method: 'POST'
})
}
export function deleteCustomEndpoint(id: string): Promise<CustomEndpointsResponse> {
return hermesApi<CustomEndpointsResponse>({
...profileScoped(),
path: `/api/providers/custom-endpoints/${encodeURIComponent(id)}`,
method: 'DELETE'
})
}
export function listOAuthProviders(): Promise<OAuthProvidersResponse> {
return hermesApi<OAuthProvidersResponse>({
...profileScoped(),
path: '/api/providers/oauth'
})
}
export function disconnectOAuthProvider(providerId: string): Promise<{ ok: boolean; provider: string }> {
return hermesApi<{ ok: boolean; provider: string }>({
...profileScoped(),
path: `/api/providers/oauth/${encodeURIComponent(providerId)}`,
method: 'DELETE'
})
}
export function startOAuthLogin(providerId: string, profile?: ProfileScope): Promise<OAuthStartResponse> {
return window.hermesDesktop.api<OAuthStartResponse>({
...capabilityScoped(profile),
path: `/api/providers/oauth/${encodeURIComponent(providerId)}/start`,
method: 'POST',
body: {}
})
}
export function submitOAuthCode(providerId: string, sessionId: string, code: string): Promise<OAuthSubmitResponse> {
return hermesApi<OAuthSubmitResponse>({
...profileScoped(),
path: `/api/providers/oauth/${encodeURIComponent(providerId)}/submit`,
method: 'POST',
body: { session_id: sessionId, code }
})
}
export function pollOAuthSession(
providerId: string,
sessionId: string,
profile?: ProfileScope
): Promise<OAuthPollResponse> {
return window.hermesDesktop.api<OAuthPollResponse>({
...capabilityScoped(profile),
path: `/api/providers/oauth/${encodeURIComponent(providerId)}/poll/${encodeURIComponent(sessionId)}`
})
}
export function cancelOAuthSession(sessionId: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
path: `/api/providers/oauth/sessions/${encodeURIComponent(sessionId)}`,
method: 'DELETE'
})
}
+153
View File
@@ -0,0 +1,153 @@
import type {
AutomationBlueprint,
CronDeliveryTarget,
CronJob,
CronJobCreatePayload,
CronJobUpdates,
SessionInfo
} from '@/types/hermes'
import { connectionScoped, hermesApi, profileScoped, STARTUP_REQUEST_TIMEOUT_MS } from './client'
// The cron trigger endpoint intentionally waits for the whole job so its
// response reflects the persisted execution result. Agent jobs can run far
// longer than the Electron fetch default; keep this override local to the one
// synchronous long-operation endpoint rather than weakening all API timeouts.
const CRON_TRIGGER_REQUEST_TIMEOUT_MS = 24 * 60 * 60 * 1000
// Cron jobs are stored per-profile (<HERMES_HOME>/cron/jobs.json), and the
// backend's list endpoint defaults to 'all'. Pass a concrete profile key to
// list just that profile's jobs, or 'all' for the unified cross-profile view.
// Omitting the arg keeps the legacy 'all' default for non-profile callers.
// profileScoped() still rides along for backend-process routing.
export function getCronJobs(profile?: string): Promise<CronJob[]> {
const suffix = profile ? `?profile=${encodeURIComponent(profile)}` : ''
return hermesApi<CronJob[]>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs${suffix}`,
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export function getCronJob(jobId: string): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}`
})
}
export async function getCronJobRuns(jobId: string, limit = 20): Promise<SessionInfo[]> {
const { runs } = await hermesApi<{ runs: SessionInfo[] }>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}/runs?limit=${limit}`
})
return runs ?? []
}
// The single source of truth for cron delivery targets (local + configured
// gateways). Both the manual cron editor and the blueprint dialog use this so
// they never offer a platform that isn't connected. Mirrors the dashboard.
export async function getCronDeliveryTargets(): Promise<CronDeliveryTarget[]> {
const { targets } = await hermesApi<{ targets: CronDeliveryTarget[] }>({
...profileScoped(),
...connectionScoped(),
path: '/api/cron/delivery-targets'
})
return targets ?? []
}
export function createCronJob(body: CronJobCreatePayload): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: '/api/cron/jobs',
method: 'POST',
body
})
}
export function updateCronJob(jobId: string, updates: CronJobUpdates): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}`,
method: 'PUT',
body: { updates }
})
}
export function pauseCronJob(jobId: string): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}/pause`,
method: 'POST'
})
}
export function resumeCronJob(jobId: string): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}/resume`,
method: 'POST'
})
}
export function triggerCronJob(jobId: string): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}/trigger`,
method: 'POST',
timeoutMs: CRON_TRIGGER_REQUEST_TIMEOUT_MS
})
}
export function deleteCronJob(jobId: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/jobs/${encodeURIComponent(jobId)}`,
method: 'DELETE'
})
}
// Automation Blueprints — parameterized cron templates the backend serves from
// cron/blueprint_catalog.py. getAutomationBlueprints returns the gallery
// (deliver options already rewritten to this machine's configured gateways);
// instantiateAutomationBlueprint fills the slots and creates a real cron job via
// the same create_job path as createCronJob.
//
// Profile-scoping is intentionally asymmetric: the GET catalog is global (the
// list endpoint takes no profile — only deliver options are rewritten from the
// configured gateways), so it carries only the profileScoped() header for
// routing. instantiate creates a real per-profile job, so it names the target
// profile explicitly via ?profile=. This mirrors the dashboard's api.ts.
export function getAutomationBlueprints(): Promise<{ blueprints: AutomationBlueprint[] }> {
return hermesApi<{ blueprints: AutomationBlueprint[] }>({
...profileScoped(),
...connectionScoped(),
path: '/api/cron/blueprints',
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export function instantiateAutomationBlueprint(
body: { blueprint: string; values: Record<string, string> },
profile: string
): Promise<CronJob> {
return hermesApi<CronJob>({
...profileScoped(),
...connectionScoped(),
path: `/api/cron/blueprints/instantiate?profile=${encodeURIComponent(profile)}`,
method: 'POST',
body
})
}
+147
View File
@@ -0,0 +1,147 @@
import type { McpCatalogResponse, McpServerSummary } from '@/types/hermes'
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped } from './client'
export interface McpTestResult {
ok: boolean
error?: string
/** `schema_chars` (converted registry-schema size, chars) is additive —
* older backends omit it and the cost overlay shows no token estimate. */
tools: { name: string; description: string; schema_chars?: number }[]
/** Capability counts (absent on older backends / failed probes). */
prompts?: number
resources?: number
}
export interface McpOAuthFlow {
flow_id: string
server_name: string
status: 'starting' | 'authorization_required' | 'approved' | 'error'
authorization_url: string | null
error: string | null
tools?: { name: string; description: string }[]
}
/** Connect to the server, list its tools, disconnect. Slow (spawns/handshakes
* for real) — well past the 15s default fetch timeout. */
export function testMcpServer(name: string, profile?: ProfileScope): Promise<McpTestResult> {
return window.hermesDesktop.api<McpTestResult>({
...capabilityScoped(profile),
path: `/api/mcp/servers/${encodeURIComponent(name)}/test`,
method: 'POST',
timeoutMs: 60_000
})
}
/** Replace the whole `mcp_servers` map (the mcp.json editor's save). Unlike
* `saveHermesConfig`, this REPLACES rather than deep-merges, so deletes,
* re-enables (dropping `enabled: false`), and removed nested fields persist. */
export function saveMcpServers(
servers: Record<string, Record<string, unknown>>,
profile?: ProfileScope
): Promise<{ ok: boolean }> {
return window.hermesDesktop.api<{ ok: boolean }>({
...capabilityScoped(profile),
path: '/api/mcp/servers',
method: 'PUT',
body: { servers }
})
}
/** Start an MCP OAuth flow and return the authorization URL. */
export function authMcpServer(name: string, profile?: ProfileScope): Promise<McpOAuthFlow> {
return window.hermesDesktop.api<McpOAuthFlow>({
...capabilityScoped(profile),
path: `/api/mcp/servers/${encodeURIComponent(name)}/auth`,
method: 'POST',
timeoutMs: 60_000
})
}
export function getMcpOAuthFlow(flowId: string, profile?: ProfileScope): Promise<McpOAuthFlow> {
return window.hermesDesktop.api<McpOAuthFlow>({
...capabilityScoped(profile),
path: `/api/mcp/oauth/flows/${encodeURIComponent(flowId)}`
})
}
/** Cancel an in-flight MCP OAuth flow server-side, freeing the per-server
* "already in progress" slot so a retry doesn't 409. */
export function cancelMcpOAuthFlow(flowId: string, profile?: null | string): Promise<{ ok: boolean; status: string }> {
return hermesApi<{ ok: boolean; status: string }>({
...profileScoped(profile),
path: `/api/mcp/oauth/flows/${encodeURIComponent(flowId)}`,
method: 'DELETE'
})
}
// ---------------------------------------------------------------------------
// MCP servers — structured list / test / enable toggle / catalog (parity with
// `hermes mcp` and the dashboard MCP page). Raw JSON editing stays in
// config.yaml via saveHermesConfig.
// ---------------------------------------------------------------------------
export function listMcpServers(): Promise<{ servers: McpServerSummary[] }> {
return hermesApi<{ servers: McpServerSummary[] }>({
...profileScoped(),
path: '/api/mcp/servers'
})
}
/** Add one server to `mcp_servers` (validated + name-collision-checked
* server-side — the same endpoint the dashboard's add form uses). */
export function addMcpServer(body: {
name: string
url?: string
command?: string
args?: string[]
env?: Record<string, string>
auth?: string
}): Promise<McpServerSummary> {
return hermesApi<McpServerSummary>({
...profileScoped(),
path: '/api/mcp/servers',
method: 'POST',
body
})
}
/** Remove one server from `mcp_servers` (the inline setup card's rollback
* when a directory install is cancelled after the config write). */
export function removeMcpServer(name: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
path: `/api/mcp/servers/${encodeURIComponent(name)}`,
method: 'DELETE'
})
}
export function setMcpServerEnabled(name: string, enabled: boolean): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
path: `/api/mcp/servers/${encodeURIComponent(name)}/enabled`,
method: 'PUT',
body: { enabled }
})
}
export function getMcpCatalog(profile?: ProfileScope): Promise<McpCatalogResponse> {
return window.hermesDesktop.api<McpCatalogResponse>({
...capabilityScoped(profile),
path: '/api/mcp/catalog'
})
}
export function installMcpCatalogEntry(
name: string,
env: Record<string, string> = {},
profile?: ProfileScope
): Promise<{ ok: boolean; name?: string; pid?: number; action?: string; background?: boolean }> {
return window.hermesDesktop.api<{ ok: boolean; name?: string; pid?: number; action?: string; background?: boolean }>({
...capabilityScoped(profile),
path: '/api/mcp/catalog/install',
method: 'POST',
body: { name, env, enable: true },
timeoutMs: 60_000
})
}
+131
View File
@@ -0,0 +1,131 @@
import type {
MessagingPlatformsResponse,
MessagingPlatformTestResponse,
MessagingPlatformUpdate,
PairingResponse,
PairingUser,
WebhookCreatePayload,
WebhookCreateResponse,
WebhookEnableResponse,
WebhooksResponse
} from '@/types/hermes'
import { hermesApi, profileScoped } from './client'
export function getMessagingPlatforms(profile?: null | string): Promise<MessagingPlatformsResponse> {
return hermesApi<MessagingPlatformsResponse>({
...profileScoped(profile),
path: '/api/messaging/platforms'
})
}
export function updateMessagingPlatform(
platformId: string,
body: MessagingPlatformUpdate,
profile?: null | string
): Promise<{ ok: boolean; platform: string }> {
return hermesApi<{ ok: boolean; platform: string }>({
...profileScoped(profile),
path: `/api/messaging/platforms/${encodeURIComponent(platformId)}`,
method: 'PUT',
body
})
}
export function testMessagingPlatform(
platformId: string,
profile?: null | string
): Promise<MessagingPlatformTestResponse> {
return hermesApi<MessagingPlatformTestResponse>({
...profileScoped(profile),
path: `/api/messaging/platforms/${encodeURIComponent(platformId)}/test`,
method: 'POST'
})
}
// -- Pairing (who may DM the bot) --------------------------------------------
// Unknown DMers get a one-time code and land in `pending` until an admin
// approves them. Approval grants on the row's `request_id`, never on the code:
// the code is the requester's proof that the channel is theirs and is never
// returned by the API, while an authenticated admin is only ever identifying
// a row they can already see.
export function getPairing(profile?: null | string): Promise<PairingResponse> {
return hermesApi<PairingResponse>({
...profileScoped(profile),
path: '/api/pairing'
})
}
export function approvePairing(
platform: string,
requestId: string,
profile?: null | string
): Promise<{ ok: boolean; user: PairingUser }> {
return hermesApi<{ ok: boolean; user: PairingUser }>({
...profileScoped(profile),
path: '/api/pairing/approve',
method: 'POST',
// These endpoints read the profile off the body, not the query string —
// `profileScoped()` alone would approve into the wrong profile's store.
body: { platform, request_id: requestId, ...profileScoped(profile) }
})
}
export function revokePairing(platform: string, userId: string, profile?: null | string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(profile),
path: '/api/pairing/revoke',
method: 'POST',
body: { platform, user_id: userId, ...profileScoped(profile) }
})
}
// -- Webhooks (subscription CRUD) --------------------------------------------
// The webhook receiver is its own gateway platform; subscriptions live in a
// shared JSON store the CLI/dashboard also drive. Enable mutates config and
// best-effort restarts the gateway; subscription changes hot-reload.
export function getWebhooks(): Promise<WebhooksResponse> {
return hermesApi<WebhooksResponse>({
...profileScoped(),
path: '/api/webhooks'
})
}
export function enableWebhooks(): Promise<WebhookEnableResponse> {
return hermesApi<WebhookEnableResponse>({
...profileScoped(),
path: '/api/webhooks/enable',
method: 'POST'
})
}
export function createWebhook(body: WebhookCreatePayload): Promise<WebhookCreateResponse> {
return hermesApi<WebhookCreateResponse>({
...profileScoped(),
path: '/api/webhooks',
method: 'POST',
body
})
}
export function deleteWebhook(name: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
path: `/api/webhooks/${encodeURIComponent(name)}`,
method: 'DELETE'
})
}
export function setWebhookEnabled(
name: string,
enabled: boolean
): Promise<{ enabled: boolean; name: string; ok: boolean }> {
return hermesApi<{ enabled: boolean; name: string; ok: boolean }>({
...profileScoped(),
path: `/api/webhooks/${encodeURIComponent(name)}/enabled`,
method: 'PUT',
body: { enabled }
})
}
+129
View File
@@ -0,0 +1,129 @@
import type {
AnalyticsResponse,
AuxiliaryModelsResponse,
MoaConfigResponse,
ModelAssignmentRequest,
ModelAssignmentResponse,
ModelInfoResponse,
ModelOptionsResponse
} from '@/types/hermes'
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped, STARTUP_REQUEST_TIMEOUT_MS } from './client'
export function getGlobalModelInfo(profile?: null | string): Promise<ModelInfoResponse> {
return hermesApi<ModelInfoResponse>({
...profileScoped(profile),
path: '/api/model/info',
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export function getUsageAnalytics(days = 30, profile?: ProfileScope): Promise<AnalyticsResponse> {
return window.hermesDesktop.api<AnalyticsResponse>({
...capabilityScoped(profile),
path: `/api/analytics/usage?days=${Math.max(1, Math.floor(days))}`
})
}
export function getGlobalModelOptions(
opts?: {
refresh?: boolean
includeUnconfigured?: boolean
explicitOnly?: boolean
},
profile?: null | string
): Promise<ModelOptionsResponse> {
const params = new URLSearchParams()
if (opts?.refresh) {
params.set('refresh', '1')
}
if (opts?.includeUnconfigured) {
params.set('include_unconfigured', '1')
}
if (opts?.explicitOnly !== false) {
params.set('explicit_only', '1')
}
return hermesApi<ModelOptionsResponse>({
...profileScoped(profile),
path: params.size > 0 ? `/api/model/options?${params.toString()}` : '/api/model/options',
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export interface RecommendedDefaultModel {
provider: string
model: string
/** True/false for Nous (free vs paid tier); null for other providers. */
free_tier: boolean | null
}
// Recommended default model for a freshly-authenticated provider. Mirrors the
// curation `hermes model` does — for Nous it honors the free/paid tier so a
// free user gets a free model instead of a paid default.
export function getRecommendedDefaultModel(
provider: string,
profile?: null | string
): Promise<RecommendedDefaultModel> {
return hermesApi<RecommendedDefaultModel>({
...profileScoped(profile),
path: `/api/model/recommended-default?provider=${encodeURIComponent(provider)}`
})
}
export function setGlobalModel(
provider: string,
model: string
): Promise<{ ok: boolean; provider: string; model: string }> {
return hermesApi<{ ok: boolean; provider: string; model: string }>({
...profileScoped(),
path: '/api/model/set',
method: 'POST',
body: {
scope: 'main',
provider,
model
}
})
}
export function getAuxiliaryModels(profile?: null | string): Promise<AuxiliaryModelsResponse> {
return hermesApi<AuxiliaryModelsResponse>({
...profileScoped(profile),
path: '/api/model/auxiliary'
})
}
export function getMoaModels(profile?: null | string): Promise<MoaConfigResponse> {
return hermesApi<MoaConfigResponse>({
...profileScoped(profile),
path: '/api/model/moa'
})
}
export function saveMoaModels(
body: MoaConfigResponse,
profile?: null | string
): Promise<MoaConfigResponse & { ok: boolean }> {
return hermesApi<MoaConfigResponse & { ok: boolean }>({
...profileScoped(profile),
path: '/api/model/moa',
method: 'PUT',
body
})
}
export function setModelAssignment(
body: ModelAssignmentRequest,
profile?: null | string
): Promise<ModelAssignmentResponse> {
return hermesApi<ModelAssignmentResponse>({
...profileScoped(profile),
path: '/api/model/set',
method: 'POST',
body
})
}
+128
View File
@@ -0,0 +1,128 @@
import type { HermesConnection } from '@/global'
import { reconnectBackoffDelayMs } from '@/lib/reconnect-backoff'
import { getApiRequestConnection, getApiRequestProfile, hermesApi, profileScoped } from './client'
/** Resolve the ACTIVE backend's connection descriptor, (connectionId,
* profile)-scoped — mirroring how store/profile resolves $connection: a
* registry agent's descriptor comes from getConnectionFor (its SOURCE
* connection), everything else from the profile-keyed local pool. The
* getConnectionFor bridge is optional (older Desktop mains); without it the
* profile-scoped pool lookup is the best available answer. */
async function activeConnection(): Promise<HermesConnection> {
const getConnectionFor = window.hermesDesktop.getConnectionFor
const connectionId = getApiRequestConnection()
if (connectionId && getConnectionFor) {
return getConnectionFor({ connectionId, profile: getApiRequestProfile() })
}
return window.hermesDesktop.getConnection(getApiRequestProfile())
}
/** Options for a plugin REST call — mirrors the app's own `hermesDesktop.api`
* shape, minus the path (which is namespace-derived). */
export interface PluginRestOptions {
method?: string
body?: unknown
/** Single-file multipart upload (see HermesApiRequest.upload). */
upload?: { filename: string; contentType?: string; bytes: ArrayBuffer }
timeoutMs?: number
}
// Normalize `path` to a leading-slash suffix relative to `/api/plugins/<id>`.
// The namespace is the boundary — reject `..` so a relative segment can't
// normalize out into another plugin's API or a core route. Check the path
// portion only (before any query/hash).
function pluginPathSuffix(caller: string, path: string): string {
const suffix = path.startsWith('/') ? path : `/${path}`
if (suffix.split(/[?#]/, 1)[0].split('/').includes('..')) {
throw new Error(`${caller}: illegal path traversal in "${path}"`)
}
return suffix
}
/** The plugin REST door. Every call is scoped BY CONSTRUCTION to the plugin's
* own backend namespace — `path` is relative to `/api/plugins/<pluginId>`
* ('/board' → `/api/plugins/kanban/board`), so a plugin can't address another
* plugin's API or a core route through it. Profile-aware like every desktop
* REST call. Broader reach (core endpoints, another namespace) is the future
* declared-capability seam; today the namespace IS the boundary. */
export async function pluginRest<T>(pluginId: string, path: string, opts: PluginRestOptions = {}): Promise<T> {
if (!window.hermesDesktop?.api) {
throw new Error('Hermes desktop bridge unavailable')
}
const suffix = pluginPathSuffix('pluginRest', path)
return hermesApi<T>({
path: `/api/plugins/${pluginId}${suffix}`,
method: opts.method,
body: opts.body,
upload: opts.upload,
timeoutMs: opts.timeoutMs,
...profileScoped()
})
}
/** The plugin WebSocket door — the live twin of `pluginRest`, scoped the same
* way: `path` is relative to `/api/plugins/<pluginId>` ('/events' → the
* plugin's own event stream). Token-mode backends auth via the same query
* credential the app's own sockets use; OAuth remotes resolve null (callers
* keep their polling fallback — every consumer must have one anyway, since a
* socket can drop). Auto-reconnects with backoff until disposed. */
export function pluginSocket(pluginId: string, path: string, onMessage: (data: unknown) => void): () => void {
const suffix = pluginPathSuffix('pluginSocket', path)
let socket: null | WebSocket = null
let disposed = false
let attempt = 0
const connect = async () => {
const connection = await activeConnection().catch(() => null)
// No bridge / OAuth cookie auth (WS tickets are single-use, core-managed):
// stay on the polling fallback rather than half-working.
if (disposed || !connection || connection.authMode === 'oauth') {
return
}
const base = connection.baseUrl.replace(/^http/, 'ws')
const join = suffix.includes('?') ? '&' : '?'
socket = new WebSocket(
`${base}/api/plugins/${pluginId}${suffix}${join}token=${encodeURIComponent(connection.token)}`
)
socket.onmessage = event => {
attempt = 0
try {
onMessage(JSON.parse(String(event.data)))
} catch {
// Non-JSON frame — plugin streams are JSON by contract; skip it.
}
}
socket.onclose = () => {
socket = null
if (!disposed) {
// Full-jitter exponential backoff: same rationale as the gateway
// socket reconnect loops — an immediate-retry loop across many
// desktop clients floods the gateway with connection attempts
// during a restart.
window.setTimeout(() => void connect(), reconnectBackoffDelayMs(attempt, { baseDelayMs: 500, capMs: 30_000 }))
attempt += 1
}
}
}
void connect()
return () => {
disposed = true
socket?.close()
}
}
+89
View File
@@ -0,0 +1,89 @@
import type {
ProfileCreatePayload,
ProfileDesktopOverlay,
ProfileSetupCommand,
ProfileSoul,
ProfilesResponse
} from '@/types/hermes'
import { hermesApi, STARTUP_REQUEST_TIMEOUT_MS } from './client'
export function getProfiles(): Promise<ProfilesResponse> {
return hermesApi<ProfilesResponse>({
path: '/api/profiles',
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
export function createProfile(body: ProfileCreatePayload): Promise<{ name: string; ok: boolean; path: string }> {
return hermesApi<{ name: string; ok: boolean; path: string }>({
path: '/api/profiles',
method: 'POST',
body
})
}
export function renameProfile(name: string, newName: string): Promise<{ name: string; ok: boolean; path: string }> {
return hermesApi<{ name: string; ok: boolean; path: string }>({
path: `/api/profiles/${encodeURIComponent(name)}`,
method: 'PATCH',
body: { new_name: newName }
})
}
export function deleteProfile(name: string): Promise<{ ok: boolean; path: string }> {
return hermesApi<{ ok: boolean; path: string }>({
path: `/api/profiles/${encodeURIComponent(name)}`,
method: 'DELETE'
})
}
export function getProfileSoul(name: string): Promise<ProfileSoul> {
return hermesApi<ProfileSoul>({
path: `/api/profiles/${encodeURIComponent(name)}/soul`
})
}
export function updateProfileSoul(name: string, content: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
path: `/api/profiles/${encodeURIComponent(name)}/soul`,
method: 'PUT',
body: { content }
})
}
export function getProfileSetupCommand(name: string): Promise<ProfileSetupCommand> {
return hermesApi<ProfileSetupCommand>({
path: `/api/profiles/${encodeURIComponent(name)}/setup-command`
})
}
/** Export a profile to a shareable .tar.gz on the backend's filesystem.
* `extraFiles` stages extra root-level files (desktop.json — the appearance/
* interface overlay) into the archive alongside the profile's own artifacts. */
export function exportProfileArchive(
name: string,
opts: { extraFiles?: Record<string, string>; output?: string } = {}
): Promise<{ archive: string; ok: boolean }> {
return hermesApi<{ archive: string; ok: boolean }>({
path: `/api/profiles/${encodeURIComponent(name)}/export`,
method: 'POST',
body: { extra_files: opts.extraFiles ?? {}, output: opts.output ?? '' },
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
/** Import a profile .tar.gz as a new profile. Returns the bundled desktop
* appearance overlay too (when the archive carried one) so the caller can
* apply theme/layout without another round-trip. */
export function importProfileArchive(
archive: string,
name?: string
): Promise<{ desktop: null | ProfileDesktopOverlay; name: string; ok: boolean; path: string }> {
return hermesApi<{ desktop: null | ProfileDesktopOverlay; name: string; ok: boolean; path: string }>({
path: '/api/profiles/import',
method: 'POST',
body: { archive, name: name || null },
timeoutMs: STARTUP_REQUEST_TIMEOUT_MS
})
}
+465
View File
@@ -0,0 +1,465 @@
import { isMissingRestEndpoint } from '@/lib/gateway-rpc'
import { recordTranscriptTail } from '@/store/transcript-tail'
import type {
PaginatedSessions,
SessionInfo,
SessionMessage,
SessionMessagesResponse,
SessionSearchResponse
} from '@/types/hermes'
import { hermesApi } from './client'
const SESSION_LIST_REQUEST_TIMEOUT_MS = 60_000
/**
* Trim a page to its window WITHOUT discarding pinned rows.
*
* The list endpoints deliberately back-fill pinned conversations past their
* LIMIT — a pin means "always reachable", so an aged-out pinned chat is
* appended after the recency window. A plain `slice(0, limit)` throws exactly
* those rows away again, which is why pins silently stopped rendering past
* some count: the sidebar could only ever show the pins that happened to fall
* inside the most-recent page.
*/
function pageWindow(sessions: SessionInfo[], limit: number): SessionInfo[] {
if (sessions.length <= limit) {
return sessions
}
const recent = sessions.slice(0, limit)
return [...recent, ...sessions.slice(limit).filter(session => session.pinned)]
}
export async function listSessions(
limit = 40,
minMessages = 0,
archived: 'exclude' | 'include' | 'only' = 'exclude',
order: 'created' | 'recent' = 'recent'
): Promise<PaginatedSessions> {
const result = await hermesApi<PaginatedSessions>({
path:
`/api/sessions?limit=${limit}&offset=0&min_messages=${Math.max(0, minMessages)}` +
`&archived=${archived}&order=${order}`,
timeoutMs: SESSION_LIST_REQUEST_TIMEOUT_MS
})
return {
...result,
sessions: pageWindow(result.sessions, limit),
offset: 0
}
}
// Unified, read-only session list aggregated across ALL profiles. Served by the
// primary backend straight off each profile's state.db — no per-profile backend
// is spawned. Single-profile users get the same rows as listSessions(), tagged
// profile="default".
// Source scoping lets callers split the unified list into independent slices:
// recents pass `excludeSources: ['cron']`, the cron-jobs section passes
// `source: 'cron'`. Without this a burst of (always-newest) cron sessions
// consumes the whole recents page and starves real conversations.
export interface SessionSourceFilter {
source?: string
excludeSources?: string[]
}
export async function listAllProfileSessions(
limit = 40,
minMessages = 0,
archived: 'exclude' | 'include' | 'only' = 'exclude',
order: 'created' | 'recent' = 'recent',
profile: 'all' | (string & {}) = 'all',
filter: SessionSourceFilter = {}
): Promise<PaginatedSessions> {
const sourceParam = filter.source ? `&source=${encodeURIComponent(filter.source)}` : ''
const excludeParam = filter.excludeSources?.length
? `&exclude_sources=${encodeURIComponent(filter.excludeSources.join(','))}`
: ''
const result = await hermesApi<PaginatedSessions>({
path:
`/api/profiles/sessions?limit=${limit}&offset=0&min_messages=${Math.max(0, minMessages)}` +
`&archived=${archived}&order=${order}&profile=${encodeURIComponent(profile)}${sourceParam}${excludeParam}`,
timeoutMs: SESSION_LIST_REQUEST_TIMEOUT_MS
})
return {
...result,
sessions: pageWindow(result.sessions, limit),
offset: 0
}
}
// Batched sidebar slices in one request: recents (scoped to the active profile),
// cron, and messaging. The backend opens each profile's state.db once and runs
// all three filtered queries, replacing three separate listAllProfileSessions
// calls that each reopened + re-counted every profile DB per refresh. Electron
// splices remote profiles per slice (see interceptSessionRequestForRemote).
export interface SidebarSessionSlice {
sessions: SessionInfo[]
/** Per-profile "the window came back full, more rows exist on disk" flags —
* what pagination needs, without a COUNT(*) per profile DB per refresh. */
profiles_truncated?: Record<string, boolean>
/** Per-profile tokens and spend over every session, not just this window.
* Absent from the legacy per-slice endpoint, which has no aggregate. */
profiles_usage?: Record<string, { cost_usd: number; tokens: number }>
}
/** Which profiles filled their per-profile window in a returned page. The
* legacy per-slice endpoint doesn't report this, so derive it from the rows:
* a profile at (or over) the cap still has more on disk. Pinned rows are
* discounted — they're back-filled past the LIMIT, so counting them fakes a
* full page and leaves a "Load more" that can never resolve. */
function profilesTruncatedFrom(sessions: SessionInfo[], cap: number): Record<string, boolean> {
const counts = new Map<string, number>()
for (const session of sessions) {
const key = session.profile || 'default'
counts.set(key, (counts.get(key) ?? 0) + (session.pinned ? 0 : 1))
}
return Object.fromEntries([...counts].map(([name, count]) => [name, count >= cap]))
}
export interface SidebarSessionsResponse {
recents: SidebarSessionSlice
cron: SidebarSessionSlice
messaging: SidebarSessionSlice
errors?: Array<{ profile: string; error: string }>
}
export interface SidebarSessionsRequest {
recentsProfile: 'all' | (string & {})
recentsLimit: number
recentsExclude: string[]
cronLimit: number
messagingLimit: number
messagingExclude: string[]
}
// The batched /sidebar endpoint shipped later than the per-slice route, so a
// newer desktop can meet an older backend that 404s it ("No such API
// endpoint"). Endpoint-missing is a capability signal, not a transient
// failure: remember it (per renderer lifetime — a runtime home change reloads
// the window and re-probes) and serve every subsequent refresh straight from
// the three proven per-slice calls instead of re-probing a known-dead route
// once per turn/broadcast.
let sidebarBatchEndpointMissing = false
// Capability flags are per-backend facts. A hard re-home reloads the window
// (module state resets naturally), but a soft gateway switch re-dials in
// place — the next backend may well have the batched route, so the switch
// paths call this to re-probe rather than leak the old backend's capability.
export function resetSidebarBatchCapability() {
sidebarBatchEndpointMissing = false
}
// Compatibility fallback: reassemble the three sidebar slices from the
// per-slice endpoint, mirroring the batched route's semantics (min_messages=1,
// archived excluded, recency order; every slice scoped to the caller's profile).
// Rides the same Electron remote-splice
// interception as the pre-batching desktop, so remote profiles stay correct.
async function listSidebarSessionsLegacy(req: SidebarSessionsRequest): Promise<SidebarSessionsResponse> {
const [recents, cron, messaging] = await Promise.all([
listAllProfileSessions(req.recentsLimit, 1, 'exclude', 'recent', req.recentsProfile, {
excludeSources: req.recentsExclude
}),
listAllProfileSessions(req.cronLimit, 1, 'exclude', 'recent', req.recentsProfile, { source: 'cron' }),
listAllProfileSessions(req.messagingLimit, 1, 'exclude', 'recent', req.recentsProfile, {
excludeSources: req.messagingExclude
})
])
const errors = [...(recents.errors ?? []), ...(cron.errors ?? []), ...(messaging.errors ?? [])]
return {
recents: {
profiles_truncated: profilesTruncatedFrom(recents.sessions, req.recentsLimit),
sessions: recents.sessions
},
cron: { sessions: cron.sessions },
messaging: { sessions: messaging.sessions },
...(errors.length ? { errors } : {})
}
}
/** The PR each of these sessions opened, recovered from its own transcript —
* for sessions whose recorded branch can't answer (they started on trunk and
* did the work in a worktree). Also returns every id it looked at, so the
* caller can remember a miss and never ask again. */
export function scanSessionPullRequests(
ids: string[]
): Promise<{ pull_requests: Record<string, { number: number; url: string }>; scanned: string[] }> {
return hermesApi<{
pull_requests: Record<string, { number: number; url: string }>
scanned: string[]
}>({
path: '/api/profiles/sessions/pull-requests',
method: 'POST',
body: { ids }
})
}
export async function listSidebarSessions(req: SidebarSessionsRequest): Promise<SidebarSessionsResponse> {
if (sidebarBatchEndpointMissing) {
return listSidebarSessionsLegacy(req)
}
const params = new URLSearchParams({
recents_profile: req.recentsProfile,
recents_limit: String(Math.max(1, req.recentsLimit)),
cron_limit: String(Math.max(1, req.cronLimit)),
messaging_limit: String(Math.max(1, req.messagingLimit))
})
if (req.recentsExclude.length) {
params.set('recents_exclude', req.recentsExclude.join(','))
}
if (req.messagingExclude.length) {
params.set('messaging_exclude', req.messagingExclude.join(','))
}
let result: SidebarSessionsResponse
try {
result = await hermesApi<SidebarSessionsResponse>({
path: `/api/profiles/sessions/sidebar?${params.toString()}`,
timeoutMs: SESSION_LIST_REQUEST_TIMEOUT_MS
})
} catch (err) {
// Safe to read a 404 as route-missing here: this GET has no path params,
// so it cannot 404 on a bad id.
if (!isMissingRestEndpoint(err)) {
throw err
}
// Older backend without the batched route (desktop/runtime version skew).
sidebarBatchEndpointMissing = true
return listSidebarSessionsLegacy(req)
}
return {
recents: { ...result.recents, sessions: result.recents?.sessions ?? [] },
cron: { ...result.cron, sessions: result.cron?.sessions ?? [] },
messaging: { ...result.messaging, sessions: result.messaging?.sessions ?? [] },
errors: result.errors
}
}
// Mutations take the owning `profile` so Electron can route them to the correct
// remote backend or local profile scope. Omit for the current/default profile.
export function setSessionArchived(id: string, archived: boolean, profile?: string | null): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}`,
method: 'PATCH',
body: { archived }
})
}
// Mirror a sidebar pin to the backend "keep" flag so the sessions.auto_archive
// sweep (which runs backend-side, blind to Desktop localStorage) never hides a
// pinned chat. Best-effort: the sidebar stays localStorage-driven for its own
// display; this only feeds the backend policy.
export function setSessionPinnedRemote(id: string, pinned: boolean, profile?: string | null): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}`,
method: 'PATCH',
body: { pinned }
})
}
// Mirror a sidebar unread toggle to the backend read-state watermark
// (sessions.last_read_at via SessionDB.set_session_read). Same profile
// routing as the other session mutations: a remote session's row lives only
// on its remote host, so the owning profile must travel with the request.
export function setSessionUnreadRemote(id: string, unread: boolean, profile?: string | null): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}`,
method: 'PATCH',
body: { unread }
})
}
export function searchSessions(query: string): Promise<SessionSearchResponse> {
return hermesApi<SessionSearchResponse>({
path: `/api/sessions/search?q=${encodeURIComponent(query)}`
})
}
// Resolves a single session row by id on one backend (the active profile, or
// the given `profile`). The backend resolves exact ids and unique prefixes and
// 404s when the id isn't on that profile — so a cheap by-id lookup replaces the
// cross-profile list scan when locating an unknown id's owner.
export function getSession(id: string, profile?: string | null): Promise<SessionInfo> {
const suffix = profile ? `?profile=${encodeURIComponent(profile)}` : ''
return hermesApi<SessionInfo>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}${suffix}`
})
}
// Reads another profile's transcript. For a remote profile Electron reroutes
// this GET to the remote backend (which serves its own state.db); for a local
// profile the primary opens that profile's state.db via ?profile=. Omit for
// the current/default profile.
export function getSessionMessages(
id: string,
profile?: string | null,
page: { limit?: number; offset?: number; order?: 'latest' | 'oldest'; includeCompacted?: boolean } = {}
): Promise<SessionMessagesResponse> {
const query = new URLSearchParams()
if (profile) {
query.set('profile', profile)
}
if (page.limit !== undefined) {
query.set('limit', String(page.limit))
}
if (page.offset !== undefined) {
query.set('offset', String(page.offset))
}
if (page.order) {
query.set('order', page.order)
}
if (page.includeCompacted !== undefined) {
query.set('include_compacted', String(page.includeCompacted))
}
const suffix = query.size ? `?${query.toString()}` : ''
return hermesApi<SessionMessagesResponse>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}/messages${suffix}`
})
}
/**
* The initial hydration page: enough tail to fill the transcript window a few
* times over, small enough that opening a long session doesn't ship (and
* convert) hundreds of rows nobody has scrolled to. Older rows load on demand
* via `getOlderSessionMessages` when "Show earlier" exhausts the in-memory
* store (see app/chat/transcript-backfill).
*/
export const LATEST_SESSION_MESSAGES_LIMIT = 120
export function getLatestSessionMessages(id: string, profile?: string | null): Promise<SessionMessagesResponse> {
// includeCompacted: durable display history must include rows preserved by
// in-place compaction (active=0, compacted=1); without them the transcript
// silently ends at the compaction boundary and earlier turns are unreachable.
return getSessionMessages(id, profile, {
limit: LATEST_SESSION_MESSAGES_LIMIT,
order: 'latest',
includeCompacted: true
}).then(page => {
// Record whether the tail was truncated (page came back full) and where
// the next older page starts, so "Show earlier" can backfill over REST
// (app/chat/transcript-backfill). Keyed under both the requested id and
// the resolved id — callers hold either.
recordTranscriptTail(id, page, profile)
if (page.session_id && page.session_id !== id) {
recordTranscriptTail(page.session_id, page, profile)
}
return page
})
}
/**
* One page of messages OLDER than the `offset` newest rows.
*
* Backend semantics (`_handle_session_messages` → `SessionDB.get_messages`
* with `latest=True`): the offset is measured back from the NEWEST message
* and the selected page is returned in chronological order. So after a tail
* hydration of N rows, `getOlderSessionMessages(id, profile, N)` returns the
* page immediately preceding it, ready to prepend.
*
* Legacy backends without pagination support return the full transcript and
* no `pagination` metadata — callers detect that via the missing field and
* treat the response as the complete history (see transcript-backfill).
*/
export function getOlderSessionMessages(
id: string,
profile: string | null | undefined,
offset: number,
limit: number = LATEST_SESSION_MESSAGES_LIMIT
): Promise<SessionMessagesResponse> {
return getSessionMessages(id, profile, { includeCompacted: true, limit, offset, order: 'latest' })
}
export async function getAllSessionMessages(
id: string,
profile?: string | null,
options: { maxJsonChars?: number } = {}
): Promise<SessionMessagesResponse> {
const messages: SessionMessage[] = []
const pageSize = 500
const maxJsonChars = options.maxJsonChars ?? 32_000_000
let jsonChars = 0
let offset = 0
let resolvedSessionId = id
while (true) {
const page = await getSessionMessages(id, profile, {
limit: pageSize,
offset,
order: 'oldest',
includeCompacted: true
})
resolvedSessionId = page.session_id
jsonChars += (JSON.stringify(page.messages) ?? '').length
if (jsonChars > maxJsonChars) {
throw new Error(
'Session transcript exceeds the Desktop safe-load limit; use the Web Dashboard export for this session.'
)
}
messages.push(...page.messages)
// Legacy backends ignore pagination and return the full transcript.
if (!page.pagination || page.messages.length === 0 || page.messages.length < page.pagination.limit) {
break
}
offset += page.messages.length
}
return { session_id: resolvedSessionId, messages }
}
export function deleteSession(id: string, profile?: string | null): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}`,
method: 'DELETE'
})
}
export function renameSession(
id: string,
title: string,
profile?: string | null
): Promise<{ ok: boolean; title: string }> {
return hermesApi<{ ok: boolean; title: string }>({
...(profile ? { profile } : {}),
path: `/api/sessions/${encodeURIComponent(id)}`,
method: 'PATCH',
body: { title, ...(profile ? { profile } : {}) }
})
}
+162
View File
@@ -0,0 +1,162 @@
import type {
SkillHubPreview,
SkillHubScanResult,
SkillHubSearchResponse,
SkillHubSourcesResponse,
SkillInfo,
StarmapGraph
} from '@/types/hermes'
import type { ActionResponse } from '@/types/hermes'
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped } from './client'
export function getSkills(profile?: ProfileScope): Promise<SkillInfo[]> {
return window.hermesDesktop.api<SkillInfo[]>({
...capabilityScoped(profile),
path: '/api/skills'
})
}
/** Raw SKILL.md text (frontmatter included) for ANY skill — bundled, hub, or
* learned — backing the Capabilities detail pane's full-skill view. */
export function getSkillContent(
name: string,
profile?: ProfileScope
): Promise<{ content: string; name: string; path: string }> {
return window.hermesDesktop.api<{ content: string; name: string; path: string }>({
...capabilityScoped(profile),
path: `/api/skills/content?name=${encodeURIComponent(name)}`
})
}
export function setSkillEnabled(
name: string,
enabled: boolean,
profile?: ProfileScope
): Promise<{ ok: boolean; name: string; enabled: boolean }> {
return window.hermesDesktop.api<{ ok: boolean; name: string; enabled: boolean }>({
...capabilityScoped(profile),
path: '/api/skills/toggle',
method: 'PUT',
body: { name, enabled }
})
}
export function getStarmapGraph(): Promise<StarmapGraph> {
return hermesApi<StarmapGraph>({
...profileScoped(),
// Backend REST contract — stays /api/learning even though the UI feature is
// now "star map". Renaming this would break against an un-upgraded backend.
path: '/api/learning/graph'
})
}
export interface LearningNodeDetail {
content: string
kind: 'memory' | 'skill'
label: string
ok: boolean
}
export function getLearningNode(id: string, profile?: ProfileScope): Promise<LearningNodeDetail> {
return window.hermesDesktop.api<LearningNodeDetail>({
...capabilityScoped(profile),
path: `/api/learning/node?id=${encodeURIComponent(id)}`
})
}
export function deleteLearningNode(id: string, profile?: ProfileScope): Promise<{ message: string; ok: boolean }> {
return window.hermesDesktop.api<{ message: string; ok: boolean }>({
...capabilityScoped(profile),
path: '/api/learning/node',
method: 'DELETE',
body: { id }
})
}
export function editLearningNode(
id: string,
content: string,
profile?: ProfileScope
): Promise<{ message: string; ok: boolean }> {
return window.hermesDesktop.api<{ message: string; ok: boolean }>({
...capabilityScoped(profile),
path: '/api/learning/node',
method: 'PUT',
body: { content, id }
})
}
// ---------------------------------------------------------------------------
// Skills hub — search / preview / scan / install (parity with `hermes skills`
// and the dashboard's Browse-hub tab). Installs spawn background actions whose
// logs are tailed via getActionStatus().
// ---------------------------------------------------------------------------
const HUB_REQUEST_TIMEOUT_MS = 45_000
export function getSkillHubSources(profile?: null | string): Promise<SkillHubSourcesResponse> {
return hermesApi<SkillHubSourcesResponse>({
...profileScoped(profile),
path: '/api/skills/hub/sources',
timeoutMs: HUB_REQUEST_TIMEOUT_MS
})
}
export function searchSkillsHub(
query: string,
source = 'all',
limit = 20,
profile?: null | string
): Promise<SkillHubSearchResponse> {
const params = new URLSearchParams({ q: query, source, limit: String(limit) })
return hermesApi<SkillHubSearchResponse>({
...profileScoped(profile),
path: `/api/skills/hub/search?${params.toString()}`,
timeoutMs: HUB_REQUEST_TIMEOUT_MS
})
}
export function previewSkillHub(identifier: string, profile?: null | string): Promise<SkillHubPreview> {
return hermesApi<SkillHubPreview>({
...profileScoped(profile),
path: `/api/skills/hub/preview?identifier=${encodeURIComponent(identifier)}`,
timeoutMs: HUB_REQUEST_TIMEOUT_MS
})
}
export function scanSkillHub(identifier: string, profile?: null | string): Promise<SkillHubScanResult> {
return hermesApi<SkillHubScanResult>({
...profileScoped(profile),
path: `/api/skills/hub/scan?identifier=${encodeURIComponent(identifier)}`,
timeoutMs: HUB_REQUEST_TIMEOUT_MS
})
}
export function installSkillFromHub(identifier: string, profile?: ProfileScope): Promise<ActionResponse> {
return window.hermesDesktop.api<ActionResponse>({
...capabilityScoped(profile),
path: '/api/skills/hub/install',
method: 'POST',
body: { identifier }
})
}
export function uninstallSkillFromHub(name: string, profile?: ProfileScope): Promise<ActionResponse> {
return window.hermesDesktop.api<ActionResponse>({
...capabilityScoped(profile),
path: '/api/skills/hub/uninstall',
method: 'POST',
body: { name }
})
}
export function updateSkillsFromHub(profile?: ProfileScope): Promise<ActionResponse> {
return window.hermesDesktop.api<ActionResponse>({
...capabilityScoped(profile),
path: '/api/skills/hub/update',
method: 'POST',
body: {}
})
}
+247
View File
@@ -0,0 +1,247 @@
import type {
ActionResponse,
ActionStatusResponse,
AudioSpeakResponse,
AudioTranscriptionResponse,
BackendUpdateCheckResponse,
CuratorStatusResponse,
DebugShareResponse,
ElevenLabsVoicesResponse,
MemoryProviderConfig,
MemoryProviderOAuthStatus,
MemoryStatusResponse
} from '@/types/hermes'
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped } from './client'
export const AUDIO_SPEAK_MIN_REQUEST_TIMEOUT_MS = 180_000
export const AUDIO_SPEAK_MAX_REQUEST_TIMEOUT_MS = 600_000
const AUDIO_SPEAK_TIMEOUT_MS_PER_CHAR = 35
export function audioSpeakRequestTimeoutMs(text: string): number {
const estimated = Math.max(
AUDIO_SPEAK_MIN_REQUEST_TIMEOUT_MS,
Math.ceil(String(text || '').length * AUDIO_SPEAK_TIMEOUT_MS_PER_CHAR)
)
return Math.min(AUDIO_SPEAK_MAX_REQUEST_TIMEOUT_MS, estimated)
}
export const AUDIO_TRANSCRIBE_MIN_REQUEST_TIMEOUT_MS = 180_000
export const AUDIO_TRANSCRIBE_MAX_REQUEST_TIMEOUT_MS = 600_000
// The transcribe payload is the base64 audio data URL itself, so its string
// length tracks clip size. ~0.1ms/char keeps short clips at the floor while
// letting multi-minute recordings scale toward the cap (a base64 char is
// ~0.75 bytes, so at 128kbps ≈ 21k chars/s of audio this budgets ~2s of
// timeout per 1s of audio before the cap clamps it).
const AUDIO_TRANSCRIBE_TIMEOUT_MS_PER_CHAR = 0.1
export function audioTranscribeRequestTimeoutMs(dataUrl: string): number {
const estimated = Math.max(
AUDIO_TRANSCRIBE_MIN_REQUEST_TIMEOUT_MS,
Math.ceil(String(dataUrl || '').length * AUDIO_TRANSCRIBE_TIMEOUT_MS_PER_CHAR)
)
return Math.min(AUDIO_TRANSCRIBE_MAX_REQUEST_TIMEOUT_MS, estimated)
}
// surface=declared serves the curated desktop schema; the dashboard consumes the raw plugin schema.
export function getMemoryProviderConfig(provider: string, profile?: null | string): Promise<MemoryProviderConfig> {
return hermesApi<MemoryProviderConfig>({
...profileScoped(profile),
path: `/api/memory/providers/${encodeURIComponent(provider)}/config?surface=declared`
})
}
export function saveMemoryProviderConfig(
provider: string,
values: Record<string, string>,
profile?: null | string
): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(profile),
path: `/api/memory/providers/${encodeURIComponent(provider)}/config?surface=declared`,
method: 'PUT',
body: { values }
})
}
// Memory-provider OAuth connect (provider-keyed; 404s for providers without an
// OAuth flow). Profile-scoped: the grant lands in the active profile's config.
export function startMemoryProviderOAuth(
provider: string,
profile?: null | string
): Promise<MemoryProviderOAuthStatus> {
return hermesApi<MemoryProviderOAuthStatus>({
...profileScoped(profile),
path: `/api/memory/providers/${encodeURIComponent(provider)}/oauth/start`,
method: 'POST'
})
}
export function getMemoryProviderOAuthStatus(
provider: string,
profile?: null | string
): Promise<MemoryProviderOAuthStatus> {
return hermesApi<MemoryProviderOAuthStatus>({
...profileScoped(profile),
path: `/api/memory/providers/${encodeURIComponent(provider)}/oauth/status`
})
}
// ---------------------------------------------------------------------------
// Memory data + curator (parity with `hermes memory` / `hermes curator`).
// ---------------------------------------------------------------------------
export function getMemoryStatus(): Promise<MemoryStatusResponse> {
return hermesApi<MemoryStatusResponse>({
...profileScoped(),
path: '/api/memory'
})
}
export function resetMemory(target: 'all' | 'memory' | 'user'): Promise<{ ok: boolean; deleted: string[] }> {
return hermesApi<{ ok: boolean; deleted: string[] }>({
...profileScoped(),
path: '/api/memory/reset',
method: 'POST',
body: { target }
})
}
export function getCuratorStatus(): Promise<CuratorStatusResponse> {
return hermesApi<CuratorStatusResponse>({
...profileScoped(),
path: '/api/curator'
})
}
export function setCuratorPaused(paused: boolean): Promise<{ ok: boolean; paused: boolean }> {
return hermesApi<{ ok: boolean; paused: boolean }>({
...profileScoped(),
path: '/api/curator/paused',
method: 'PUT',
body: { paused }
})
}
export function runCurator(): Promise<ActionResponse> {
return hermesApi<ActionResponse>({
...profileScoped(),
path: '/api/curator/run',
method: 'POST',
body: {}
})
}
export function restartGateway(): Promise<ActionResponse> {
return hermesApi<ActionResponse>({
...profileScoped(),
path: '/api/gateway/restart',
method: 'POST'
})
}
export function updateHermes(): Promise<ActionResponse> {
return hermesApi<ActionResponse>({
...profileScoped(),
path: '/api/hermes/update',
method: 'POST'
})
}
/** Query the connected backend's own update state. In remote mode this is the
* authoritative source for the backend's behind-count + "what's changed",
* distinct from the Electron client clone's git state. */
export function checkHermesUpdate(force = false): Promise<BackendUpdateCheckResponse> {
return hermesApi<BackendUpdateCheckResponse>({
...profileScoped(),
path: `/api/hermes/update/check${force ? '?force=true' : ''}`
})
}
export function getActionStatus(name: string, lines = 200, profile?: ProfileScope): Promise<ActionStatusResponse> {
return window.hermesDesktop.api<ActionStatusResponse>({
...capabilityScoped(profile),
path: `/api/actions/${encodeURIComponent(name)}/status?lines=${Math.max(1, lines)}`
})
}
export function transcribeAudio(dataUrl: string, mimeType?: string): Promise<AudioTranscriptionResponse> {
return hermesApi<AudioTranscriptionResponse>({
path: '/api/audio/transcribe',
method: 'POST',
...profileScoped(),
body: {
data_url: dataUrl,
mime_type: mimeType
},
// Transcription blocks until provider STT, file handling, and response
// encoding finish. Remote providers and long clips regularly exceed the
// default 15s Electron backend timeout.
timeoutMs: audioTranscribeRequestTimeoutMs(dataUrl)
})
}
export function speakText(text: string): Promise<AudioSpeakResponse> {
return hermesApi<AudioSpeakResponse>({
...profileScoped(),
path: '/api/audio/speak',
method: 'POST',
body: { text },
// TTS blocks until provider synthesis, file read, and base64 encoding
// finish. Remote providers and large messages regularly exceed the
// default 15s Electron backend timeout.
timeoutMs: audioSpeakRequestTimeoutMs(text)
})
}
export function getElevenLabsVoices(profile?: null | string): Promise<ElevenLabsVoicesResponse> {
return hermesApi<ElevenLabsVoicesResponse>({
path: '/api/audio/elevenlabs/voices',
...profileScoped(profile)
})
}
/** `gh` CLI presence + auth state, for the composer's GitHub skill pill
* (GitHub is deliberately not an MCP — the github/* skills are the
* integration). Backend caches for 5 minutes; `refresh` bypasses. */
export function getGhAuthStatus(refresh = false): Promise<{ available: boolean; authenticated: boolean }> {
return hermesApi<{ available: boolean; authenticated: boolean }>({
...profileScoped(),
path: `/api/git/gh-auth${refresh ? '?refresh=true' : ''}`
})
}
// ---------------------------------------------------------------------------
// Maintenance operations (parity with `hermes doctor` / `hermes security
// audit` / `hermes backup` / `hermes debug share` and the dashboard System
// page). All except debug share are spawn-based background actions tailed via
// getActionStatus().
// ---------------------------------------------------------------------------
export function runDoctor(): Promise<ActionResponse> {
return hermesApi<ActionResponse>({ path: '/api/ops/doctor', method: 'POST', body: {} })
}
export function runSecurityAudit(): Promise<ActionResponse> {
return hermesApi<ActionResponse>({ path: '/api/ops/security-audit', method: 'POST', body: {} })
}
export function runBackup(): Promise<ActionResponse & { archive?: string }> {
return hermesApi<ActionResponse & { archive?: string }>({
path: '/api/ops/backup',
method: 'POST',
body: {}
})
}
export function runDebugShare(): Promise<DebugShareResponse> {
return hermesApi<DebugShareResponse>({
path: '/api/ops/debug-share',
method: 'POST',
body: {},
// Synchronous upload of report + logs to the paste service.
timeoutMs: 120_000
})
}
+141
View File
@@ -0,0 +1,141 @@
import type {
ActionResponse,
ComputerUseStatus,
TerminalBackendsResponse,
ToolsetConfig,
ToolsetInfo,
ToolsetModelsResponse
} from '@/types/hermes'
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped } from './client'
// The optional trailing `profile` on every capability fetcher below is the
// Capabilities view's profile-scope override: it lets the Skills/Tools/MCP
// panels configure ANY profile without swapping the app-wide active profile.
// Omitting it (every pre-existing caller) means `profileScoped(undefined)`
// falls back to the app-wide `_apiProfile`, so behavior is byte-identical.
export function getToolsets(profile?: ProfileScope): Promise<ToolsetInfo[]> {
return window.hermesDesktop.api<ToolsetInfo[]>({
...capabilityScoped(profile),
path: '/api/tools/toolsets'
})
}
export function setToolsetEnabled(
name: string,
enabled: boolean,
profile?: ProfileScope
): Promise<{ ok: boolean; name: string; enabled: boolean }> {
return window.hermesDesktop.api<{ ok: boolean; name: string; enabled: boolean }>({
...capabilityScoped(profile),
path: `/api/tools/toolsets/${encodeURIComponent(name)}`,
method: 'PUT',
body: { enabled }
})
}
export function getToolsetConfig(name: string, profile?: ProfileScope): Promise<ToolsetConfig> {
return window.hermesDesktop.api<ToolsetConfig>({
...capabilityScoped(profile),
path: `/api/tools/toolsets/${encodeURIComponent(name)}/config`
})
}
export function getToolsetModels(
name: string,
provider?: string,
profile?: ProfileScope
): Promise<ToolsetModelsResponse> {
const suffix = provider ? `?provider=${encodeURIComponent(provider)}` : ''
return window.hermesDesktop.api<ToolsetModelsResponse>({
...capabilityScoped(profile),
path: `/api/tools/toolsets/${encodeURIComponent(name)}/models${suffix}`
})
}
export function selectToolsetModel(
name: string,
model: string,
provider?: string,
profile?: ProfileScope
): Promise<{ ok: boolean; name: string; model: string }> {
return window.hermesDesktop.api<{ ok: boolean; name: string; model: string }>({
...capabilityScoped(profile),
path: `/api/tools/toolsets/${encodeURIComponent(name)}/model`,
method: 'PUT',
body: { model, provider }
})
}
export interface SelectToolsetProviderResponse {
ok: boolean
name: string
provider: string
/** Present when the selection was scoped to one web capability. */
capability?: string
/** Present (true) when a managed Nous row was selected but the Portal
* entitlement is missing — the row won't activate until the user signs
* in to Nous Portal. */
needs_nous_auth?: boolean
/** The managed feature key (e.g. "browser") when needs_nous_auth is set. */
feature?: string
}
export function selectToolsetProvider(
name: string,
provider: string,
capability?: 'search' | 'extract',
profile?: ProfileScope
): Promise<SelectToolsetProviderResponse> {
return window.hermesDesktop.api<SelectToolsetProviderResponse>({
...capabilityScoped(profile),
path: `/api/tools/toolsets/${encodeURIComponent(name)}/provider`,
method: 'PUT',
body: capability ? { provider, capability } : { provider }
})
}
export function runToolsetPostSetup(
name: string,
key: string,
profile?: ProfileScope
): Promise<ActionResponse & { key: string }> {
return window.hermesDesktop.api<ActionResponse & { key: string }>({
...capabilityScoped(profile),
path: `/api/tools/toolsets/${encodeURIComponent(name)}/post-setup`,
method: 'POST',
body: { key }
})
}
export function getTerminalBackends(): Promise<TerminalBackendsResponse> {
return hermesApi<TerminalBackendsResponse>({
...profileScoped(),
path: '/api/tools/terminal/backends'
})
}
export function selectTerminalBackend(backend: string): Promise<{ ok: boolean; backend: string }> {
return hermesApi<{ ok: boolean; backend: string }>({
...profileScoped(),
path: '/api/tools/terminal/backend',
method: 'PUT',
body: { backend }
})
}
export function getComputerUseStatus(): Promise<ComputerUseStatus> {
return hermesApi<ComputerUseStatus>({
...profileScoped(),
path: '/api/tools/computer-use/status'
})
}
export function grantComputerUsePermissions(): Promise<ActionResponse> {
return hermesApi<ActionResponse>({
...profileScoped(),
path: '/api/tools/computer-use/permissions/grant',
method: 'POST'
})
}
+4 -1
View File
@@ -492,7 +492,10 @@ function ArtifactImageCard({ artifact, failedImage, onImageError, onOpenChat }:
}, [artifact.href, artifact.id, artifact.value, onImageError])
return (
<article className="group/artifact overflow-hidden rounded-lg border border-(--ui-stroke-tertiary) bg-(--ui-chat-bubble-background)">
<article
className="group/artifact overflow-hidden rounded-lg border border-(--ui-stroke-tertiary) bg-(--ui-chat-bubble-background)"
data-tour="artifact-card"
>
<div
className={cn(
'relative flex h-40 w-full items-center justify-center overflow-hidden border-b border-(--ui-stroke-tertiary) bg-(--ui-bg-quinary) p-1.5',
@@ -22,6 +22,24 @@ export const COMPOSER_STACK_BREAKPOINT_PX = 320
// chevron frees is spent keeping the row single for another stretch.
export const COMPOSER_COMPACT_PILL_PX = 560
// The ladder keeps going below the stack breakpoint — a pane can be far
// narrower than even the stacked controls row. Both rungs are budgeted
// against that row's real cost: menu ~24 + surface padding 16 + the cluster
// (~190; ~218 mid-turn with the queue button).
//
// At 260 the three voice toggles fold into the one menu HUD mode already
// uses, clearing the mid-turn worst case with margin. Each stage sits clear
// of the floor below it rather than arriving the instant the previous one
// gives out — the mistake COMPOSER_COMPACT_PILL_PX documents.
export const COMPOSER_FOLD_VOICE_PX = 260
// Type and send, nothing else. A pane can be dragged to MIN_PANE_PX (80), and
// even with voice folded the row still costs ~150, so the last rung drops the
// pill AND the voice menu. Both stay reachable — the model by hotkey and the
// full picker, dictation from any wider pane — and Send fits with room to
// spare at any width the layout tree allows (~74 all-in).
export const COMPOSER_MINIMAL_PX = 180
// A single editor line is ~28px (--composer-input-min-height 1.625rem + 0.5rem
// vertical padding). Anything taller means the text wrapped to a second line,
// which is when the composer should expand to the stacked layout.
@@ -0,0 +1,25 @@
import { cn } from '@/lib/utils'
// Shared class names for the composer's control row, in a module of their own
// so both the row (`controls.tsx`) and the menus it renders can wear them
// without importing each other in a cycle.
export const ICON_BTN = 'size-(--composer-control-size) shrink-0 rounded-md'
export const GHOST_ICON_BTN = cn(
ICON_BTN,
'text-(--ui-text-tertiary) hover:bg-(--chrome-action-hover) hover:text-foreground'
)
// Send/voice-conversation primary: solid foreground-on-background circle
// (reads as black-on-white in light mode, white-on-black in dark mode) to
// match the reference composer's high-contrast CTA. Keeps the pill itself
// neutral and lets the action visually dominate the row.
export const PRIMARY_ICON_BTN = cn(
'size-(--composer-control-primary-size,var(--composer-control-size)) shrink-0 rounded-full p-0',
'bg-foreground text-background hover:bg-foreground/90',
'disabled:bg-foreground/30 disabled:text-background disabled:opacity-100'
)
/** A toggle that is currently ON — dictation, spoken replies, the wake word. */
export const ACTIVE_ICON_BTN = 'bg-primary/10 text-primary hover:bg-primary/15 hover:text-primary'
@@ -3,6 +3,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest'
import type { ChatBarState } from '@/app/chat/composer/types'
import { I18nProvider } from '@/i18n'
import { $hudMode } from '@/store/hud'
import { applyWakeStartResult, applyWakeStatus, resetWakeWordState } from '@/store/wake-word'
import { ComposerControls } from './controls'
@@ -57,6 +58,75 @@ async function expectShortcutTooltip(label: string, shortcut: string) {
afterEach(() => {
cleanup()
$hudMode.set(false)
})
// The HUD is a Spotlight bar a few hundred pixels wide: the four voice
// controls fold into one menu there, and the way out of HUD mode joins the
// row instead of floating above the bar in a reserved strip. The docked
// composer keeps every control inline and shows no exit.
describe('HUD mode', () => {
it('keeps the voice controls inline and offers no exit in the docked composer', () => {
renderControls()
expect(screen.getByLabelText('Voice dictation')).toBeTruthy()
expect(screen.getByLabelText('Read replies aloud')).toBeTruthy()
expect(screen.queryByLabelText('Exit HUD mode')).toBeNull()
expect(screen.queryByLabelText('Voice')).toBeNull()
})
it('folds them into one menu and offers the way out in the HUD', () => {
$hudMode.set(true)
renderControls()
expect(screen.getByLabelText('Voice')).toBeTruthy()
expect(screen.getByLabelText('Exit HUD mode')).toBeTruthy()
// Folded away, not duplicated — the whole point is the row's width back.
expect(screen.queryByLabelText('Voice dictation')).toBeNull()
expect(screen.queryByLabelText('Read replies aloud')).toBeNull()
})
// A collapsed menu that looked idle while the mic was open would be a worse
// trade than the space it saves, so the trigger reports the live state.
it('reports a live voice state on the collapsed trigger', () => {
$hudMode.set(true)
renderControls({ voiceStatus: 'recording' })
expect(screen.getByLabelText('Stop dictation')).toBeTruthy()
expect(screen.queryByLabelText('Voice')).toBeNull()
})
})
// A tile can be narrower than the controls cost, and the row is inside an
// overflow-hidden surface — so anything that doesn't fold gets clipped off the
// right edge, send button first. The ladder keeps going past `stacked`: voice
// folds into the same menu the HUD uses, then the model pill drops. Send is
// the last thing standing.
describe('narrow tiles', () => {
it('folds the voice controls into one menu without entering HUD mode', () => {
renderControls({ foldVoice: true })
expect(screen.getByLabelText('Voice')).toBeTruthy()
expect(screen.queryByLabelText('Voice dictation')).toBeNull()
expect(screen.queryByLabelText('Read replies aloud')).toBeNull()
// Folding is a width decision, not the HUD: no exit affordance appears.
expect(screen.queryByLabelText('Exit HUD mode')).toBeNull()
})
it('keeps Send at the tightest width, with everything else dropped', () => {
renderControls({ foldVoice: true, minimal: true })
expect(screen.getByLabelText('Send')).toBeTruthy()
expect(screen.queryByLabelText('Voice')).toBeNull()
})
it('keeps Stop reachable mid-turn at the tightest width', () => {
renderControls({ busy: true, busyAction: 'stop', foldVoice: true, hasComposerPayload: false, minimal: true })
expect(screen.getByLabelText('Stop')).toBeTruthy()
})
})
describe('ComposerControls shortcut tooltips', () => {
+71 -28
View File
@@ -7,26 +7,18 @@ import { useI18n } from '@/i18n'
import { triggerHaptic } from '@/lib/haptics'
import { AudioLines, Ear, EarOff, iconSize, Layers3, Loader2, Square, Volume2, VolumeX } from '@/lib/icons'
import { cn } from '@/lib/utils'
import { $hudMode, closeHud } from '@/store/hud'
import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
import { ACTIVE_ICON_BTN, GHOST_ICON_BTN, PRIMARY_ICON_BTN } from './control-classes'
import type { ConversationStatus } from './hooks/use-voice-conversation'
import { ModelPill } from './model-pill'
import type { ChatBarState, VoiceStatus } from './types'
import { VoiceMenu } from './voice-menu'
export const ICON_BTN = 'size-(--composer-control-size) shrink-0 rounded-md'
export const GHOST_ICON_BTN = cn(
ICON_BTN,
'text-(--ui-text-tertiary) hover:bg-(--chrome-action-hover) hover:text-foreground'
)
// Send/voice-conversation primary: solid foreground-on-background circle
// (reads as black-on-white in light mode, white-on-black in dark mode) to
// match the reference composer's high-contrast CTA. Keeps the pill itself
// neutral and lets the action visually dominate the row.
export const PRIMARY_ICON_BTN = cn(
'size-(--composer-control-primary-size,var(--composer-control-size)) shrink-0 rounded-full p-0',
'bg-foreground text-background hover:bg-foreground/90',
'disabled:bg-foreground/30 disabled:text-background disabled:opacity-100'
)
// Re-exported: `context-menu.tsx` and other row neighbours have always reached
// for these here, and the row is where they read as belonging.
export { ACTIVE_ICON_BTN, GHOST_ICON_BTN, ICON_BTN, PRIMARY_ICON_BTN } from './control-classes'
interface ConversationProps {
active: boolean
@@ -47,7 +39,9 @@ export function ComposerControls({
compactModelPill = false,
conversation,
disabled,
foldVoice = false,
hasComposerPayload,
minimal = false,
state,
voiceStatus,
onDictate,
@@ -61,7 +55,9 @@ export function ComposerControls({
compactModelPill?: boolean
conversation: ConversationProps
disabled: boolean
foldVoice?: boolean
hasComposerPayload: boolean
minimal?: boolean
state: ChatBarState
voiceStatus: VoiceStatus
onDictate: () => void
@@ -70,6 +66,7 @@ export function ComposerControls({
}) {
const { t } = useI18n()
const c = t.composer
const hudMode = useStore($hudMode)
if (conversation.active) {
return <ConversationPill {...conversation} disabled={disabled} />
@@ -80,13 +77,40 @@ export function ComposerControls({
// only when the composer is empty and a turn is running.
const showStop = busy && !hasComposerPayload
const showQueueButton = busyAction !== 'stop' && hasComposerPayload
// The HUD is a Spotlight bar a few hundred pixels wide, so the four separate
// voice toggles fold into one menu there and leave the row to the input. A
// narrow tile hits the same wall from the other direction and folds for the
// same reason — same controls, same state, different budget. Below that
// even the menu goes: at `minimal` the row is the send button and nothing
// else, which is the one thing that must survive every width.
const foldedVoice = hudMode || foldVoice
return (
<div className="ml-auto flex shrink-0 items-center gap-(--composer-control-gap)">
<ModelPill compact={compactModelPill} disabled={disabled} model={state.model} />
const voiceControls = foldedVoice ? (
<VoiceMenu
autoSpeak={autoSpeak}
disabled={disabled}
onDictate={onDictate}
onStartConversation={conversation.onStart}
onToggleAutoSpeak={onToggleAutoSpeak}
state={state}
voiceStatus={voiceStatus}
/>
) : (
<>
<DictationButton disabled={disabled} onToggle={onDictate} state={state.voice} status={voiceStatus} />
<AutoSpeakButton active={autoSpeak} disabled={disabled} onToggle={onToggleAutoSpeak} />
<WakeWordButton disabled={disabled} />
</>
)
return (
<div className="ml-auto flex min-w-0 shrink items-center gap-(--composer-control-gap)">
{minimal ? null : (
<>
<ModelPill compact={compactModelPill} disabled={disabled} model={state.model} />
{voiceControls}
</>
)}
{showQueueButton ? (
<Tip label={<TipKeybindLabel actionId="composer.queue" text={c.queueMessage} />}>
<Button
@@ -142,10 +166,37 @@ export function ComposerControls({
</Button>
</Tip>
)}
{/* The way out of HUD mode, riding the controls row rather than floating
above the bar. The old chip lived in a 26px transparent strip reserved
over the composer (--hud-chip-strip), which under glass is bare
untinted material with a hidden button in it — a band of chrome above
the surface, paid for in every state, for a control that is invisible
until hovered. Here it costs no reserved space and sits with the other
things you can press. */}
{hudMode ? <ExitHudButton /> : null}
</div>
)
}
function ExitHudButton() {
const { t } = useI18n()
return (
<Tip label={t.titlebar.exitHud}>
<Button
aria-label={t.titlebar.exitHud}
className={cn(GHOST_ICON_BTN, 'p-0')}
onClick={closeHud}
size="icon"
type="button"
variant="ghost"
>
<Codicon name="screen-normal" size="0.875rem" />
</Button>
</Tip>
)
}
function ConversationPill({
disabled,
level,
@@ -269,11 +320,7 @@ function AutoSpeakButton({ active, disabled, onToggle }: { active: boolean; disa
<Button
aria-label={label}
aria-pressed={active}
className={cn(
GHOST_ICON_BTN,
'p-0',
active && 'bg-primary/10 text-primary hover:bg-primary/15 hover:text-primary'
)}
className={cn(GHOST_ICON_BTN, 'p-0', active && ACTIVE_ICON_BTN)}
disabled={disabled}
onClick={() => {
triggerHaptic(active ? 'close' : 'open')
@@ -318,11 +365,7 @@ function WakeWordButton({ disabled, pausedForVoice = false }: { disabled: boolea
<Button
aria-label={label}
aria-pressed={wake.listening && !pausedForVoice}
className={cn(
GHOST_ICON_BTN,
'p-0',
wake.listening && !pausedForVoice && 'bg-primary/10 text-primary hover:bg-primary/15 hover:text-primary'
)}
className={cn(GHOST_ICON_BTN, 'p-0', wake.listening && !pausedForVoice && ACTIVE_ICON_BTN)}
disabled={disabled || pausedForVoice || wake.pending}
onClick={() => {
triggerHaptic(wake.listening ? 'close' : 'open')
@@ -365,7 +408,7 @@ function DictationButton({
GHOST_ICON_BTN,
'p-0',
'data-[active=true]:bg-accent data-[active=true]:text-foreground',
status === 'recording' && 'bg-primary/10 text-primary hover:bg-primary/15 hover:text-primary',
status === 'recording' && ACTIVE_ICON_BTN,
status === 'transcribing' && 'bg-primary/10 text-primary'
)}
data-active={active}
+59 -6
View File
@@ -67,6 +67,8 @@ const cssEscape = (value: string): string => {
}
interface SubmitDetail {
/** Unique mounted composer surface captured at click time. */
surfaceId: string
target: ComposerTarget
text: string
/** `hidden` types the persisted user row so no bubble renders — the
@@ -155,6 +157,38 @@ const dispatch = <T>(name: string, detail: T) => {
window.setTimeout(() => window.dispatchEvent(new CustomEvent<T>(name, { detail })), 0)
}
/** Submit is the one bus mutation that must preserve the chat visible at click
* time. Deferring it lets a parent click handler/tab reveal switch the active
* keep-alive pane before subscribers run, so the task is dropped or claimed by
* another composer. Other bus events intentionally defer for focus restoration.
*/
const dispatchNow = <T>(name: string, detail: T) => {
if (typeof window !== 'undefined') {
window.dispatchEvent(new CustomEvent<T>(name, { detail }))
}
}
/** Unique identity for the visible composer surface addressed by a submit. */
const getVisibleComposerSurfaceId = (target: ComposerTarget): string | null => {
if (typeof document === 'undefined') {
return null
}
const surface = queryVisible<HTMLElement>(`[data-composer-target="${cssEscape(target)}"]`)
return surface?.dataset.composerSurfaceId || null
}
const composerSurfaceIsVisible = (target: ComposerTarget, surfaceId: string): boolean => {
if (typeof document === 'undefined') {
return false
}
return queryAllVisible<HTMLElement>(`[data-composer-target="${cssEscape(target)}"]`).some(
surface => surface.dataset.composerSurfaceId === surfaceId
)
}
const subscribe = <T>(name: string, handler: (detail: T) => void) => {
if (typeof window === 'undefined') {
return () => undefined
@@ -264,17 +298,36 @@ export const onComposerInsertRefsRequest = (handler: (detail: InsertRefsDetail)
* the agent a task without the user round-tripping through the input. */
export const requestComposerSubmit = (
text: string,
{ target = 'active', displayKind }: { target?: ComposerTarget | 'active'; displayKind?: 'hidden' } = {}
) => {
{
displayKind,
surfaceId: requestedSurfaceId,
target = 'active'
}: { displayKind?: 'hidden'; surfaceId?: null | string; target?: ComposerTarget | 'active' } = {}
): boolean => {
const trimmed = text.trim()
if (trimmed) {
dispatch<SubmitDetail>(SUBMIT_EVENT, {
target: resolve(target),
if (!trimmed) {
return false
}
const resolvedTarget = resolve(target)
const surfaceId = requestedSurfaceId === undefined ? getVisibleComposerSurfaceId(resolvedTarget) : requestedSurfaceId
// Fail closed: without an exact visible surface identity, broadcasting a
// submit could make more than one keep-alive/new-chat composer claim it.
if (!surfaceId || (requestedSurfaceId !== undefined && !composerSurfaceIsVisible(resolvedTarget, surfaceId))) {
return false
}
dispatchNow<SubmitDetail>(SUBMIT_EVENT, {
surfaceId,
target: resolvedTarget,
text: trimmed,
...(displayKind ? { displayKind } : {})
})
}
return true
}
export const onComposerSubmitRequest = (handler: (detail: SubmitDetail) => void) =>
@@ -10,7 +10,13 @@ import {
} from '@/app/chat/surface-vars'
import { useResizeObserver } from '@/hooks/use-resize-observer'
import { COMPOSER_COMPACT_PILL_PX, COMPOSER_SINGLE_LINE_MAX_PX, COMPOSER_STACK_BREAKPOINT_PX } from '../composer-utils'
import {
COMPOSER_COMPACT_PILL_PX,
COMPOSER_FOLD_VOICE_PX,
COMPOSER_MINIMAL_PX,
COMPOSER_SINGLE_LINE_MAX_PX,
COMPOSER_STACK_BREAKPOINT_PX
} from '../composer-utils'
interface UseComposerMetricsArgs {
composerDockRef: RefObject<HTMLDivElement | null>
@@ -20,13 +26,36 @@ interface UseComposerMetricsArgs {
poppedOut: boolean
}
/** Every width-driven collapse stage, resolved from the composer's own width. */
export interface ComposerFit {
compactPill: boolean
foldVoice: boolean
minimal: boolean
tight: boolean
}
const ROOMY: ComposerFit = { compactPill: false, foldVoice: false, minimal: false, tight: false }
const fitForWidth = (width: number): ComposerFit => ({
compactPill: width < COMPOSER_COMPACT_PILL_PX,
foldVoice: width < COMPOSER_FOLD_VOICE_PX,
minimal: width < COMPOSER_MINIMAL_PX,
tight: width < COMPOSER_STACK_BREAKPOINT_PX
})
const sameFit = (a: ComposerFit, b: ComposerFit) =>
a.compactPill === b.compactPill && a.foldVoice === b.foldVoice && a.minimal === b.minimal && a.tight === b.tight
interface UseComposerMetricsResult extends ComposerFit {
stacked: boolean
}
/**
* Owns the composer's *sizing* engine: the stacked-vs-inline layout decision
* and the measured-height CSS vars the thread reads for bottom clearance. All
* work is edge-gated — the ResizeObserver only fires on real size changes, the
* height vars are 8px-bucketed so per-keystroke growth never invalidates the
* tree's computed style, and `tight` only flips when it crosses the breakpoint.
* Returns `stacked` (the only value the render needs).
* tree's computed style, and the fit only re-renders when it crosses a stage.
*/
export function useComposerMetrics({
composerDockRef,
@@ -34,14 +63,9 @@ export function useComposerMetrics({
composerSurfaceRef,
editorRef,
poppedOut
}: UseComposerMetricsArgs): {
compactPill: boolean
stacked: boolean
} {
}: UseComposerMetricsArgs): UseComposerMetricsResult {
const [expanded, setExpanded] = useState(false)
const [tight, setTight] = useState(false)
// Wider than `tight`: the pill goes icon-only before the row has to stack.
const [compactPill, setCompactPill] = useState(false)
const [fit, setFit] = useState<ComposerFit>(ROOMY)
// Edge signals, not the live text: these only re-render when emptiness / the
// presence of a non-trailing newline actually flips, so typing within a line
@@ -84,8 +108,7 @@ export function useComposerMetrics({
// until a wrap or row change actually happens.
const lastBucketedHeightRef = useRef(0)
const lastBucketedSurfaceHeightRef = useRef(0)
const lastTightRef = useRef<boolean | null>(null)
const lastCompactPillRef = useRef<boolean | null>(null)
const lastFitRef = useRef(ROOMY)
// Mirrored into a ref so `syncComposerMetrics` stays referentially stable —
// it's the shared ResizeObserver's handler, and a new identity every render
// would re-register the observation.
@@ -121,18 +144,11 @@ export function useComposerMetrics({
const surfaceHeight = composerSurfaceRef.current?.getBoundingClientRect().height
if (width > 0) {
const nextTight = width < COMPOSER_STACK_BREAKPOINT_PX
const nextFit = fitForWidth(width)
if (nextTight !== lastTightRef.current) {
lastTightRef.current = nextTight
setTight(nextTight)
}
const nextCompactPill = width < COMPOSER_COMPACT_PILL_PX
if (nextCompactPill !== lastCompactPillRef.current) {
lastCompactPillRef.current = nextCompactPill
setCompactPill(nextCompactPill)
if (!sameFit(nextFit, lastFitRef.current)) {
lastFitRef.current = nextFit
setFit(nextFit)
}
}
@@ -189,7 +205,7 @@ export function useComposerMetrics({
}
}, [composerRef])
// Both decisions come from the composer's OWN measured width, never the
// Every decision comes from the composer's OWN measured width, never the
// viewport's. There used to be a `(max-width: 30rem)` media query in here as
// well, and it quietly outranked everything: any window under 480px stacked
// the row AND compacted the pill in the same instant, regardless of how much
@@ -199,7 +215,14 @@ export function useComposerMetrics({
// stack) by 160px. The ResizeObserver knows the real width; the viewport is
// not a proxy for it.
//
// The pill still compacts whenever the row stacks, so the controls row can't
// over-run once it has the width to itself.
return { compactPill: compactPill || tight, stacked: expanded || tight }
// The ladder is monotonic: each stage implies the ones above it, so the pill
// is always compact by the time the row stacks, and the voice controls are
// always folded before minimal drops them.
return {
compactPill: fit.compactPill || fit.tight,
foldVoice: fit.foldVoice || fit.minimal,
minimal: fit.minimal,
stacked: expanded || fit.tight,
tight: fit.tight
}
}
@@ -1,6 +1,8 @@
import { act, cleanup, renderHook, waitFor } from '@testing-library/react'
import { type Dispatch, type PropsWithChildren, type SetStateAction, useLayoutEffect, useState } from 'react'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { PaneVisibleContext } from '@/components/pane-shell/pane-visibility'
import { $clarifyRequests } from '@/store/clarify'
import type { ComposerAttachment } from '@/store/composer'
import { $gateway } from '@/store/gateway'
@@ -12,23 +14,41 @@ import {
setSudoRequest
} from '@/store/prompts'
import { type ComposerTarget, requestComposerSubmit } from '../focus'
import { ComposerScopeProvider, ComposerSurfaceProvider, MAIN_COMPOSER_SCOPE } from '../scope'
import { useComposerSubmit } from './use-composer-submit'
interface SubmitHarnessOptions {
attachments?: ComposerAttachment[]
busy?: boolean
compacting?: boolean
inputDisabled?: boolean
scopeTarget?: ComposerTarget
sessionKey?: string | null
submitOnHide?: boolean
surfaceId?: string | null
text?: string
visible?: boolean
}
let surfaceSequence = 0
function renderSubmitHook({
attachments = [],
busy = false,
compacting = false,
text = ''
inputDisabled = false,
scopeTarget = 'main',
sessionKey = 'stored-session',
submitOnHide = false,
surfaceId,
text = '',
visible = true
}: SubmitHarnessOptions = {}) {
const resolvedSurfaceId = surfaceId === undefined ? `test-surface-${++surfaceSequence}` : surfaceId
const draftRef = { current: text }
const editor = document.createElement('div')
const editor = window.document.createElement('div')
editor.dataset.slot = 'composer-rich-input'
editor.textContent = text
const editorRef = { current: editor }
@@ -36,16 +56,45 @@ function renderSubmitHook({
const onSteer = vi.fn(async () => true)
const onSubmit = vi.fn(async () => true)
const queueCurrentDraft = vi.fn(() => true)
let updatePaneVisible: Dispatch<SetStateAction<boolean>> | undefined
const clearDraft = vi.fn(() => {
draftRef.current = ''
editorRef.current!.textContent = ''
})
const hook = renderHook(() =>
const Wrapper = ({ children }: PropsWithChildren) => {
const [paneVisible, setPaneVisible] = useState(visible)
updatePaneVisible = setPaneVisible
useLayoutEffect(() => {
if (submitOnHide && !paneVisible) {
requestComposerSubmit('ship while hiding', { target: scopeTarget })
}
}, [paneVisible])
return (
<ComposerScopeProvider value={{ ...MAIN_COMPOSER_SCOPE, target: scopeTarget }}>
<ComposerSurfaceProvider value={resolvedSurfaceId}>
<PaneVisibleContext.Provider value={paneVisible}>
<div
data-composer-surface-id={resolvedSurfaceId ?? undefined}
data-composer-target={scopeTarget}
data-pane-hidden={paneVisible ? undefined : ''}
>
{children}
</div>
</PaneVisibleContext.Provider>
</ComposerSurfaceProvider>
</ComposerScopeProvider>
)
}
const hook = renderHook(
() =>
useComposerSubmit({
activeQueueSessionKey: 'stored-session',
activeQueueSessionKeyRef: { current: 'stored-session' },
activeQueueSessionKey: sessionKey,
activeQueueSessionKeyRef: { current: sessionKey },
attachments,
busy,
compacting,
@@ -56,7 +105,7 @@ function renderSubmitHook({
editorRef,
exitQueuedEdit: vi.fn(() => false),
focusInput: vi.fn(),
inputDisabled: false,
inputDisabled,
loadIntoComposer: vi.fn(),
onCancel,
onSteer,
@@ -67,12 +116,154 @@ function renderSubmitHook({
sessionId: 'runtime-session',
setComposerText: vi.fn(),
stashAt: vi.fn()
})
}),
{ wrapper: Wrapper }
)
return { clearDraft, hook, onCancel, onSteer, onSubmit, queueCurrentDraft }
return {
clearDraft,
hook,
onCancel,
onSteer,
onSubmit,
queueCurrentDraft,
composerSurfaceId: resolvedSurfaceId,
setPaneVisible(nextVisible: boolean) {
if (!updatePaneVisible) {
throw new Error('Pane visibility setter was not initialized')
}
updatePaneVisible(nextVisible)
}
}
}
describe('useComposerSubmit external request routing', () => {
afterEach(() => {
cleanup()
vi.restoreAllMocks()
})
it('does not fan out a main ship across keep-alives or other projects', async () => {
const visibleMain = renderSubmitHook({ sessionKey: 'session-a' })
const hiddenMain = renderSubmitHook({ sessionKey: 'session-b', visible: false })
const visibleTile = renderSubmitHook({ scopeTarget: 'tile:project-b', sessionKey: 'tile-session' })
const hiddenTile = renderSubmitHook({
scopeTarget: 'tile:project-c',
sessionKey: 'other-tile',
visible: false
})
expect(requestComposerSubmit('ship this branch', { target: 'main' })).toBe(true)
await waitFor(() =>
expect(visibleMain.onSubmit).toHaveBeenCalledWith('ship this branch', {
composerScope: 'session-a'
})
)
expect(visibleMain.onSubmit).toHaveBeenCalledTimes(1)
expect(hiddenMain.onSubmit).not.toHaveBeenCalled()
expect(visibleTile.onSubmit).not.toHaveBeenCalled()
expect(hiddenTile.onSubmit).not.toHaveBeenCalled()
})
it('routes a tile-targeted submit to that tile only', async () => {
const main = renderSubmitHook({ sessionKey: 'main-session' })
const tile = renderSubmitHook({ scopeTarget: 'tile:project-b', sessionKey: 'tile-session' })
expect(requestComposerSubmit('ship project B', { target: 'tile:project-b' })).toBe(true)
await waitFor(() =>
expect(tile.onSubmit).toHaveBeenCalledWith('ship project B', {
composerScope: 'tile-session'
})
)
expect(main.onSubmit).not.toHaveBeenCalled()
})
it('uses the captured surface id when two visible composers share a target', async () => {
const first = renderSubmitHook({ sessionKey: 'session-first' })
const second = renderSubmitHook({ sessionKey: 'session-second' })
requestComposerSubmit('ship exactly one session', { surfaceId: second.composerSurfaceId, target: 'main' })
await waitFor(() =>
expect(second.onSubmit).toHaveBeenCalledWith('ship exactly one session', {
composerScope: 'session-second'
})
)
expect(first.onSubmit).not.toHaveBeenCalled()
})
it('submits to the session visible at click time even when the same click switches tabs', async () => {
const hiddenA = renderSubmitHook({ sessionKey: 'session-a', visible: false })
const visibleB = renderSubmitHook({ sessionKey: 'session-b' })
act(() => {
requestComposerSubmit('ship session B', { target: 'main' })
visibleB.setPaneVisible(false)
hiddenA.setPaneVisible(true)
})
await waitFor(() =>
expect(visibleB.onSubmit).toHaveBeenCalledWith('ship session B', {
composerScope: 'session-b'
})
)
expect(hiddenA.onSubmit).not.toHaveBeenCalled()
})
it('does not fan out when visible composers do not have queue session keys yet', async () => {
const firstNewSession = renderSubmitHook({ sessionKey: null })
const secondNewSession = renderSubmitHook({ sessionKey: null })
act(() => {
requestComposerSubmit('ship the visible new session', { target: 'main' })
})
await waitFor(() => expect(firstNewSession.onSubmit).toHaveBeenCalledTimes(1))
expect(secondNewSession.onSubmit).not.toHaveBeenCalled()
})
it('fails closed when the visible composer has no surface identity', () => {
const unidentified = renderSubmitHook({ surfaceId: null })
expect(requestComposerSubmit('do not broadcast this', { target: 'main' })).toBe(false)
expect(unidentified.onSubmit).not.toHaveBeenCalled()
})
it('fails closed when a pinned origin surface is no longer visible', () => {
const hidden = renderSubmitHook({ sessionKey: 'hidden-origin', visible: false })
const visible = renderSubmitHook({ sessionKey: 'other-visible' })
expect(
requestComposerSubmit('do not send to a stale origin', {
surfaceId: hidden.composerSurfaceId,
target: 'main'
})
).toBe(false)
expect(hidden.onSubmit).not.toHaveBeenCalled()
expect(visible.onSubmit).not.toHaveBeenCalled()
})
it('does not submit through a composer whose pane is hidden during the request', () => {
const main = renderSubmitHook({ submitOnHide: true })
act(() => main.setPaneVisible(false))
expect(main.onSubmit).not.toHaveBeenCalled()
})
it('does not submit through a disabled composer', () => {
const disabled = renderSubmitHook({ inputDisabled: true })
requestComposerSubmit('do not send this', { target: 'main' })
expect(disabled.onSubmit).not.toHaveBeenCalled()
})
})
describe('useComposerSubmit busy-turn routing', () => {
afterEach(() => {
cleanup()
@@ -1,5 +1,6 @@
import { type RefObject, useEffect, useRef } from 'react'
import { type RefObject, useLayoutEffect, useRef } from 'react'
import { usePaneVisible } from '@/components/pane-shell/pane-visibility'
import { SLASH_COMMAND_RE } from '@/lib/chat-runtime'
import { triggerHaptic } from '@/lib/haptics'
import { hasClarifyRequest, skipClarifyRequest } from '@/store/clarify'
@@ -13,7 +14,7 @@ import { cloneAttachments, type QueueEditState } from '../composer-utils'
import { onComposerSubmitRequest } from '../focus'
import { pathifyRefs } from '../path-refs'
import { composerPlainText } from '../rich-editor'
import { useComposerScope } from '../scope'
import { useComposerScope, useComposerSurfaceId } from '../scope'
import type { ChatBarProps } from '../types'
interface UseComposerSubmitArgs {
@@ -76,7 +77,9 @@ export function useComposerSubmit({
setComposerText,
stashAt
}: UseComposerSubmitArgs) {
const paneVisible = usePaneVisible()
const scope = useComposerScope()
const surfaceId = useComposerSurfaceId()
// Shared send primitive: fire onSubmit, and if the gateway rejects (accepted
// === false) or throws, re-load + re-stash the draft so the words survive.
@@ -103,19 +106,26 @@ export function useComposerSubmit({
}
// External "submit this prompt" requests (e.g. the review pane's agent-ship
// button) route through the same send path. A ref keeps the listener stable
// while always calling the latest dispatchSubmit closure.
// button) route through the same send path. Match both the composer target
// and the exact visible surface captured at click time — every tile stays
// mounted, and a session can be rendered in more than one pane.
const dispatchSubmitRef = useRef(dispatchSubmit)
dispatchSubmitRef.current = dispatchSubmit
useEffect(
useLayoutEffect(
() =>
onComposerSubmitRequest(({ target, text, displayKind }) => {
if (target === 'main' && !inputDisabled) {
onComposerSubmitRequest(({ surfaceId: requestedSurfaceId, target, text, displayKind }) => {
if (
target === scope.target &&
surfaceId !== null &&
requestedSurfaceId === surfaceId &&
paneVisible &&
!inputDisabled
) {
dispatchSubmitRef.current(text, undefined, displayKind)
}
}),
[inputDisabled]
[inputDisabled, paneVisible, scope.target, surfaceId]
)
const submitDraft = () => {
@@ -36,6 +36,26 @@ import { detectTrigger, textBeforeCaret, type TriggerState } from '../text-utils
* prose off and stranded a partial `folder:` in front of the chip, because the
* window it removed wasn't the token the user was typing.
*/
/** The keyup half of trigger detection, shared by both composers.
*
* If the open popover already consumed this key in keydown (Arrow/Enter/Tab/
* Escape), skip the refresh: those keys never edit text, and for Escape the
* keydown already closed the menu — refreshing here would re-detect the
* still-present `/` and instantly reopen it. It reads a ref set during keydown
* rather than `trigger`, because by keyup time React has re-rendered and
* `trigger` may already be null. */
export function triggerKeyUpHandler(consumedRef: MutableRefObject<boolean>, refreshTrigger: () => void) {
return () => {
if (consumedRef.current) {
consumedRef.current = false
return
}
window.setTimeout(refreshTrigger, 0)
}
}
export function rebuildAroundCaret(editor: HTMLDivElement, tokenLength: number, insert: DocumentFragment | string) {
const current = composerPlainText(editor)
const caret = caretOffsetInEditor(editor)
@@ -2,6 +2,8 @@ import { render } from '@testing-library/react'
import { createRef, type RefObject } from 'react'
import { describe, expect, it, vi } from 'vitest'
import { placeCaretAtEnd } from '../test-utils'
import { useComposerUndo } from './use-composer-undo'
/** Mount the hook against a real contentEditable, exposing its API. */
@@ -36,19 +38,10 @@ function makeEditor(text: string) {
return { editor, ref }
}
const caretAtEnd = (editor: HTMLElement) => {
const range = document.createRange()
const selection = window.getSelection()!
range.selectNodeContents(editor)
range.collapse(false)
selection.removeAllRanges()
selection.addRange(range)
}
describe('useComposerUndo', () => {
it('restores the pre-edit text, which is what a paste destroyed', () => {
const { editor, ref } = makeEditor('before')
caretAtEnd(editor)
placeCaretAtEnd(editor)
const { api, view } = mountUndo(ref, () => editor.textContent || '')
@@ -69,7 +62,7 @@ describe('useComposerUndo', () => {
it('withUndoPoint banks only when the edit actually ran', () => {
const { editor, ref } = makeEditor('text')
caretAtEnd(editor)
placeCaretAtEnd(editor)
const { api, view } = mountUndo(ref, () => editor.textContent || '')
@@ -95,7 +88,7 @@ describe('useComposerUndo', () => {
it('claims a native historyUndo aimed at the focused editor', () => {
const { editor, ref } = makeEditor('kept')
editor.focus()
caretAtEnd(editor)
placeCaretAtEnd(editor)
const { api, view } = mountUndo(ref, () => editor.textContent || '')
@@ -162,7 +155,7 @@ describe('useComposerUndo', () => {
it('reset drops history so undo cannot cross a draft swap', () => {
const { editor, ref } = makeEditor('session A')
caretAtEnd(editor)
placeCaretAtEnd(editor)
const { api, view } = mountUndo(ref, () => editor.textContent || '')
@@ -4,6 +4,7 @@ import { useCallback, useEffect, useRef, useState } from 'react'
import { useI18n } from '@/i18n'
import { chatMessageText, collectUnspokenTurnSpeech } from '@/lib/chat-messages'
import { triggerHaptic } from '@/lib/haptics'
import { markAssistantIdSpoken, resolveSpokenReply } from '@/lib/spoken-reply'
import { clearWakeIndicator, syncWakeIndicatorWithVoice } from '@/lib/wake-indicator'
import { $voiceConversationStartRequest, takeVoiceConversationStart } from '@/store/composer'
import { resetBrowseState } from '@/store/composer-input-history'
@@ -62,7 +63,6 @@ export function useComposerVoice({
// A tile's composer speaks ITS transcript, not the primary chat's.
const { $messages } = useComposerScope()
const [voiceConversationActive, setVoiceConversationActive] = useState(false)
const lastSpokenIdRef = useRef<string | null>(null)
const ownsWakeIndicatorRef = useRef(false)
const voiceStartRequest = useStore($voiceConversationStartRequest)
@@ -77,8 +77,9 @@ export function useComposerVoice({
const pendingResponse = () => {
const messages = $messages.get()
const last = messages.findLast(m => m.role === 'assistant' && !m.hidden)
const spoken = resolveSpokenReply(sessionId, messages)
if (!last || last.id === lastSpokenIdRef.current) {
if (!last || last.id === spoken?.id) {
return null
}
@@ -100,14 +101,18 @@ export function useComposerVoice({
* in order — narration interims AND the final answer, not just whichever
* bubble happens to be last. See `collectUnspokenTurnSpeech`.
*/
const pendingTurnResponse = () => collectUnspokenTurnSpeech($messages.get(), lastSpokenIdRef.current)
const pendingTurnResponse = () => {
const messages = $messages.get()
return collectUnspokenTurnSpeech(messages, resolveSpokenReply(sessionId, messages)?.id ?? null)
}
const consumePendingResponse = () => {
const messages = $messages.get()
const last = messages.findLast(m => m.role === 'assistant' && !m.hidden)
if (last) {
lastSpokenIdRef.current = last.id
markAssistantIdSpoken(sessionId, messages, last.id)
}
}
+14 -20
View File
@@ -50,7 +50,7 @@ import { useComposerPlaceholder } from './hooks/use-composer-placeholder'
import { useComposerPopout } from './hooks/use-composer-popout'
import { useComposerQueue } from './hooks/use-composer-queue'
import { useComposerSubmit } from './hooks/use-composer-submit'
import { useComposerTrigger } from './hooks/use-composer-trigger'
import { triggerKeyUpHandler, useComposerTrigger } from './hooks/use-composer-trigger'
import { useComposerUndo } from './hooks/use-composer-undo'
import { useComposerUrlDialog } from './hooks/use-composer-url-dialog'
import { useComposerVoice } from './hooks/use-composer-voice'
@@ -321,7 +321,7 @@ export function ChatBar({
return onCancel()
}, [activeQueueSessionKeyRef, onCancel])
const { compactPill, stacked } = useComposerMetrics({
const { compactPill, foldVoice, minimal, stacked } = useComposerMetrics({
composerDockRef,
composerRef,
composerSurfaceRef,
@@ -904,21 +904,7 @@ export function ChatBar({
}
}
const handleEditorKeyUp = () => {
// If this keyup belongs to a key the open trigger popover already consumed
// in keydown (Arrow/Enter/Tab/Escape), skip the refresh. Those keys never
// edit text, and for Escape the keydown already closed the menu — a refresh
// here would re-detect the still-present `/` and instantly reopen it. We
// read a ref set during keydown rather than `trigger`, because by keyup
// time React has re-rendered and `trigger` may already be null.
if (triggerKeyConsumedRef.current) {
triggerKeyConsumedRef.current = false
return
}
window.setTimeout(refreshTrigger, 0)
}
const handleEditorKeyUp = triggerKeyUpHandler(triggerKeyConsumedRef, refreshTrigger)
const {
dragActive,
@@ -998,7 +984,9 @@ export function ChatBar({
status: conversation.status
}}
disabled={disabled}
foldVoice={foldVoice}
hasComposerPayload={hasComposerPayload}
minimal={minimal}
onDictate={dictate}
onQueue={queueDraft}
onToggleAutoSpeak={handleToggleAutoSpeak}
@@ -1256,7 +1244,13 @@ export function ChatBar({
{hudMode && busy && <span aria-hidden className="arc-border arc-composer" />}
<div
className={cn(
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
// grid-cols-[minmax(0,1fr)]: the implicit `auto` column sized
// itself to its items' min-content, so a status row whose
// content out-measured a narrow pane silently widened the
// track past the surface — and every `w-full` child (the fade,
// the input/controls row) laid out against that phantom width
// and got clipped by overflow-hidden, send button first.
'group/composer-surface relative z-4 isolate grid grid-cols-[minmax(0,1fr)] grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
COMPOSER_DROP_FADE_CLASS,
dragActive && COMPOSER_DROP_ACTIVE_CLASS
)}
@@ -1278,7 +1272,7 @@ export function ChatBar({
// A tile's rail reviews ITS worktree: pin the pane's scope to
// this surface's cwd. Main keeps the classic follow-the-
// active-session scope (null).
onOpen={() => toggleReview(scope.target === 'main' ? null : (cwd ?? null))}
onOpen={() => toggleReview(scope.target === 'main' ? null : (cwd ?? null), scope.target)}
onOpenWorktree={openInWorktree}
onSwitchBranch={handleSwitchBranch}
repoPath={cwd}
@@ -1336,7 +1330,7 @@ export function ChatBar({
<ContribSlot area={COMPOSER_AREAS.leading} />
</div>
<div className="min-w-0 [grid-area:input]">{input}</div>
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<div className="flex min-w-0 items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<ContribSlot area={COMPOSER_AREAS.actions} />
{controls}
</div>
@@ -98,7 +98,8 @@ describe('ModelPill per-surface model label', () => {
$provider: atom('anthropic'),
$reasoningEffort: atom('high'),
$runtimeId: atom('tile-runtime'),
$storedId: atom('stored-tile')
$storedId: atom('stored-tile'),
$turnStartedAt: atom<number | null>(null)
}
render(
@@ -18,8 +18,11 @@ import { onComposerModelMenuRequest } from './focus'
import { useComposerScope } from './scope'
import type { ChatBarState } from './types'
// `shrink` (not `shrink-0`) with a truncating label: the pill is the one
// control in the row that can give width back continuously, so it absorbs the
// squeeze between collapse stages instead of pushing Send past the edge.
const PILL = cn(
'h-(--composer-control-size) max-w-40 shrink-0 gap-1 rounded-md px-2 text-xs font-normal',
'h-(--composer-control-size) min-w-0 max-w-40 shrink gap-1 rounded-md px-2 text-xs font-normal',
'text-(--ui-text-tertiary) hover:bg-(--chrome-action-hover) hover:text-foreground'
)

Some files were not shown because too many files have changed in this diff Show More