Merge remote-tracking branch 'origin/main' into fix-73914-atomic-cancel

# Conflicts:
#	tests/hermes_cli/test_web_oauth_dispatch.py
This commit is contained in:
joaomarcos
2026-07-31 03:35:08 -03:00
2964 changed files with 99831 additions and 405554 deletions
+28
View File
@@ -53,7 +53,19 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures
# (connection reset, auth token timeout, rate limiting). The action
# generates a unique builder name per invocation so the retry doesn't
# collide with the failed first attempt. A genuine persistent failure
# still fails the job — only the first attempt has continue-on-error.
# Refs: docker/setup-buildx-action#510
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
# Build once, load into the local daemon for testing. Cached
@@ -146,7 +158,15 @@ jobs:
- name: Checkout trusted source
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
@@ -208,7 +228,15 @@ jobs:
pattern: digest-*
merge-multiple: true
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
+16 -2
View File
@@ -97,9 +97,18 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# The trailing extras beyond all/dev are the lazy-install features
# (tools/lazy_deps.py) that tests exercise for real: provider.anthropic,
# stt/tts.mistral, image.fal, terminal.modal, terminal.daytona,
# memory.hindsight, search.parallel. The hermetic test env forbids
# mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so the SDKs those tests need must be in the
# venv up front — resolved from uv.lock like everything else, which
# also honors the exact supply-chain pins these extras carry.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
@@ -216,9 +225,14 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# Same extras as the test job's sync above: the hermetic test env
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
# in the venv up front.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
+4
View File
@@ -190,3 +190,7 @@ infographics/
infograficos/
infografico/
native/fts5_cjk/*.so
# Runtime marker written by hermes update when a lazy dependency refresh is
# interrupted; consumed by launch-time recovery. Never commit it (was tracked
# by accident via 3a69e34702, removed in the #72002 salvage).
.lazy-refresh-incomplete
+146
View File
@@ -0,0 +1,146 @@
# Message reactions (desktop tapbacks)
Two-way emoji reactions on individual messages in the desktop transcript: the
user reacts to any message, the agent reacts to a user message, and both sides
read the other's reactions as conversational signal.
## What already exists
Hermes already models reactions on the **platform** side — the desktop is the
only surface without them.
| Surface | Reaction support | Where |
|---|---|---|
| Agent → platform message | `send_message(action="react"/"unreact")` | `tools/send_message_tool.py:266` `_handle_react()` |
| Photon / iMessage | tapbacks in + out, routed only for messages we sent | `plugins/platforms/photon/adapter.py:1240-1283` |
| Telegram | `setMessageReaction`, config-gated | `plugins/platforms/telegram/adapter.py:9669+` |
| Slack / Matrix / Feishu / Discord | inbound reaction events → hooks | `gateway/run.py:4688` `_handle_reaction_event()` → `HookRegistry.emit("reaction:added")` |
| Adapter contract | `add_reaction()` / `remove_reaction()` coroutines, `set_reaction_handler()` | `gateway/platforms/base.py:3330` |
| Core "affection" detector | regex on user text → `vibe`, drives CLI pet / TUI heart / desktop hearts | `agent/reactions.py`, `agent/turn_context.py:592-604` |
Two things follow from that table:
1. **The agent-facing verb already exists.** `send_message(action="react")` is
the established shape. A desktop reaction should extend that tool, not add a
new core tool — every new tool ships on every API call (AGENTS.md footprint
ladder).
2. **The inbound convention already exists.** Photon turns a tapback into a
normal message event with `reply_to_message_id` + `reply_to_is_own_message`,
and the gateway prefixes `[Replying to your previous message: "…"]`
(`gateway/run.py:13125-13132`). Desktop reactions should read the same way to
the model.
Nothing exists on the desktop side: `grep -ri reaction` across `apps/desktop`
finds only the pet-overlay hearts.
## Prior art
**iOS Tapback** ([Apple](https://support.apple.com/guide/iphone/react-with-tapbacks-iph018d3c336/ios)):
double-tap or touch-and-hold a message → floating pill above the bubble with
heart / thumbs-up / thumbs-down / haha / ‼️ / ❓, swipe left for suggested emoji
and stickers, or tap the emoji button for the full keyboard. **One tapback per
message per person** — tapping the same one again removes it, tapping a
different one replaces it. Multiple people's tapbacks stack on the badge.
**Platform data models** converge on the same shape:
| Platform | Model | Add / remove |
|---|---|---|
| Slack | `{name, count, users[]}` | [`reactions.add`](https://docs.slack.dev/reference/methods/reactions.add) / `reactions.remove`, emits `reaction_added` |
| Discord | `{emoji, count, me}` on the message object | `PUT`/`DELETE .../reactions/{emoji}/@me` |
| Telegram | `reaction: [{type:"emoji", emoji:"👍"}]` — replaces the whole set | `setMessageReaction`, `is_big` for the big animation |
Telegram's "set the whole array" is the closest match to iOS semantics and the
simplest thing to persist.
**assistant-ui has no reaction primitive.** `@assistant-ui/react` 0.14.24 (MIT,
vendored at `apps/desktop/node_modules`): zero hits for "reaction" in `core/src`,
`react/src`, `dist/`, or the 2.2 MB `llms-full.txt` docs dump. What exists is a
hard-coded binary `FeedbackAdapter` (`"positive" | "negative"`,
`core/src/adapters/feedback.ts`) that throws when unconfigured and only writes
back onto assistant messages. Not usable for emoji, not usable on user messages.
**But `metadata.custom` is the supported extension channel** and this repo
already uses it: `ThreadUserMessage`/`ThreadAssistantMessage`/`ThreadSystemMessage`
all carry `metadata.custom: Record<string, unknown>` (`core/src/types/message.ts:319-366`),
and `chat-runtime.ts:397` already ships `custom: { attachmentRefs }` through it.
**Emoji picker survey** (npm week of 2026-07-22, sizes measured from the
published ESM entry):
| Library | License | Weekly DL | gzip | Headless | Latest |
|---|---|---|---|---|---|
| **frimousse** | MIT | 573k | **8.5 kB** | ✅ fully unstyled, composable parts | 0.3.0 · 2025-07-15 |
| emoji-picker-react | MIT | 1.31M | 87 kB | ❌ own CSS-in-JS (flairup) | 4.19.1 · 2026-04-27 |
| emoji-mart | MIT | 2.22M | ~120 kB w/ data | ❌ Preact + shadow styling | 5.6.0 · **2024-04-25**, 217 open issues |
| emoji-picker-element | Apache-2.0 | 183k | — | ❌ Web Component / Shadow DOM | 1.29.1 · 2026-03-01 |
No picker is currently a dependency (only `emoji-regex`, transitive). Already
paid for and reusable: `radix-ui` (Popover), `motion`, `@tanstack/react-virtual`,
Tailwind v4.
## Recommendation
**Hand-roll the tapback pill; add frimousse only behind the "+".** Six fixed
emoji in a pill is ~40 lines of JSX against existing tokens — pulling 87 kB of
`emoji-picker-react` to render six buttons, plus a CSS engine that fights
`DESIGN.md`, is backwards. frimousse is headless, dependency-free, 10× smaller,
and exposes `emojibaseUrl` so the data can be bundled as a Vite asset instead of
hitting jsDelivr (Electron must work offline).
### Data model
One reaction per author per message, Telegram-style whole-set replacement:
```ts
type MessageReaction = { emoji: string; author: 'user' | 'agent'; at: number }
```
Persisted in the existing `messages.display_metadata` JSON column
(`hermes_state_common.py:215`) — no new table. It already survives insert,
compaction, and every read projection, and
`set_latest_matching_message_display_kind()` (`hermes_state.py:5292`) is the
precedent for stamping metadata onto an already-persisted row.
### Model context
Reactions must reach the model **without breaking prompt caching**. The
`api_messages` build loop strips `display_metadata` from every outgoing copy
(`agent/conversation_loop.py:1443-1446`) precisely so display state never
becomes a provider field. Two candidate paths:
| Path | Cache impact | Notes |
|---|---|---|
| Rewrite the reacted-to message's content to carry the annotation | **Breaks the cached prefix** — mutates past context | Rejected. AGENTS.md: prompt caching is sacred. |
| Deliver the reaction as the *next* turn's leading annotation, mirroring photon | Prefix untouched; only the new turn carries it | Matches `[Replying to your previous message: "…"]` (`gateway/run.py:13125`), which the agent already understands |
The second is the same trick the platform adapters already use, so the model
sees a familiar shape and no existing conversation is rewritten.
### Attach points
| Concern | File | Lines |
|---|---|---|
| Assistant hover bar | `apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx` | 134–175 |
| User hover cluster | `apps/desktop/src/components/assistant-ui/thread/user-message.tsx` | 296–336 |
| Callback threading (ref caveat 79–99) | `apps/desktop/src/components/assistant-ui/thread/index.tsx` | 109–133 |
| `metadata.custom` → runtime | `apps/desktop/src/lib/chat-runtime.ts` | 384–432 |
| RPC client ↔ server pattern | `sidebar/session-actions-menu.tsx:62-89` ↔ `tui_gateway/server.py:8322` | — |
| Persistence | `hermes_state_common.py:192-216`, `hermes_state.py:5292-5324` | — |
| Prompt injection / strip | `agent/conversation_loop.py` | 1430–1529 |
### Known gaps to solve first
- **No durable message id crosses the gateway RPC path.** `_history_to_messages()`
(`tui_gateway/server.py:6545`) builds `{"role", "text"}` and drops the id. The
REST path carries `messages.id` incidentally via `SELECT *` but TS
`SessionMessage` (`types/hermes.ts:513-533`) doesn't declare it. Renderer ids
are ephemeral and change shape between rehydrated (`<ts>-<i>-<role>`), live
(`assistant-<ms>`), and optimistic (`user-<ms>-<rand>`) messages. A reaction
needs a stable key — this is the first thing to fix.
- **WeakMap identity cache** in `apps/desktop/src/app/chat/runtime-repository.ts:26-66`
keys normalized `ThreadMessage` by `ChatMessage` identity. A reaction change
must produce a **new** `ChatMessage` object or the UI renders stale.
- **Rewind rewrites rows** (`replace_messages`), so anything keyed by row id
needs cascade handling — an argument for keeping reactions in
`display_metadata` on the row itself rather than a side table.
+3 -2
View File
@@ -1284,14 +1284,15 @@ def profile_env(tmp_path, monkeypatch):
### Python
**ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces
hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8,
`-n auto` xdist workers, in-tree subprocess-isolation plugin). Direct `pytest`
per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist,
worker count auto-scaled from CPU count). Direct `pytest`
on a 16+ core developer machine with API keys set diverges from CI in ways
that have caused multiple "works locally, fails in CI" incidents (and the reverse).
```bash
scripts/run_tests.sh # full suite, CI-parity
scripts/run_tests.sh tests/gateway/ # one directory
scripts/run_tests.sh tests/agent/test_foo.py::test_x # one test
scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular)
scripts/run_tests.sh -v --tb=long # pass-through pytest flags
```
+3 -2
View File
@@ -201,7 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes
### Run tests
```bash
# Preferred — matches CI (hermetic env, 4 xdist workers); see AGENTS.md
# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation
# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md
scripts/run_tests.sh
# Alternative (activate the venv first). The wrapper is still recommended
@@ -848,7 +849,7 @@ that touches the OS, assume *any* platform can hit your code path.
Tests that use POSIX-only syscalls need a skip marker. Common ones:
- Symlinks → `@pytest.mark.skipif(sys.platform == "win32", ...)`
- `0o600` file modes → `@pytest.mark.skipif(sys.platform.startswith("win"), ...)`
- `signal.SIGALRM` → Unix-only (see `tests/conftest.py::_enforce_test_timeout`)
- `signal.SIGALRM` → Unix-only (per-test timeouts no longer use it directly; see the win32 timeout-method shim in `tests/conftest.py::pytest_configure`)
- `os.setsid` / `os.fork` → Unix-only
- Live Winsock / Windows-specific regression tests →
`@pytest.mark.skipif(sys.platform != "win32", reason="Windows-specific regression")`
+22 -2
View File
@@ -200,6 +200,22 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
done && \
npm cache clean --force
# ---------- Photon iMessage sidecar deps (baked, NS-606) ----------
# The photon plugin's Node sidecar needs its own node_modules
# (spectrum-ts). The install tree is immutable at runtime, so a lazy
# `npm ci` on first connect would hit EROFS — bake the deps here instead
# (deterministic installs, NS-559). The patch script is copied alongside
# the manifests because package.json's postinstall runs it, which also
# means the spectrum-ts patch is applied at build time. Layer-cached:
# only re-runs when the sidecar manifests/patch change.
COPY plugins/platforms/photon/sidecar/package.json \
plugins/platforms/photon/sidecar/package-lock.json \
plugins/platforms/photon/sidecar/patch-spectrum-mixed-attachments.mjs \
plugins/platforms/photon/sidecar/
RUN cd plugins/platforms/photon/sidecar && \
npm ci --no-audit --fetch-retries=5 && \
npm cache clean --force
# ---------- Layer-cached Python dependency install ----------
# Copy only pyproject.toml + uv.lock so the Python dep resolve + wheel
# download + native-extension compile layer is cached unless those inputs
@@ -211,7 +227,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# frontend stats the readme path during dep resolution, so we `touch` an
# empty placeholder — the real README is restored by `COPY . .` below.
#
# `uv sync --frozen --no-install-project --extra all --extra messaging`
# `uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp`
# installs the deps reachable through the composite `[all]` extra
# (handpicked set intended for the production image — excludes `[dev]`),
# plus gateway messaging adapters that should work in the published image
@@ -224,6 +240,10 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# so Docker users can use these providers without requiring runtime
# lazy-install access to PyPI (often blocked in containerized envs).
#
# The [otlp] extra contains the SDK/exporter imported by Hermes when Gateway
# Health export is enabled. Collector and observability-backend dependencies
# remain external and are not part of the Hermes production image.
#
# The hindsight memory provider's client (hindsight-client) is baked in
# for the same reason: it lazy-installs into /opt/hermes/.venv at first
# use, which lives inside the (immutable) image layer rather than the
@@ -241,7 +261,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# The editable link is created after the source copy below.
COPY pyproject.toml uv.lock ./
RUN touch ./README.md
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
# ---------- Frontend build (cached independently from Python source) ----------
# Copy only the frontend source trees first so that Python-only changes don't
+1 -1
View File
@@ -26,7 +26,7 @@ Use any model you want — [Nous Portal](https://portal.nousresearch.com), OpenR
<tr><td><b>A closed learning loop</b></td><td>Agent-curated memory with periodic nudges. Autonomous skill creation after complex tasks. Skills self-improve during use. FTS5 session search with LLM summarization for cross-session recall. <a href="https://github.com/plastic-labs/honcho">Honcho</a> dialectic user modeling. Compatible with the <a href="https://agentskills.io">agentskills.io</a> open standard.</td></tr>
<tr><td><b>Scheduled automations</b></td><td>Built-in cron scheduler with delivery to any platform. Daily reports, nightly backups, weekly audits — all in natural language, running unattended.</td></tr>
<tr><td><b>Delegates and parallelizes</b></td><td>Spawn isolated subagents for parallel workstreams. Write Python scripts that call tools via RPC, collapsing multi-step pipelines into zero-context-cost turns.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Six terminal backends — local, Docker, SSH, Singularity, Modal, and Daytona. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Seven terminal backends — local, Docker, SSH, Singularity, Modal, Daytona, and Vercel Sandbox. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Research-ready</b></td><td>Batch trajectory generation, trajectory compression for training the next generation of tool-calling models.</td></tr>
</table>
+7 -3
View File
@@ -173,9 +173,13 @@ modelo de autorización, pero las reglas a continuación se aplican uniformement
**Superficies en Hermes Agent:**
- **Adaptadores de plataforma del gateway.** Integraciones de mensajería en
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
y adaptadores análogos incluidos como plugins.
- **Adaptadores de plataforma del gateway.** La mayoría de las integraciones
de mensajería se distribuyen como plugins empaquetados en
`plugins/platforms/<name>/` (Telegram, Discord, Slack, email, SMS, etc.).
Los tipos base compartidos y un conjunto menor de adaptadores
legacy/directos viven en `gateway/platforms/` (`base.py`, Signal, servidor
API, webhooks, …), con descubrimiento y carga diferida vía
`gateway/platform_registry.py`.
- **Superficies HTTP expuestas en red.** El adaptador del servidor API, el
plugin del dashboard, los endpoints HTTP del plugin kanban, y cualquier
otro plugin que vincule un socket de escucha.
+6 -3
View File
@@ -177,9 +177,12 @@ authorization model, but the rules below apply uniformly.
**Surfaces in Hermes Agent:**
- **Gateway platform adapters.** Messaging integrations in
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
and analogous adapters shipped as plugins.
- **Gateway platform adapters.** Most messaging integrations ship as
bundled plugins under `plugins/platforms/<name>/` (Telegram, Discord,
Slack, email, SMS, etc.). Shared base types and a smaller set of
legacy/direct adapters live under `gateway/platforms/`
(`base.py`, Signal, API server, webhooks, …), with discovery and
deferred loading via `gateway/platform_registry.py`.
- **Network-exposed HTTP surfaces.** The API server adapter, the
dashboard plugin, the kanban plugin's HTTP endpoints, and any
other plugin that binds a listening socket.
+9 -12
View File
@@ -118,7 +118,7 @@ def _provider_default_routes(provider: str) -> set[str]:
from hermes_cli.providers import HERMES_OVERLAYS, get_provider
overlay = HERMES_OVERLAYS.get(provider)
provider_def = get_provider(provider)
provider_def = get_provider(provider, allow_network=False)
for value in (
getattr(overlay, "base_url_override", ""),
getattr(provider_def, "base_url", ""),
@@ -455,8 +455,7 @@ def init_agent(
command: str = None,
args: list[str] | None = None,
model: str = "",
max_iterations: int = 500, # Default tool-calling iterations (shared with subagents)
tool_delay: float = 1.0,
max_iterations: int = 90, # Default tool-calling iterations (shared with subagents)
enabled_toolsets: List[str] = None,
disabled_toolsets: List[str] = None,
save_trajectories: bool = False,
@@ -529,8 +528,7 @@ def init_agent(
requested_provider (str): Original provider identity before runtime canonicalization
api_mode (str): API mode override: "chat_completions" or "codex_responses"
model (str): Model name to use (default: "anthropic/claude-opus-4.6")
max_iterations (int): Maximum number of tool calling iterations (default: 500)
tool_delay (float): Delay between tool calls in seconds (default: 1.0)
max_iterations (int): Maximum number of tool calling iterations (default: 90)
enabled_toolsets (List[str]): Only enable tools from these toolsets (optional)
disabled_toolsets (List[str]): Disable tools from these toolsets (optional)
save_trajectories (bool): Whether to save conversation trajectories to JSONL files (default: False)
@@ -576,7 +574,6 @@ def init_agent(
# Shared iteration budget — parent creates, children inherit.
# Consumed by every LLM turn across parent + all subagents.
agent.iteration_budget = iteration_budget or IterationBudget(max_iterations)
agent.tool_delay = tool_delay
agent.save_trajectories = save_trajectories
agent.verbose_logging = verbose_logging
agent.quiet_mode = quiet_mode
@@ -843,7 +840,7 @@ def init_agent(
# sessions with >5-minute pauses between turns (#14971).
agent._cache_ttl = "5m"
try:
from hermes_cli.config import load_config as _load_pc_cfg
from hermes_cli.config import load_config_readonly as _load_pc_cfg
_pc_cfg = _load_pc_cfg().get("prompt_caching", {}) or {}
_ttl = _pc_cfg.get("cache_ttl", "5m")
@@ -1093,7 +1090,7 @@ def init_agent(
# Guardrail config — read from config.yaml at init time.
agent._bedrock_guardrail_config = None
try:
from hermes_cli.config import load_config as _load_br_cfg
from hermes_cli.config import load_config_readonly as _load_br_cfg
_gr = _load_br_cfg().get("bedrock", {}).get("guardrail", {})
if _gr.get("guardrail_identifier") and _gr.get("guardrail_version"):
agent._bedrock_guardrail_config = {
@@ -1164,8 +1161,8 @@ def init_agent(
client_kwargs["default_headers"] = hermes_xai_default_headers()
elif "default_headers" not in client_kwargs:
# Fall back to profile.default_headers for providers that
# declare custom headers (e.g. Kimi User-Agent on non-kimi.com
# endpoints).
# declare custom headers (e.g. Vercel AI Gateway attribution,
# Kimi User-Agent on non-kimi.com endpoints).
try:
from providers import get_provider_profile as _gpf
_ph = _gpf(agent.provider)
@@ -1480,7 +1477,7 @@ def init_agent(
# reads the JSON files directly. See run_agent._save_session_log.
agent._session_json_enabled = False
try:
from hermes_cli.config import load_config as _load_sess_cfg
from hermes_cli.config import load_config_readonly as _load_sess_cfg
_sess_cfg = (_load_sess_cfg().get("sessions") or {})
agent._session_json_enabled = bool(_sess_cfg.get("write_json_snapshots", False))
except Exception:
@@ -1551,7 +1548,7 @@ def init_agent(
# Load config once for memory, skills, and compression sections
try:
from hermes_cli.config import load_config as _load_agent_config
from hermes_cli.config import load_config_readonly as _load_agent_config
_agent_cfg = _load_agent_config()
except Exception:
_agent_cfg = {}
+52 -102
View File
@@ -249,12 +249,42 @@ def sanitize_tool_call_arguments(
*,
logger=None,
session_id: str = None,
cursor: Optional[dict] = None,
) -> int:
"""Repair corrupted assistant tool-call argument JSON in-place."""
"""Repair corrupted assistant tool-call argument JSON in-place.
``cursor`` (optional) is a caller-owned dict used to skip re-validating
messages already validated on a previous call. It stores, under
``"prefix"``, the exact message *objects* (strong references) validated
last time, in order. On the next call, the longest contiguous prefix of
``messages`` whose objects are ``is``-identical to the stored prefix is
skipped; scanning starts at the first divergence (conservative: any
reordering, truncation, compression rewrite, or mid-list insertion breaks
identity at that index and everything from there is re-scanned).
Safety argument for skipping: a message in the matched prefix was fully
scanned before — every tool_call argument was either already valid JSON
or was rewritten to ``"{}"`` (valid). The only code paths that mutate
``function["arguments"]`` on live history dicts between calls are the
surrogate / non-ASCII sanitizers, which substitute characters *inside*
JSON string values and cannot invalidate JSON syntax. Compression,
repair, undo, and steer paths replace or reorder message dicts, which
breaks the identity match and forces a re-scan. Holding strong
references (the objects themselves, not ``id()``s) makes address reuse
aliasing (#50372-style) impossible.
"""
log = logger or logging.getLogger(__name__)
if not isinstance(messages, list):
return 0
start_index = 0
if cursor is not None:
prev_prefix = cursor.get("prefix")
if isinstance(prev_prefix, list):
limit = min(len(prev_prefix), len(messages))
while start_index < limit and messages[start_index] is prev_prefix[start_index]:
start_index += 1
repaired = 0
marker = _ra().AIAgent._TOOL_CALL_ARGUMENTS_CORRUPTION_MARKER
@@ -275,7 +305,7 @@ def sanitize_tool_call_arguments(
existing_text = str(existing)
tool_msg["content"] = f"{marker}\n{existing_text}"
message_index = 0
message_index = start_index
while message_index < len(messages):
msg = messages[message_index]
if not isinstance(msg, dict) or msg.get("role") != "assistant":
@@ -356,6 +386,12 @@ def sanitize_tool_call_arguments(
message_index += 1
if cursor is not None:
# Strong references to the exact objects validated this call, in
# order. Any future divergence (compression, undo, repair, steer)
# breaks identity at the divergent index and re-scans from there.
cursor["prefix"] = messages[:]
return repaired
@@ -3295,89 +3331,17 @@ def intent_ack_continuation_enabled(agent) -> bool:
def copy_reasoning_content_for_api(agent, source_msg: dict, api_msg: dict) -> None:
"""Copy provider-facing reasoning fields onto an API replay message."""
if source_msg.get("role") != "assistant":
return
"""Copy provider-facing reasoning fields onto an API replay message.
needs_thinking_pad = agent._needs_thinking_reasoning_pad()
Forwarder — the strip-vs-repad POLICY is owned by
``agent.message_sanitization.apply_reasoning_content_policy`` (audit F4);
this only supplies the agent's cached provider-direction flag.
"""
from agent.message_sanitization import apply_reasoning_content_policy
# 1. Explicit reasoning_content already set.
#
# When the active provider enforces the thinking-mode echo-back
# (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their
# own space-placeholder written at creation time and any valid reasoning
# from the same provider. Sessions persisted BEFORE #17341 have
# empty-string placeholders pinned at creation time; DeepSeek V4 Pro
# rejects those with HTTP 400, so upgrade "" → " " on replay.
#
# When the active provider does NOT enforce echo-back, strip the field
# entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq,
# SambaNova, …) reject ANY reasoning_content key in input messages with
# HTTP 400/422 ("Extra inputs are not permitted"), even an empty string
# or a single-space pad. This is the cross-provider fallback case: a
# reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a
# fallback to a strict provider replays that pad and 422s. Stripping
# here covers the rebuild path; reapply_reasoning_echo_for_provider()
# covers the already-built api_messages path. Refs #45655.
existing = source_msg.get("reasoning_content")
if isinstance(existing, str):
if not needs_thinking_pad:
api_msg.pop("reasoning_content", None)
elif existing == "":
api_msg["reasoning_content"] = " "
else:
api_msg["reasoning_content"] = existing
return
# 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi,
# if the source turn has tool_calls AND a 'reasoning' field but no
# 'reasoning_content' key, the 'reasoning' text was written by a
# prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message
# pins reasoning_content at creation time for tool-call turns, so the
# shape (reasoning set, reasoning_content absent, tool_calls present)
# is unreachable from same-provider DeepSeek history after this fix.
# Inject a single space to satisfy the API without leaking another
# provider's chain of thought to DeepSeek/Kimi. Space (not "")
# because DeepSeek V4 Pro rejects empty-string reasoning_content
# in thinking mode (refs #17341).
normalized_reasoning = source_msg.get("reasoning")
if (
needs_thinking_pad
and source_msg.get("tool_calls")
and isinstance(normalized_reasoning, str)
and normalized_reasoning
):
api_msg["reasoning_content"] = " "
return
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content'
# for providers that use the internal 'reasoning' key.
# This must happen before the unconditional empty-string fallback so
# genuine reasoning content is not overwritten (#15812 regression in
# PR #15478). Only promote for providers that enforce echo-back —
# strict providers reject the field (refs #45655).
if isinstance(normalized_reasoning, str) and normalized_reasoning:
if needs_thinking_pad:
api_msg["reasoning_content"] = normalized_reasoning
else:
api_msg.pop("reasoning_content", None)
return
# 4. DeepSeek / Kimi thinking mode: all assistant messages need
# reasoning_content. Inject a single space to satisfy the provider's
# requirement when no explicit reasoning content is present. Covers
# both tool-call turns (already-poisoned history with no reasoning
# at all) and plain text turns. Space (not "") because DeepSeek V4
# Pro tightened validation and rejects empty string with HTTP 400
# ("The reasoning content in the thinking mode must be passed back
# to the API"). Refs #17341.
if needs_thinking_pad:
api_msg["reasoning_content"] = " "
return
# 5. reasoning_content was present but not a string (e.g. None after
# context compaction). Don't pass null to the API.
api_msg.pop("reasoning_content", None)
apply_reasoning_content_policy(
source_msg, api_msg, agent._needs_thinking_reasoning_pad()
)
def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int:
@@ -3409,25 +3373,11 @@ def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int:
Returns the number of assistant turns whose reasoning_content was added or
removed.
"""
needs_pad = agent._needs_thinking_reasoning_pad()
changed = 0
for api_msg in api_messages:
if api_msg.get("role") != "assistant":
continue
if needs_pad:
if api_msg.get("reasoning_content"):
continue
copy_reasoning_content_for_api(agent, api_msg, api_msg)
if api_msg.get("reasoning_content"):
changed += 1
else:
# Strict provider — strip any stale reasoning_content pad left
# over from a reasoning primary so the fallback request doesn't
# 400/422 on it.
if "reasoning_content" in api_msg:
api_msg.pop("reasoning_content", None)
changed += 1
return changed
from agent.message_sanitization import reapply_reasoning_echo
return reapply_reasoning_echo(
api_messages, agent._needs_thinking_reasoning_pad()
)
def _iter_httpx_pool_objects(http_client: Any):
+29 -18
View File
@@ -543,6 +543,7 @@ _API_KEY_PROVIDER_AUX_MODELS_FALLBACK: Dict[str, str] = {
"kimi-coding-cn": "kimi-k2-turbo-preview",
"gmi": "google/gemini-3.1-flash-lite-preview",
"anthropic": "claude-haiku-4-5-20251001",
"ai-gateway": "google/gemini-3-flash",
"opencode-zen": "gemini-3-flash",
"opencode-go": "glm-5",
"kilocode": "google/gemini-3.6-flash",
@@ -676,15 +677,15 @@ def build_or_headers(or_config: dict | None = None) -> dict:
Overrides ``openrouter.response_cache_ttl`` in config.yaml.
*or_config* is the ``openrouter`` section from config.yaml. When *None*,
falls back to reading config from disk via ``load_config()``.
falls back to reading config from disk via ``load_config_readonly()``.
"""
headers = dict(_OR_HEADERS_BASE)
# Resolve config from disk if not provided.
if or_config is None:
try:
from hermes_cli.config import load_config
or_config = load_config().get("openrouter", {})
from hermes_cli.config import load_config_readonly
or_config = load_config_readonly().get("openrouter", {})
except Exception:
or_config = {}
@@ -729,6 +730,15 @@ def build_nvidia_nim_headers(base_url: str | None) -> dict:
return {}
# Vercel AI Gateway app attribution headers. HTTP-Referer maps to
# referrerUrl and X-Title maps to appName in the gateway's analytics.
from hermes_cli import __version__ as _HERMES_VERSION
_AI_GATEWAY_HEADERS = {
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
}
# Nous Portal extra_body for product attribution.
# Callers should pass this as extra_body in chat.completions.create()
@@ -2317,8 +2327,8 @@ def _read_main_model() -> str:
if isinstance(override, str) and override.strip():
return override.strip()
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
model_cfg = cfg.get("model", {})
if isinstance(model_cfg, str) and model_cfg.strip():
return model_cfg.strip()
@@ -2344,8 +2354,8 @@ def _read_main_provider() -> str:
if isinstance(override, str) and override.strip():
return override.strip().lower()
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
model_cfg = cfg.get("model", {})
if isinstance(model_cfg, dict):
provider = model_cfg.get("provider", "")
@@ -3040,12 +3050,12 @@ def _try_azure_foundry(
try:
from hermes_cli.runtime_provider import _resolve_azure_foundry_runtime
from hermes_cli.auth import AuthError
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
except ImportError:
return None, None
try:
cfg = load_config()
cfg = load_config_readonly()
model_cfg = cfg.get("model") if isinstance(cfg, dict) else {}
if not isinstance(model_cfg, dict):
model_cfg = {}
@@ -3159,8 +3169,8 @@ def _try_anthropic(explicit_api_key: str = None) -> Tuple[Optional[Any], Optiona
# see issue #52608.
base_url = _pool_runtime_base_url(entry, _ANTHROPIC_DEFAULT_BASE_URL) if pool_present else _ANTHROPIC_DEFAULT_BASE_URL
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
model_cfg = cfg.get("model")
if isinstance(model_cfg, dict):
cfg_provider = str(model_cfg.get("provider") or "").strip().lower()
@@ -4764,10 +4774,10 @@ def _try_main_fallback_chain(
participate in the same order as the main agent.
"""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
from hermes_cli.fallback_config import get_fallback_chain
chain = get_fallback_chain(load_config())
chain = get_fallback_chain(load_config_readonly())
except Exception as exc:
logger.debug("Auxiliary %s: could not load main fallback chain: %s", task or "call", exc)
return None, None, ""
@@ -5725,7 +5735,8 @@ def resolve_provider_client(
else:
# Fall back to profile.default_headers for providers that declare
# client-level attribution headers on their profile (e.g. GMI
# User-Agent for traffic identification).
# User-Agent for traffic identification, Vercel AI Gateway
# Referer/Title for analytics).
try:
from providers import get_provider_profile as _gpf_main
_ph_main = _gpf_main(provider)
@@ -5986,11 +5997,11 @@ def _main_model_supports_vision(provider: str, model: Optional[str]) -> bool:
"""
try:
from agent.image_routing import _lookup_supports_vision
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
except ImportError:
return True
try:
supports = _lookup_supports_vision(provider, model, load_config())
supports = _lookup_supports_vision(provider, model, load_config_readonly())
except Exception: # pragma: no cover - defensive
return True
if supports is None:
@@ -6959,8 +6970,8 @@ def _get_auxiliary_task_config(task: str) -> Dict[str, Any]:
if not task:
return {}
try:
from hermes_cli.config import load_config
config = load_config()
from hermes_cli.config import load_config_readonly
config = load_config_readonly()
except ImportError:
return {}
aux = config.get("auxiliary", {}) if isinstance(config, dict) else {}
+2 -2
View File
@@ -70,8 +70,8 @@ def _resolve_review_runtime(agent: Any) -> Dict[str, Any]:
"routed": False,
}
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception:
return parent
aux = cfg.get("auxiliary", {}) if isinstance(cfg.get("auxiliary"), dict) else {}
+1 -1
View File
@@ -34,7 +34,7 @@ from __future__ import annotations
import logging
import math
import os
from dataclasses import dataclass, field
from dataclasses import dataclass
from typing import Any, Optional
logger = logging.getLogger(__name__)
+1 -1
View File
@@ -17,7 +17,7 @@ from __future__ import annotations
import logging
import os
import uuid
from dataclasses import dataclass, field
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import Any, Optional
+56 -8
View File
@@ -227,6 +227,53 @@ def _env_float(name: str, default: float) -> float:
return default
def _estimate_chunk_bytes(chunk: Any) -> int:
"""Cheap per-chunk size estimate for the stream diagnostic counters.
The previous implementation used ``len(repr(chunk))`` — a full recursive
repr of a pydantic model on EVERY streaming chunk (5.5-8.8 µs each,
~20-30 ms of pure CPU on a 3,000-chunk response, in the hottest loop in
the agent). The counter only feeds a retry-diagnostic log line, so an
estimate based on the delta payload lengths is plenty (2.1-2.4 µs, ~3x
cheaper, and independent of model/pydantic field count). Chat Completions
chunks are sized from their delta content/reasoning/tool-argument strings
plus a small framing constant; anything shape-unknown (Anthropic events,
stub providers) falls back to a flat constant so `bytes` stays monotonic
and roughly proportional to traffic.
"""
size = 40 # SSE/JSON framing floor per chunk
try:
choices = getattr(chunk, "choices", None)
if choices:
delta = getattr(choices[0], "delta", None)
if delta is not None:
for attr in ("content", "reasoning_content", "reasoning"):
v = getattr(delta, attr, None)
if isinstance(v, str):
size += len(v)
tool_calls = getattr(delta, "tool_calls", None)
if tool_calls:
for tc in tool_calls:
fn = getattr(tc, "function", None)
if fn is not None:
args = getattr(fn, "arguments", None)
if isinstance(args, str):
size += len(args)
name = getattr(fn, "name", None)
if isinstance(name, str):
size += len(name)
else:
# Non-chat-completions shapes (Anthropic events etc.): try the
# common text fields, else keep the framing floor.
for attr in ("text", "partial_json"):
v = getattr(getattr(chunk, "delta", None), attr, None)
if isinstance(v, str):
size += len(v)
except Exception:
pass
return size
def _codex_wait_notice_recovery(
*,
stale_timeout: float,
@@ -3108,12 +3155,13 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
_diag["chunks"] = int(_diag.get("chunks", 0)) + 1
if _diag.get("first_chunk_at") is None:
_diag["first_chunk_at"] = last_chunk_time["t"]
# Approximate byte size from the chunk's repr — exact wire
# bytes aren't exposed by the SDK, but len(repr(chunk)) is
# a stable proxy for "how much content arrived" that
# survives stub provider differences.
# Approximate byte size from the chunk's delta payload —
# exact wire bytes aren't exposed by the SDK. A full
# repr() per chunk was 5.5-8.8 µs of pure CPU on the
# hottest loop in the agent; the delta-length estimate
# is ~3x cheaper and stays proportional to traffic.
try:
_diag["bytes"] = int(_diag.get("bytes", 0)) + len(repr(chunk))
_diag["bytes"] = int(_diag.get("bytes", 0)) + _estimate_chunk_bytes(chunk)
except Exception:
pass
except Exception:
@@ -3554,7 +3602,7 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
_diag["chunks"] = int(_diag.get("chunks", 0)) + 1
if _diag.get("first_chunk_at") is None:
_diag["first_chunk_at"] = last_chunk_time["t"]
_diag["bytes"] = int(_diag.get("bytes", 0)) + len(repr(event))
_diag["bytes"] = int(_diag.get("bytes", 0)) + _estimate_chunk_bytes(event)
except Exception:
pass
if agent._interrupt_requested:
@@ -3980,9 +4028,9 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
# env var ``HERMES_LOCAL_STREAM_STALE_TIMEOUT`` overrides for escape-hatch.
_local_default = 900.0
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
_cfg = load_config()
_cfg = load_config_readonly() # read-only consumer — no deepcopy
_agent_cfg = _cfg.get("agent") if isinstance(_cfg, dict) else None
if isinstance(_agent_cfg, dict):
_v = _agent_cfg.get("local_stream_stale_timeout")
+5 -4
View File
@@ -18,6 +18,7 @@ import uuid
from types import SimpleNamespace
from typing import Any, Dict, List, Optional
from agent.message_sanitization import deterministic_call_id
from agent.prompt_builder import DEFAULT_AGENT_IDENTITY
logger = logging.getLogger(__name__)
@@ -182,13 +183,13 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str:
def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Thin wrapper over the single policy owner
``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept
as a module-level name because run_agent and tests import it from here.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return f"call_{digest}"
return deterministic_call_id(fn_name, arguments, index)
def _clamp_responses_call_id(call_id: str) -> str:
-1
View File
@@ -18,7 +18,6 @@ from __future__ import annotations
import json
import logging
import os
import time
from types import SimpleNamespace
from typing import Any, Callable, Dict, List
+2 -2
View File
@@ -337,9 +337,9 @@ def _coding_mode(config: Optional[dict[str, Any]]) -> str:
"""Return the normalized ``agent.coding_context`` mode (auto/focus/on/off)."""
if config is None:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
config = load_config()
config = load_config_readonly()
except Exception:
config = {}
raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto")
+83 -6
View File
@@ -170,6 +170,35 @@ def _fresh_compaction_message_copy(msg: Dict[str, Any]) -> Dict[str, Any]:
return fresh
def _template_visible_role(message: Any) -> Optional[str]:
"""Role as counted by strict chat-template alternation checks.
Mistral-family templates (Devstral, Mistral Small 3.x, Magistral)
enforce user/assistant alternation at render time but EXEMPT the tool
flow from the check: ``tool`` results and assistant messages carrying
``tool_calls`` are skipped. A summary role chosen against the *literal*
neighbouring roles can therefore still violate alternation as the
template sees it. The canonical failure: the protected head ends
``[user, assistant(tool_calls), tool]``, so the literal last role is
``tool`` and the summary is pinned to ``role="user"`` -- but the last
role the template counts is ``user``, the template sees user -> user,
and llama.cpp / Mistral-hosted backends reject the ENTIRE request with
a Jinja alternation error (HTTP 500). Because the summary persists in
the stored conversation, every retry replays the same poisoned history
and the session is unrecoverable.
Returns ``None`` for messages the alternation check skips.
"""
if not isinstance(message, dict):
return None
role = message.get("role")
if role == "tool":
return None
if role == "assistant" and message.get("tool_calls"):
return None
return role
def _strip_persistence_markers(messages: List[Dict[str, Any]]) -> None:
"""Enforce the compaction invariant: no assembled message carries a
session-store persistence marker.
@@ -5420,8 +5449,43 @@ This compaction should PRIORITISE preserving all information related to the focu
# last_head_role reads the assembled (post-strip) head; first_tail_role
# reads the assembled (post-strip) tail_messages — a stripped stale
# handoff must not influence alternation-safe role selection.
last_head_role = compressed[-1].get("role", "user") if compressed else "user"
first_tail_role = tail_messages[0].get("role", "user") if tail_messages else None
# Both are TEMPLATE-VISIBLE roles (``_template_visible_role``), not the
# literal list neighbours: strict Mistral-style templates skip tool
# results and assistant tool-call messages when enforcing
# user/assistant alternation, so the summary must alternate against
# the nearest message the template actually counts. Selecting against
# the literal neighbour (previously ``compressed[-1]``) emitted the
# summary as role="user" behind a ``[user, assistant(tool_calls),
# tool]`` head — which every Mistral-strict backend rejects with a
# Jinja alternation 500, permanently poisoning the session.
last_head_role: Optional[str] = "user"
if compressed:
last_head_role = next(
(
role
for role in (
_template_visible_role(m) for m in reversed(compressed)
)
if role is not None
),
# Head holds only template-exempt messages: the summary will
# be the first message the template counts, and the sequence
# must open with "user" (handled below alongside the forced
# cases).
None,
)
first_tail_role = None
if tail_messages:
first_tail_role = next(
(
role
for role in (
_template_visible_role(m) for m in tail_messages
)
if role is not None
),
None,
)
# When the only protected head message is the system prompt, the
# summary becomes the first *visible* message in the API request
# (most adapters — Anthropic, Bedrock — send the system prompt as
@@ -5455,9 +5519,15 @@ This compaction should PRIORITISE preserving all information related to the focu
)
if not _user_survives:
_force_user_leading = True
# Pick a role that avoids consecutive same-role with both neighbors.
# Priority: avoid colliding with head (already committed), then tail.
if last_head_role in {"assistant", "tool"} or _force_user_leading:
# Pick a role that alternates with both template-visible neighbors.
# Priority: alternate against the head (already committed), then tail.
# ``None`` (all-exempt head) means the summary opens the visible
# sequence, which strict templates require to start with "user".
if (
last_head_role is None
or last_head_role in {"assistant", "tool"}
or _force_user_leading
):
summary_role = "user"
else:
summary_role = "assistant"
@@ -5465,7 +5535,14 @@ This compaction should PRIORITISE preserving all information related to the focu
# collide with the head, flip it.
if first_tail_role is not None and summary_role == first_tail_role:
flipped = "assistant" if summary_role == "user" else "user"
if flipped != last_head_role and not _force_user_leading:
# ``last_head_role is None`` (all-exempt head) pins the summary to
# "user" above; flipping to "assistant" would make the visible
# sequence open with "assistant", which strict templates reject.
if (
flipped != last_head_role
and last_head_role is not None
and not _force_user_leading
):
summary_role = flipped
else:
# Both roles would create consecutive same-role messages
+19 -3
View File
@@ -22,9 +22,7 @@ import os
import random
import re
import ssl
import threading
import time
import uuid
from typing import Any, Dict, List, Optional
from agent.codex_responses_adapter import _summarize_user_message_for_log
@@ -40,7 +38,6 @@ from agent.conversation_compression import (
from agent.context_engine import automatic_compaction_status_message
from agent.display import KawaiiSpinner
from agent.error_classifier import FailoverReason, classify_api_error
from agent.iteration_budget import IterationBudget
from agent.turn_context import (
_compression_warrants_another_preflight_pass,
build_turn_context,
@@ -1386,10 +1383,23 @@ def run_conversation(
# However, providers like Moonshot AI require a separate 'reasoning_content' field
# on assistant messages with tool_calls. We handle both cases here.
request_logger = getattr(agent, "logger", None) or logging.getLogger(__name__)
# Per-agent validation cursor: skips re-json.loads-ing tool_call
# arguments on history messages already validated in a previous
# iteration. Identity-keyed (strong refs) — compression/undo/repair
# rewriting the list breaks the prefix match and forces a re-scan
# from the divergence point. See sanitize_tool_call_arguments.
_sanitize_cursor = getattr(agent, "_sanitize_args_cursor", None)
if _sanitize_cursor is None:
_sanitize_cursor = {}
try:
agent._sanitize_args_cursor = _sanitize_cursor
except Exception:
pass
repaired_tool_calls = agent._sanitize_tool_call_arguments(
messages,
logger=request_logger,
session_id=agent.session_id,
cursor=_sanitize_cursor,
)
if repaired_tool_calls > 0:
request_logger.info(
@@ -1435,6 +1445,12 @@ def run_conversation(
api_msg.pop("display_kind", None)
api_msg.pop("display_metadata", None)
# Durable row identity stamped by _rows_to_conversation so the
# desktop can address a specific persisted message (reactions).
# Bookkeeping, never a provider field — only the chat-completions
# transport strips underscore keys, so drop it centrally here.
api_msg.pop("_row_id", None)
# Inject ephemeral context into the current turn's user message.
# Sources: memory manager prefetch + plugin pre_llm_call hooks
# with target="user_message" (the default). Both are
+4 -5
View File
@@ -138,8 +138,8 @@ def is_paused() -> bool:
def _load_config() -> Dict[str, Any]:
"""Read curator.* config from ~/.hermes/config.yaml. Tolerates missing file."""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator: %s", e)
return {}
@@ -902,7 +902,6 @@ def _reconcile_classification(
Every removed skill is placed in exactly one bucket.
"""
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
heur_pruned = {e["name"] for e in heuristic.get("pruned", [])}
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
@@ -1876,9 +1875,9 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
_acp_args = None
_model_name = ""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
from hermes_cli.runtime_provider import resolve_runtime_provider
_cfg = load_config()
_cfg = load_config_readonly()
_binding = _resolve_review_runtime(_cfg)
_provider, _model_name = _binding.provider, _binding.model
_rp = resolve_runtime_provider(
+2 -2
View File
@@ -147,8 +147,8 @@ def _utc_id(now: Optional[datetime] = None) -> str:
def _load_config() -> Dict[str, Any]:
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator backup: %s", e)
return {}
+2 -2
View File
@@ -197,8 +197,8 @@ def _config_language_cached() -> str | None:
(e.g. after the setup wizard).
"""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
lang = (cfg.get("display") or {}).get("language")
if lang:
return _normalize_lang(lang)
+2 -2
View File
@@ -91,9 +91,9 @@ def get_active_provider() -> Optional[ImageGenProvider]:
"""
configured: Optional[str] = None
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
section = cfg.get("image_gen") if isinstance(cfg, dict) else None
if isinstance(section, dict):
raw = section.get("provider")
-1
View File
@@ -403,7 +403,6 @@ def _category_counts(payload: dict[str, Any]) -> list[tuple[str, int]]:
def category_color_map(payload: dict[str, Any]) -> dict[str, str]:
"""Deterministic, evenly-spread hue per skill category (theme-independent)."""
clusters = _category_counts(payload)
n = max(1, len(clusters))
# Golden-angle hue spacing so adjacent categories never collide in color.
return {cat: rgb_to_hex(_hsl_to_rgb((i * 137.508) % 360, 0.55, 0.62)) for i, (cat, _c) in enumerate(clusters)}
+1 -1
View File
@@ -55,7 +55,7 @@ def register_subparser(subparsers: argparse._SubParsersAction) -> None:
help="Even attempt servers marked manual-install (best effort)",
)
sub_restart = sub.add_parser(
sub.add_parser(
"restart",
help="Tear down running LSP clients (next edit re-spawns)",
)
+21 -1
View File
@@ -40,7 +40,7 @@ from __future__ import annotations
import logging
import os
import threading
from typing import Tuple
from typing import List, Tuple
# Dedicated logger name so the documented grep recipe survives a
# ``logging.getLogger(__name__)`` rename of any internal module.
@@ -188,6 +188,25 @@ def log_spawn_failed(server_id: str, workspace_root: str, exc: BaseException) ->
)
def log_reaped(keys: List[Tuple[str, str]], idle_timeout: float) -> None:
"""Idle clients were shut down by the reaper. INFO — one line per
sweep so users can correlate memory drops with LSP activity.
Also clears the ``log_active`` announce cache for the reaped keys so
a later respawn re-announces at INFO instead of logging a misleading
DEBUG "reused client".
"""
with _announce_lock:
for key in keys:
_announced_active.discard(key)
summary = ", ".join(f"{sid} ({root})" for sid, root in keys)
_emit(
"reaper",
logging.INFO,
f"reaped {len(keys)} idle client(s) after {idle_timeout:.0f}s: {summary}",
)
def reset_announce_caches() -> None:
"""Test-only: clear the dedup caches. Production code never calls this."""
with _announce_lock:
@@ -209,5 +228,6 @@ __all__ = [
"log_timeout",
"log_server_error",
"log_spawn_failed",
"log_reaped",
"reset_announce_caches",
]
+3 -5
View File
@@ -30,7 +30,6 @@ import logging
import os
import shutil
import subprocess
import sys
import threading
from pathlib import Path
from typing import Any, Dict, Optional
@@ -124,10 +123,9 @@ def _is_windows() -> bool:
def hermes_lsp_bin_dir() -> Path:
"""Return the Hermes-owned bin staging dir for LSP servers."""
home = os.environ.get("HERMES_HOME")
if home is None:
home = os.path.join(os.path.expanduser("~"), ".hermes")
p = Path(home) / "lsp" / "bin"
from hermes_constants import get_hermes_home
p = get_hermes_home() / "lsp" / "bin"
p.mkdir(parents=True, exist_ok=True)
return p
+79 -5
View File
@@ -59,6 +59,7 @@ from agent.lsp.workspace import (
logger = logging.getLogger("agent.lsp.manager")
DEFAULT_IDLE_TIMEOUT = 600 # seconds; servers idle for >10min get reaped
MIN_IDLE_TIMEOUT = 30 # floor for config values; must exceed any per-op wait budget
class _BackgroundLoop:
@@ -176,6 +177,7 @@ class LSPService:
self._spawning: Dict[Tuple[str, str], asyncio.Future] = {}
self._last_used: Dict[Tuple[str, str], float] = {}
self._state_lock = threading.Lock()
self._idle_reaper_task: Optional[asyncio.Task] = None
# Delta baseline: file path → snapshot of diagnostics taken
# immediately before a write. ``get_diagnostics_sync`` filters
@@ -183,6 +185,9 @@ class LSPService:
# introduced by the current edit.
self._delta_baseline: Dict[str, List[Dict[str, Any]]] = {}
if self._enabled and self._idle_timeout > 0:
self._loop.run(self._start_idle_reaper(), timeout=2.0)
@classmethod
def create_from_config(cls) -> Optional["LSPService"]:
"""Build a service from ``hermes_cli.config`` settings.
@@ -191,8 +196,8 @@ class LSPService:
itself returns ``is_active()`` False when LSP is disabled.
"""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e: # noqa: BLE001
logger.debug("LSP config load failed: %s", e)
return None
@@ -205,6 +210,16 @@ class LSPService:
wait_mode = lsp_cfg.get("wait_mode", "document")
wait_timeout = float(lsp_cfg.get("wait_timeout", DIAGNOSTICS_DOCUMENT_WAIT))
install_strategy = lsp_cfg.get("install_strategy", "auto")
try:
idle_timeout = float(lsp_cfg.get("idle_timeout", DEFAULT_IDLE_TIMEOUT))
except (TypeError, ValueError):
idle_timeout = DEFAULT_IDLE_TIMEOUT
if 0 < idle_timeout < MIN_IDLE_TIMEOUT:
# A timeout below the per-operation wait budget could reap a
# client mid-flight; the resulting outer timeout would then
# mark the (server, workspace) pair broken for the process
# lifetime. Clamp to a safe floor (0 still disables).
idle_timeout = MIN_IDLE_TIMEOUT
servers_cfg = lsp_cfg.get("servers") or {}
disabled = []
binary_overrides: Dict[str, List[str]] = {}
@@ -235,6 +250,7 @@ class LSPService:
env_overrides=env_overrides,
init_overrides=init_overrides,
disabled_servers=disabled,
idle_timeout=idle_timeout,
)
# ------------------------------------------------------------------
@@ -434,6 +450,7 @@ class LSPService:
# ``_clients`` with a half-initialized state.
with self._state_lock:
client = self._clients.pop(key, None)
self._last_used.pop(key, None)
if client is not None:
try:
# Fire-and-forget shutdown — give it a second to cleanup,
@@ -470,7 +487,7 @@ class LSPService:
except Exception as e: # noqa: BLE001
logger.debug("snapshot open/wait failed: %s", e)
return []
self._last_used[(client.server_id, client.workspace_root)] = time.time()
self._touch(client)
if not fresh:
# No fresh data for the pre-edit content — an empty baseline
# is safe: worst case the delta filter removes less, never
@@ -499,7 +516,7 @@ class LSPService:
except Exception as e: # noqa: BLE001
logger.debug("open/wait failed for %s: %s", file_path, e)
return None
self._last_used[(client.server_id, client.workspace_root)] = time.time()
self._touch(client)
if not fresh:
return None
return list(client.diagnostics_for(file_path, fresh_only=True))
@@ -539,6 +556,7 @@ class LSPService:
with self._state_lock:
client = self._clients.get(key)
if client is not None and client.is_running:
self._last_used[key] = time.time()
eventlog.log_active(srv.server_id, per_server_root)
return client
spawning = self._spawning.get(key)
@@ -589,7 +607,7 @@ class LSPService:
return None
with self._state_lock:
self._clients[key] = client
self._last_used[key] = time.time()
self._last_used[key] = time.time()
eventlog.log_active(srv.server_id, per_server_root)
spawn_future.set_result(client)
return client
@@ -597,7 +615,63 @@ class LSPService:
with self._state_lock:
self._spawning.pop(key, None)
async def _start_idle_reaper(self) -> None:
self._idle_reaper_task = asyncio.create_task(self._idle_reaper_loop())
def _touch(self, client: LSPClient) -> None:
"""Refresh the last-used timestamp for a client we just used.
Guarded on membership so a reaped-mid-operation client can't
resurrect an orphan ``_last_used`` entry after the reaper popped
the key. All writers and the reaper run on the background loop
thread; the lock keeps this consistent with the reader anyway.
"""
key = (client.server_id, client.workspace_root)
with self._state_lock:
if key in self._clients:
self._last_used[key] = time.time()
async def _idle_reaper_loop(self) -> None:
interval = min(60.0, self._idle_timeout)
while True:
await asyncio.sleep(interval)
try:
await self._reap_idle_once()
except asyncio.CancelledError:
raise
except Exception as e: # noqa: BLE001
# A transient sweep error must not kill the reaper —
# otherwise one bad shutdown permanently re-opens the
# unbounded-accumulation leak this loop exists to fix.
logger.debug("LSP idle reaper sweep error: %s", e)
async def _reap_idle_once(self) -> None:
cutoff = time.time() - self._idle_timeout
with self._state_lock:
idle_keys = [
key
for key in self._clients
if self._last_used.get(key, 0) < cutoff
]
clients = [self._clients.pop(key) for key in idle_keys]
for key in idle_keys:
self._last_used.pop(key, None)
if clients:
eventlog.log_reaped(
[(c.server_id, c.workspace_root) for c in clients],
self._idle_timeout,
)
await asyncio.gather(
*(client.shutdown() for client in clients),
return_exceptions=True,
)
async def _shutdown_async(self) -> None:
reaper = self._idle_reaper_task
self._idle_reaper_task = None
if reaper is not None:
reaper.cancel()
await asyncio.gather(reaper, return_exceptions=True)
with self._state_lock:
clients = list(self._clients.values())
self._clients.clear()
+6 -6
View File
@@ -710,9 +710,9 @@ def _find_pses_bundle(ctx: ServerContext) -> Optional[str]:
env_path = os.environ.get("PSES_BUNDLE_PATH")
if env_path:
candidates.append(env_path)
home = os.environ.get("HERMES_HOME") or os.path.join(
os.path.expanduser("~"), ".hermes"
)
from hermes_constants import get_hermes_home
home = str(get_hermes_home())
candidates.append(os.path.join(home, "lsp", "PowerShellEditorServices"))
for cand in candidates:
@@ -796,9 +796,9 @@ def _spawn_powershell_es(root: str, ctx: ServerContext) -> Optional[SpawnSpec]:
def hermes_lsp_session_dir() -> str:
"""Return (and create) the dir for PSES session/log scratch files."""
home = os.environ.get("HERMES_HOME") or os.path.join(
os.path.expanduser("~"), ".hermes"
)
from hermes_constants import get_hermes_home
home = str(get_hermes_home())
d = os.path.join(home, "lsp", "pses")
os.makedirs(d, exist_ok=True)
return d
+375
View File
@@ -14,6 +14,7 @@ re-exports from ``run_agent`` remain in place so existing imports
from __future__ import annotations
import hashlib
import json
import logging
import re
@@ -474,4 +475,378 @@ __all__ = [
"_sanitize_tools_non_ascii",
"_strip_images_from_messages",
"_sanitize_structure_non_ascii",
# call_id policy owners (F4 consolidation)
"deterministic_call_id",
"coalesce_tool_call_id",
"uniquify_tool_call_ids",
# reasoning_content policy owners (F4 consolidation)
"reasoning_echo_family",
"matches_reasoning_echo_family",
"needs_reasoning_echo",
"apply_reasoning_content_policy",
"reapply_reasoning_echo",
]
# ---------------------------------------------------------------------------
# call_id policy — single owner (audit F4, incident chain I4)
# ---------------------------------------------------------------------------
#
# Three forked policy sites converged here:
# * agent/codex_responses_adapter.py `_deterministic_call_id` — hash
# synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2).
# * run_agent.AIAgent._get_tool_call_id_static — `call_id or id`
# coalescing for dicts and SDK objects.
# * run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with
# deterministic `_d<n>` suffixes (#58327 loss class).
#
# NOT consolidated (different scheme on purpose):
# agent/transports/codex_event_projector._deterministic_call_id maps codex
# app-server ITEM ids (`codex_<type>_<item_id>`), not chat tool-call
# content; merging the two would change ids and invalidate prompt caches.
#
# HARD INVARIANT: everything here must stay deterministic (never uuid4) and
# byte-identical for existing inputs — these ids feed prompt-cache prefixes.
def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return f"call_{digest}"
def coalesce_tool_call_id(tc: Any) -> str:
"""Extract the effective call ID from a tool_call entry (dict or object).
Single owner for the ``call_id or id`` coalescing rule: Codex Responses
tool calls carry ``call_id`` (authoritative pairing key), Chat
Completions ones carry ``id`` only. Returns ``""`` when neither is set.
"""
if isinstance(tc, dict):
return (tc.get("call_id", "") or tc.get("id", "") or "").strip()
return (getattr(tc, "call_id", "") or getattr(tc, "id", "") or "").strip()
def uniquify_tool_call_ids(tool_calls: list) -> list:
"""Ensure every tool call in a single assistant turn has a distinct id.
Some models/providers reuse one call id across different calls in a
single batch (observed with native Kimi Responses replays, Ollama-
compatible endpoints, and degraded models at long context; same bug
class as openclaw/openclaw#110518 / #110956). Duplicate ids are lossy
downstream: the pre-API sanitizer keeps only the first call/result
pair per id (#58327), so the later call's result silently vanishes
from every replayed payload, and strict providers (Anthropic
tool_use, DeepSeek) reject duplicate ids outright.
The first occurrence keeps its id; later collisions get a
deterministic ``<id>_d<n>`` suffix — never a random UUID, which would
break prompt-cache prefix stability across replays. Mutates the
entries in place (SDK models / SimpleNamespace / dicts) and returns
the same list. Blank/missing ids are left for the deterministic
fallback in ``build_assistant_message``.
"""
seen: set = set()
for tc in tool_calls or []:
# Same coalescing rule as ``coalesce_tool_call_id`` but tolerant of
# non-string ids (degraded models can emit ints/None here).
if isinstance(tc, dict):
raw = tc.get("call_id") or tc.get("id") or ""
else:
raw = getattr(tc, "call_id", None) or getattr(tc, "id", None) or ""
raw = raw.strip() if isinstance(raw, str) else ""
if not raw:
continue
# Composite Responses ids ("call_x|fc_y") collide on the call
# half — that's the pairing key providers enforce per turn.
cid = raw.split("|", 1)[0]
if not cid:
continue
if cid not in seen:
seen.add(cid)
continue
n = 2
new_id = f"{cid}_d{n}"
while new_id in seen:
n += 1
new_id = f"{cid}_d{n}"
seen.add(new_id)
def _renamed(value):
# Preserve a composite id's response-item half so the
# provider's real fc_/item id survives the rename.
if isinstance(value, str) and "|" in value:
return f"{new_id}|{value.split('|', 1)[1]}"
return new_id
try:
if isinstance(tc, dict):
if tc.get("id"):
tc["id"] = _renamed(tc["id"])
else:
tc["id"] = new_id
if tc.get("call_id"):
tc["call_id"] = new_id
else:
tc.id = _renamed(getattr(tc, "id", None))
if getattr(tc, "call_id", None):
tc.call_id = new_id
except Exception:
logger.warning(
"Could not uniquify duplicate tool call id %s", cid
)
continue
_fn = tc.get("function") if isinstance(tc, dict) else getattr(tc, "function", None)
_fn_name = (_fn.get("name") if isinstance(_fn, dict) else getattr(_fn, "name", None)) or "?"
logger.warning(
"Model reused tool call id %s within one turn; renamed the "
"duplicate to %s (tool=%s) to keep call/result pairing "
"lossless.", cid, new_id, _fn_name,
)
return tool_calls
# ---------------------------------------------------------------------------
# reasoning_content policy — single owner (audit F4)
# ---------------------------------------------------------------------------
#
# The strip-vs-repad decision was previously forked across the wire files in
# separate incident commits (2b3a4f0af8 strip for strict providers,
# b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi pad). The
# POLICY — which provider direction gets which treatment — lives here as one
# rule table + apply functions; adapters keep only SYNTAX mapping (e.g.
# anthropic_adapter turning reasoning_content into a thinking block).
#
# Direction table:
# require-side (echo-back enforced; replays 400 without the field):
# kimi — provider kimi-coding/kimi-coding-cn, or host api.kimi.com /
# moonshot.ai / moonshot.cn. Host-driven on purpose:
# aggregators re-exporting kimi models reject the echo.
# deepseek — provider "deepseek", model contains "deepseek", or host
# api.deepseek.com (#15250; V4 rejects empty-string pads,
# hence the " " single-space pad, #17341).
# mimo — provider "xiaomi", model contains "mimo", or host
# *.xiaomimimo.com.
# strict side (field rejected with 400/422 "Extra inputs are not
# permitted"): everyone else — Mistral, Cerebras, Groq, SambaNova, …
# (#45655). Strip the key entirely, even a single-space pad.
_REASONING_ECHO_RULES: tuple = (
# (family, exact providers (raw), exact providers (lowered),
# model substrings (lowered), base_url hosts)
("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (),
("api.kimi.com", "moonshot.ai", "moonshot.cn")),
("deepseek", frozenset(), frozenset({"deepseek"}), ("deepseek",),
("api.deepseek.com",)),
("mimo", frozenset(), frozenset({"xiaomi"}), ("mimo",),
("api.xiaomimimo.com", "xiaomimimo.com")),
)
def _family_rule(family: str) -> tuple:
for rule in _REASONING_ECHO_RULES:
if rule[0] == family:
return rule
raise KeyError(family)
def matches_reasoning_echo_family(
family: str, provider: Any, model: Any, base_url: Any
) -> bool:
"""True when (provider, model, base_url) matches one echo-back family.
Families can overlap (e.g. a deepseek-named model pointed at a kimi
host); this membership test is independent per family so per-family
predicates keep their original semantics.
"""
from utils import base_url_host_matches
_, raw_providers, lowered_providers, model_subs, hosts = _family_rule(family)
provider_lower = (provider or "").lower()
model_lower = (model or "").lower()
if provider in raw_providers or provider_lower in lowered_providers:
return True
if any(sub in model_lower for sub in model_subs):
return True
return any(base_url_host_matches(base_url, host) for host in hosts)
def reasoning_echo_family(provider: Any, model: Any, base_url: Any) -> "str | None":
"""Classify the provider direction for the reasoning_content echo policy.
Returns ``"kimi"``, ``"deepseek"``, or ``"mimo"`` (first match in table
order) when the target endpoint enforces reasoning_content echo-back on
assistant turns, else ``None`` (strict/indifferent side — the field must
be stripped).
"""
for rule in _REASONING_ECHO_RULES:
if matches_reasoning_echo_family(rule[0], provider, model, base_url):
return rule[0]
return None
def needs_reasoning_echo(provider: Any, model: Any, base_url: Any) -> bool:
"""True when the endpoint requires reasoning_content echo-back."""
return reasoning_echo_family(provider, model, base_url) is not None
def apply_reasoning_content_policy(
source_msg: dict, api_msg: dict, needs_thinking_pad: bool
) -> None:
"""Copy provider-facing reasoning fields onto an API replay message.
``needs_thinking_pad`` is the require-side flag (see
``needs_reasoning_echo`` / the agent's cached
``_needs_thinking_reasoning_pad``). Mutates ``api_msg`` in place.
"""
if source_msg.get("role") != "assistant":
return
# 1. Explicit reasoning_content already set.
#
# When the active provider enforces the thinking-mode echo-back
# (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their
# own space-placeholder written at creation time and any valid reasoning
# from the same provider. Sessions persisted BEFORE #17341 have
# empty-string placeholders pinned at creation time; DeepSeek V4 Pro
# rejects those with HTTP 400, so upgrade "" → " " on replay.
#
# When the active provider does NOT enforce echo-back, strip the field
# entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq,
# SambaNova, …) reject ANY reasoning_content key in input messages with
# HTTP 400/422 ("Extra inputs are not permitted"), even an empty string
# or a single-space pad. This is the cross-provider fallback case: a
# reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a
# fallback to a strict provider replays that pad and 422s. Stripping
# here covers the rebuild path; ``reapply_reasoning_echo`` covers the
# already-built api_messages path. Refs #45655.
existing = source_msg.get("reasoning_content")
if isinstance(existing, str):
if not needs_thinking_pad:
api_msg.pop("reasoning_content", None)
elif existing == "":
api_msg["reasoning_content"] = " "
else:
api_msg["reasoning_content"] = existing
return
# 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi,
# if the source turn has tool_calls AND a 'reasoning' field but no
# 'reasoning_content' key, the 'reasoning' text was written by a
# prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message
# pins reasoning_content at creation time for tool-call turns, so the
# shape (reasoning set, reasoning_content absent, tool_calls present)
# is unreachable from same-provider DeepSeek history after this fix.
# Inject a single space to satisfy the API without leaking another
# provider's chain of thought to DeepSeek/Kimi. Space (not "")
# because DeepSeek V4 Pro rejects empty-string reasoning_content
# in thinking mode (refs #17341).
normalized_reasoning = source_msg.get("reasoning")
if (
needs_thinking_pad
and source_msg.get("tool_calls")
and isinstance(normalized_reasoning, str)
and normalized_reasoning
):
api_msg["reasoning_content"] = " "
return
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content'
# for providers that use the internal 'reasoning' key.
# This must happen before the unconditional empty-string fallback so
# genuine reasoning content is not overwritten (#15812 regression in
# PR #15478). Only promote for providers that enforce echo-back —
# strict providers reject the field (refs #45655).
if isinstance(normalized_reasoning, str) and normalized_reasoning:
if needs_thinking_pad:
api_msg["reasoning_content"] = normalized_reasoning
else:
api_msg.pop("reasoning_content", None)
return
# 4. DeepSeek / Kimi thinking mode: all assistant messages need
# reasoning_content. Inject a single space to satisfy the provider's
# requirement when no explicit reasoning content is present. Covers
# both tool-call turns (already-poisoned history with no reasoning
# at all) and plain text turns. Space (not "") because DeepSeek V4
# Pro tightened validation and rejects empty string with HTTP 400
# ("The reasoning content in the thinking mode must be passed back
# to the API"). Refs #17341.
if needs_thinking_pad:
api_msg["reasoning_content"] = " "
return
# 5. reasoning_content was present but not a string (e.g. None after
# context compaction). Don't pass null to the API.
api_msg.pop("reasoning_content", None)
def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int:
"""Re-pad (or strip) assistant turns' reasoning_content for the active provider.
``api_messages`` is built once, before the retry loop, while the *primary*
provider is active. A mid-conversation fallback can then switch providers,
so the reasoning fields baked into ``api_messages`` are shaped for the
*prior* provider and must be reconciled against the *current* one:
* Switching TO a require-side provider (DeepSeek / Kimi / MiMo thinking
mode): assistant turns built when the prior provider did NOT need the
echo-back go out without ``reasoning_content`` and the new provider
rejects them with HTTP 400 ("The reasoning_content in the thinking mode
must be passed back"). Re-apply the pad.
* Switching TO a strict provider that rejects the field (Mistral,
Cerebras, Groq, SambaNova, …): assistant turns built under a reasoning
primary carry a ``reasoning_content`` pad (often a single space ``" "``),
and the strict provider rejects it with HTTP 400/422 ("Extra inputs are
not permitted"). Strip the field. This is the exact cross-provider
fallback bug from #45655 — a DeepSeek primary pads history with ``" "``,
the request falls back to Mistral, and Mistral 422s on the stale pad.
Calling this immediately before building the request kwargs reconciles the
fields against the *current* provider. It is idempotent and safe to call
every iteration; it covers every fallback path.
Returns the number of assistant turns whose reasoning_content was added or
removed.
"""
changed = 0
for api_msg in api_messages:
if api_msg.get("role") != "assistant":
continue
if needs_thinking_pad:
if api_msg.get("reasoning_content"):
continue
apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad)
if api_msg.get("reasoning_content"):
changed += 1
else:
# Strict provider — strip any stale reasoning_content pad left
# over from a reasoning primary so the fallback request doesn't
# 400/422 on it.
if "reasoning_content" in api_msg:
api_msg.pop("reasoning_content", None)
changed += 1
return changed
# ---------------------------------------------------------------------------
# Image / multimodal parts — evaluated, NOT consolidated (verdict: syntax)
# ---------------------------------------------------------------------------
#
# The per-adapter image handling is format-specific SYNTAX, not shared policy:
# * anthropic_adapter (~1817): data-URL → Anthropic `source: {type: base64}`
# block mapping — Anthropic wire shape only.
# * codex_responses_adapter (~113/165/812): chat `image_url` parts →
# Responses `input_image` items and image counting for log summaries —
# Responses wire shape only.
# * transports/chat_completions: pass-through (native format).
# The one genuinely shared image POLICY — removing images when a server
# rejects them while preserving tool_call_id pairing — already has a single
# owner here: ``_strip_images_from_messages`` above.
+195 -11
View File
@@ -13,18 +13,40 @@ import os
import re
import time
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from typing import Any, Dict, List, Optional, Tuple, TYPE_CHECKING
from urllib.parse import urlparse
import requests
import yaml
if TYPE_CHECKING: # pragma: no cover — runtime import is lazy (see below)
import requests
from utils import atomic_json_write, base_url_host_matches, base_url_hostname
from hermes_constants import OPENROUTER_MODELS_URL
logger = logging.getLogger(__name__)
# ``requests`` (with urllib3) costs ~27 ms of the `import cli` waterfall and
# is only used inside the fetch functions below. It's resolved lazily:
# ``_ensure_requests()`` populates the module global on the runtime path, and
# the PEP 562 ``__getattr__`` covers external attribute access — notably
# ``patch("agent.model_metadata.requests.get")`` in tests, which resolves the
# attribute at patch time.
def _ensure_requests():
if "requests" not in globals():
import requests as _requests
globals()["requests"] = _requests
return globals()["requests"]
def __getattr__(name: str):
if name == "requests":
return _ensure_requests()
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
def _resolve_requests_verify() -> bool | str:
"""Resolve SSL verify setting for `requests` calls from env vars.
@@ -50,7 +72,7 @@ def _resolve_requests_verify() -> bool | str:
_PROVIDER_PREFIXES: frozenset[str] = frozenset({
"openrouter", "nous", "openai-codex", "copilot", "copilot-acp",
"gemini", "ollama-cloud", "zai", "kimi-coding", "kimi-coding-cn", "stepfun", "minimax", "minimax-oauth", "minimax-cn", "anthropic", "deepseek", "deepinfra",
"opencode-zen", "opencode-go", "kilocode", "alibaba", "novita",
"opencode-zen", "opencode-go", "ai-gateway", "kilocode", "alibaba", "novita",
"qwen-oauth",
"xiaomi",
"arcee",
@@ -62,7 +84,7 @@ _PROVIDER_PREFIXES: frozenset[str] = frozenset({
"glm", "z-ai", "z.ai", "zhipu", "github", "github-copilot",
"github-models", "kimi", "moonshot", "kimi-cn", "moonshot-cn", "claude", "deep-seek", "deep-infra",
"ollama",
"stepfun", "opencode", "zen", "go", "kilo", "dashscope", "aliyun", "qwen",
"stepfun", "opencode", "zen", "go", "vercel", "kilo", "dashscope", "aliyun", "qwen",
"mimo", "xiaomi-mimo",
"tencent", "tokenhub", "tencent-cloud", "tencentmaas",
"arcee-ai", "arceeai",
@@ -123,6 +145,67 @@ _ENDPOINT_MODEL_CACHE_TTL = 300
_ENDPOINT_PROBE_TTL_SECONDS = 3600.0
_endpoint_probe_path_cache: Dict[str, tuple] = {}
# ── Disk L2 for local-endpoint probe results ────────────────────────────────
# The in-process caches above die with the process, so every CLI cold start
# with a local model re-paid the probe waterfall in AIAgent.__init__:
# detect_local_server_type (up to 4 HTTP GETs, ≤2 s each on a hung server)
# + /api/show (≤3 s). A short-TTL disk cache makes back-to-back CLI
# invocations hit disk instead of the network. Only SUCCESSFUL probes are
# persisted (a down server must not pin a negative verdict), and the TTL is
# short enough that swapping the server on a port (stop Ollama, start
# LM Studio) is picked up within minutes — strictly fresher than the 1 h
# in-process TTL that already accepts that staleness.
_LOCAL_PROBE_DISK_TTL_SECONDS = 300.0
def _local_probe_disk_cache_path() -> Path:
from hermes_constants import get_hermes_home
return get_hermes_home() / "cache" / "local_endpoint_probes.json"
def _load_local_probe_disk_cache() -> Dict[str, Any]:
try:
with _local_probe_disk_cache_path().open("r", encoding="utf-8") as f:
data = json.load(f)
return data if isinstance(data, dict) else {}
except Exception:
return {}
def _local_probe_disk_get(kind: str, key: str) -> Optional[Any]:
"""Return a fresh cached value for ``kind:key``, else None."""
entry = _load_local_probe_disk_cache().get(f"{kind}:{key}")
if not isinstance(entry, dict):
return None
try:
if (time.time() - float(entry["ts"])) >= _LOCAL_PROBE_DISK_TTL_SECONDS:
return None
return entry["value"]
except Exception:
return None
def _local_probe_disk_put(kind: str, key: str, value: Any) -> None:
"""Persist a successful probe result. Best-effort; prunes stale entries."""
try:
now = time.time()
data = _load_local_probe_disk_cache()
data = {
k: v
for k, v in data.items()
if isinstance(v, dict)
and (now - float(v.get("ts", 0))) < _LOCAL_PROBE_DISK_TTL_SECONDS
}
data[f"{kind}:{key}"] = {"value": value, "ts": now}
atomic_json_write(
_local_probe_disk_cache_path(),
data,
indent=0,
separators=(",", ":"),
)
except Exception as e:
logger.debug("Failed to save local probe disk cache: %s", e)
def _get_model_metadata_cache_path() -> Path:
"""Return path to the OpenRouter model metadata disk cache."""
@@ -752,6 +835,13 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
if cached is not None and (time.monotonic() - cached[1]) < _ENDPOINT_PROBE_TTL_SECONDS:
return cached[0]
# Disk L2: a fresh cross-process verdict skips the HTTP waterfall
# entirely (back-to-back CLI invocations, cron ticks).
disk_hit = _local_probe_disk_get("server_type", server_url)
if isinstance(disk_hit, str):
_endpoint_probe_path_cache[server_url] = (disk_hit, time.monotonic())
return disk_hit
headers = _auth_headers(api_key)
result: Optional[str] = None
@@ -804,6 +894,7 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
if result is not None:
_endpoint_probe_path_cache[server_url] = (result, time.monotonic())
_local_probe_disk_put("server_type", server_url, result)
return result
@@ -926,6 +1017,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any
return _model_metadata_cache
try:
_ensure_requests()
# Tuple (connect, read) — flat timeout=10 means urllib3 can block 10s per
# retry stage through proxies that 403 CONNECT, ballooning to minutes
# (#46620). 5s connect / 10s read fails fast on unreachable hosts.
@@ -982,6 +1074,7 @@ def fetch_endpoint_model_metadata(
normalized = _normalize_base_url(base_url)
if not normalized or _is_openrouter_base_url(normalized):
return {}
_ensure_requests()
if not force_refresh:
cached = _endpoint_model_metadata_cache.get(normalized)
@@ -1523,6 +1616,13 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
if server_type != "ollama":
return None
# Disk L2: /api/show results are stable for a given (model, server) on
# human timescales — skip the HTTP roundtrip on fresh cross-process hits.
_disk_key = f"{server_url}|{bare_model}"
disk_hit = _local_probe_disk_get("ollama_num_ctx", _disk_key)
if isinstance(disk_hit, int) and disk_hit > 0:
return disk_hit
headers = _auth_headers(api_key)
try:
@@ -1540,7 +1640,9 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
parts = line.strip().split()
if len(parts) >= 2:
try:
return int(parts[-1])
_ctx = int(parts[-1])
_local_probe_disk_put("ollama_num_ctx", _disk_key, _ctx)
return _ctx
except ValueError:
pass
@@ -1548,7 +1650,9 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
model_info = data.get("model_info", {})
for key, value in model_info.items():
if "context_length" in key and isinstance(value, (int, float)):
return int(value)
_ctx = int(value)
_local_probe_disk_put("ollama_num_ctx", _disk_key, _ctx)
return _ctx
except Exception:
pass
return None
@@ -1882,6 +1986,7 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) ->
"x-api-key": api_key,
"anthropic-version": "2023-06-01",
}
_ensure_requests()
resp = requests.get(url, headers=headers, timeout=(5, 10), verify=_resolve_requests_verify())
if resp.status_code != 200:
return None
@@ -1989,6 +2094,7 @@ def _fetch_codex_oauth_context_lengths_with_source(
headers["ChatGPT-Account-Id"] = acct_id
try:
_ensure_requests()
resp = requests.get(
"https://chatgpt.com/backend-api/codex/models?client_version=1.0.0",
headers=headers,
@@ -2745,14 +2851,92 @@ def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
image — the Anthropic pricing model — instead of counting raw base64
character length. Without this, a single ~1MB screenshot would be
estimated at ~250K tokens and trigger premature context compression.
Per-message results are memoized (see ``_estimate_message_tokens_cached``)
keyed on a deep *identity fingerprint* of the message, so re-walking a
long history every iteration only pays for messages whose object graph
actually changed. The memo is exact: equal fingerprints imply identical
leaf objects and structure, hence an identical estimate.
"""
_IMAGE_TOKEN_COST = 1500
text_tokens = 0
image_tokens = 0
total = 0
for msg in messages:
text_tokens += _estimate_message_tokens_without_images(msg)
image_tokens += _count_image_tokens(msg, _IMAGE_TOKEN_COST)
return text_tokens + image_tokens
total += _estimate_message_tokens_cached(msg, _IMAGE_TOKEN_COST)
return total
# --- Per-message token-estimate memo -------------------------------------
#
# ``estimate_messages_tokens_rough`` is called on the full history every
# loop iteration (conversation_loop preflight), repeatedly during compaction
# telemetry, and inside an O(n^2) shrink loop in moa_loop. The per-message
# helpers are pure functions of the message's value, so a memo keyed on a
# fingerprint that uniquely determines the value is exactly equivalent.
#
# Fingerprint design (soundness argument):
# * strings are fingerprinted by ``id()`` AND pinned (a strong reference is
# stored in the cache entry). While the entry lives, that id cannot be
# reused by another object, so id-equality implies object-equality —
# strings are immutable, so value-equality too (no #50372-style aliasing).
# * ints/floats/bools/None are fingerprinted by value.
# * dicts/lists recurse structurally, preserving key order — ``str(shadow)``
# depends on insertion order, so order is part of the key.
# * any other type aborts the memo and falls through to a direct compute.
# Equal fingerprints therefore imply deep-equal messages built from identical
# immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical estimate.
#
# Because the api_messages build shallow-copies history dicts each iteration,
# the copies share the same content strings — so unchanged history messages
# hit the memo even though the outer dicts are fresh objects every turn.
_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {}
_MSG_TOKENS_CACHE_MAX = 4096
def _msg_fingerprint(value: Any, pins: list) -> Any:
if value is None or value is True or value is False:
return value
t = type(value)
if t is str:
pins.append(value)
return ("s", id(value))
if t is int or t is float:
return ("n", t.__name__, value)
if t is dict:
return ("d", tuple(
(_msg_fingerprint(k, pins), _msg_fingerprint(v, pins))
for k, v in value.items()
))
if t is list:
return ("l", tuple(_msg_fingerprint(v, pins) for v in value))
if t is tuple:
return ("t", tuple(_msg_fingerprint(v, pins) for v in value))
raise ValueError("unfingerprintable message value")
def _estimate_message_tokens_cached(msg: Any, image_cost: int) -> int:
try:
pins: list = []
key = _msg_fingerprint(msg, pins)
hash(key)
except Exception:
return (
_estimate_message_tokens_without_images(msg)
+ _count_image_tokens(msg, image_cost)
)
cached = _MSG_TOKENS_CACHE.get(key)
if cached is not None:
return cached[1]
tokens = (
_estimate_message_tokens_without_images(msg)
+ _count_image_tokens(msg, image_cost)
)
_MSG_TOKENS_CACHE[key] = (pins, tokens)
while len(_MSG_TOKENS_CACHE) > _MSG_TOKENS_CACHE_MAX:
try:
_MSG_TOKENS_CACHE.pop(next(iter(_MSG_TOKENS_CACHE)))
except (StopIteration, KeyError, RuntimeError):
break
return tokens
def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
+236 -58
View File
@@ -8,11 +8,15 @@ of 4000+ models across 109+ providers. Provides:
(reasoning, tools, vision, PDF, audio), modalities, knowledge cutoff,
open-weights flag, family grouping, deprecation status
Data resolution order (like TypeScript OpenCode):
1. Bundled snapshot (ships with the package — offline-first)
2. Disk cache (~/.hermes/models_dev_cache.json)
3. Network fetch (https://models.dev/api.json)
4. Background refresh every 60 minutes
Data resolution order:
1. In-memory cache (fresh, or stale served immediately while a single
background daemon thread refreshes)
2. Disk cache (~/.hermes/models_dev_cache.json — any age; stale data is
served rather than blocking callers on the network)
3. Network fetch (https://models.dev/api.json) — only when no cache
exists at all; failed refreshes back off for 5 minutes process-wide
Latency-sensitive callers (gateway route-identity checks) pass
``allow_network=False`` and never touch the network.
Other modules should import the dataclasses and query functions from here
rather than parsing the raw JSON themselves.
@@ -20,6 +24,7 @@ rather than parsing the raw JSON themselves.
import json
import logging
import threading
import time
from dataclasses import dataclass
from pathlib import Path
@@ -33,10 +38,15 @@ logger = logging.getLogger(__name__)
MODELS_DEV_URL = "https://models.dev/api.json"
_MODELS_DEV_CACHE_TTL = 3600 # 1 hour in-memory
_MODELS_DEV_RETRY_DELAY = 300 # 5 minutes after a failed refresh
# In-memory cache
_models_dev_cache: Dict[str, Any] = {}
_models_dev_cache_time: float = 0
_models_dev_retry_after: float = 0
_models_dev_fetch_lock = threading.Lock()
_models_dev_refresh_lock = threading.Lock()
_models_dev_refresh_in_flight = False
# ---------------------------------------------------------------------------
@@ -158,6 +168,7 @@ PROVIDER_TO_MODELS_DEV: Dict[str, str] = {
"alibaba": "alibaba",
"qwen-oauth": "alibaba",
"copilot": "github-copilot",
"ai-gateway": "vercel",
"opencode-zen": "opencode",
"opencode-go": "opencode-go",
"kilocode": "kilo",
@@ -237,27 +248,157 @@ def _save_disk_cache(data: Dict[str, Any]) -> None:
logger.debug("Failed to save models.dev disk cache: %s", e)
def fetch_models_dev(force_refresh: bool = False) -> Dict[str, Any]:
def _fetch_models_dev_from_network() -> Dict[str, Any]:
"""Fetch the live models.dev registry without touching local caches.
Raises on network errors and on an empty/invalid registry payload.
"""
# Tuple (connect, read): a flat timeout=15 let a blackholed connect
# stall the first-turn critical path for the full 15 s. 5 s connect
# fails fast on unreachable hosts; 10 s read still tolerates a slow
# registry response (matches the OpenRouter fetch convention in
# agent/model_metadata.py).
response = requests.get(MODELS_DEV_URL, timeout=(5, 10))
response.raise_for_status()
data = response.json()
if not isinstance(data, dict) or not data:
raise ValueError("models.dev returned an empty or invalid registry")
return data
def _mark_stale_cache_grace() -> None:
"""Give stale cache data a short in-memory grace before retrying refresh.
Only ever moves the timestamp forward: if a background refresh completed
between the caller's staleness check and this call, the fresh timestamp
is preserved instead of being rewound to a 5-minute grace.
"""
global _models_dev_cache_time
grace_time = time.time() - _MODELS_DEV_CACHE_TTL + _MODELS_DEV_RETRY_DELAY
if grace_time > _models_dev_cache_time:
_models_dev_cache_time = grace_time
def _commit_registry(data: Dict[str, Any], *, where: str) -> None:
"""Persist a freshly fetched registry: disk + in-mem + clear backoff.
Callers must hold ``_models_dev_fetch_lock`` so a failing refresh on one
path can never stomp the state a succeeding refresh on the other path
just committed (e.g. a failing background worker re-arming the backoff
immediately after a successful ``force_refresh``).
"""
global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after
_save_disk_cache(data)
_models_dev_cache = data
_models_dev_cache_time = time.time()
_models_dev_retry_after = 0
logger.debug(
"Refreshed models.dev registry (%s): %d providers, %d total models",
where,
len(data),
sum(len(p.get("models", {})) for p in data.values() if isinstance(p, dict)),
)
def _note_refresh_failure(exc: Exception, *, where: str) -> None:
"""Record a failed refresh: arm the process-wide 5-minute backoff.
Callers must hold ``_models_dev_fetch_lock`` (see ``_commit_registry``).
"""
global _models_dev_retry_after
_models_dev_retry_after = time.time() + _MODELS_DEV_RETRY_DELAY
logger.debug(
"models.dev refresh failed (%s); retry suppressed for %ds: %s",
where,
_MODELS_DEV_RETRY_DELAY,
exc,
)
def _background_refresh_models_dev() -> None:
"""Best-effort refresh after serving stale cache data."""
global _models_dev_refresh_in_flight
try:
data = _fetch_models_dev_from_network()
with _models_dev_fetch_lock:
_commit_registry(data, where="background")
except Exception as e:
with _models_dev_fetch_lock:
_note_refresh_failure(e, where="background")
finally:
with _models_dev_refresh_lock:
_models_dev_refresh_in_flight = False
def _start_background_refresh_models_dev() -> None:
"""Start one daemon refresh worker if none is already running.
Honors the process-wide failure backoff: after a failed refresh,
no new background worker is spawned until ``_models_dev_retry_after``.
"""
global _models_dev_refresh_in_flight
if time.time() < _models_dev_retry_after:
return
with _models_dev_refresh_lock:
if _models_dev_refresh_in_flight:
return
_models_dev_refresh_in_flight = True
thread = threading.Thread(
target=_background_refresh_models_dev,
name="models-dev-refresh",
daemon=True,
)
try:
thread.start()
except Exception as e:
# Thread/fd exhaustion: clear the flag so refresh isn't disabled
# for the rest of the process lifetime. Callers still get stale data.
with _models_dev_refresh_lock:
_models_dev_refresh_in_flight = False
logger.debug("Failed to start models.dev refresh thread: %s", e)
def fetch_models_dev(
force_refresh: bool = False, *, allow_network: bool = True
) -> Dict[str, Any]:
"""Fetch models.dev registry. Cache hierarchy: in-mem → disk → network.
Returns the full registry dict keyed by provider ID, or empty dict on failure.
Cache hierarchy (when ``force_refresh=False``):
1. In-memory cache, populated and < TTL old → return immediately.
2. **Disk cache file < TTL old by mtime → load, populate in-mem, return.**
No network call. Saves ~500 ms per cold-start agent construction;
``models.dev`` only changes when providers add new models, so a
1 hour staleness window is acceptable (same TTL as in-mem cache).
3. Network fetch → on success, save to disk + in-mem and return.
4. Network fails → fall back to ANY available disk cache (even stale)
with a short 5 min in-mem grace period before retrying network.
1. Fresh in-memory cache → return immediately.
2. Stale in-memory cache → return immediately and refresh in a single
background daemon thread. Callers never block on the network while
any cache exists; ``models.dev`` only changes when providers add
new models, so stale data is preferable to a foreground timeout.
3. Disk cache file (any age) → load, populate in-mem, return
immediately. Stale disk caches trigger the same background refresh.
4. No cache at all → singleflight foreground network fetch. On
success, save to disk + in-mem and return.
5. Any failed refresh (foreground or background) suppresses further
automatic refreshes for 5 minutes process-wide.
When ``force_refresh=True`` (used by ``hermes config refresh``, the
\"refresh model catalog\" code path), stages 1 and 2 are skipped. The
function always hits the network and only falls back to disk if the
network call fails.
\"refresh model catalog\" code path), cache fast paths and the failure
backoff are bypassed; the function hits the network and only falls back
to cached data if the call fails. When ``allow_network=False``, any
memory or disk cache is returned regardless of age and no request is
made — used by latency-sensitive paths (gateway route-identity checks)
that must never wait on the network.
"""
global _models_dev_cache, _models_dev_cache_time
global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after
if not allow_network:
if _models_dev_cache:
return _models_dev_cache
disk_data = _load_disk_cache()
if disk_data:
_models_dev_cache = disk_data
disk_age = _disk_cache_age_seconds()
_models_dev_cache_time = (
time.time() - disk_age if disk_age is not None else 0
)
return _models_dev_cache
# Stage 1: fresh in-memory cache wins. This is the hot path on
# long-lived processes — no I/O, no system calls.
@@ -268,54 +409,82 @@ def fetch_models_dev(force_refresh: bool = False) -> Dict[str, Any]:
):
return _models_dev_cache
# Stage 2: fresh-by-mtime disk cache short-circuits the network call.
# Only kicks in on cold-start processes (in-mem cache is empty or
# expired) and only when the user hasn't asked for a forced refresh.
# Skipped if the disk cache file is missing, unreadable, or older
# than _MODELS_DEV_CACHE_TTL.
# Stage 2: stale in-memory cache is still better than blocking provider
# resolution on a foreground network timeout. Refresh it in the background.
if not force_refresh and _models_dev_cache:
_mark_stale_cache_grace()
_start_background_refresh_models_dev()
logger.debug(
"Using stale in-memory models.dev cache; refreshing in background"
)
return _models_dev_cache
# Stage 3: disk cache short-circuits the network call.
# Only kicks in on cold-start processes (in-mem cache is empty) and only
# when the user hasn't asked for a forced refresh. A stale disk cache is
# deliberately usable: provider/model resolution should not hang just
# because models.dev is unreachable.
if not force_refresh:
disk_age = _disk_cache_age_seconds()
if disk_age is not None and disk_age < _MODELS_DEV_CACHE_TTL:
if disk_age is not None:
disk_data = _load_disk_cache()
if disk_data:
_models_dev_cache = disk_data
# Anchor in-mem TTL to the disk file's age so we don't
# extend an already-aging cache by another full hour.
_models_dev_cache_time = time.time() - disk_age
logger.debug(
"Loaded models.dev from fresh disk cache "
"(%d providers, age=%.0fs)", len(disk_data), disk_age,
)
if disk_age < _MODELS_DEV_CACHE_TTL:
# Anchor in-mem TTL to the disk file's age so we don't
# extend an already-aging cache by another full hour.
_models_dev_cache_time = time.time() - disk_age
logger.debug(
"Loaded models.dev from fresh disk cache "
"(%d providers, age=%.0fs)", len(disk_data), disk_age,
)
else:
_mark_stale_cache_grace()
_start_background_refresh_models_dev()
logger.debug(
"Using stale models.dev disk cache (age=%.0fs); "
"refreshing in background",
disk_age,
)
return _models_dev_cache
# Stage 3: network fetch.
try:
response = requests.get(MODELS_DEV_URL, timeout=15)
response.raise_for_status()
data = response.json()
if isinstance(data, dict) and data:
_models_dev_cache = data
_models_dev_cache_time = time.time()
_save_disk_cache(data)
logger.debug(
"Fetched models.dev registry: %d providers, %d total models",
len(data),
sum(len(p.get("models", {})) for p in data.values() if isinstance(p, dict)),
)
# Failed automatic refreshes are process-wide. Avoid making every caller
# retry the same unreachable endpoint while no usable cache exists.
if not force_refresh and time.time() < _models_dev_retry_after:
return _models_dev_cache
# Stage 4: singleflight foreground network fetch — only reached when no
# memory or disk cache exists (or on force_refresh). Recheck state after
# acquiring the lock because another caller may have refreshed or
# established backoff while we waited.
with _models_dev_fetch_lock:
now = time.time()
if not force_refresh:
if _models_dev_cache:
return _models_dev_cache
if now < _models_dev_retry_after:
return _models_dev_cache
try:
data = _fetch_models_dev_from_network()
_commit_registry(data, where="foreground")
return data
except Exception as e:
logger.debug("Failed to fetch models.dev: %s", e)
except Exception as e:
_note_refresh_failure(e, where="foreground")
# Stage 4: network failed — fall back to whatever disk cache exists,
# even if it's stale. Give it a short 5 min in-mem TTL so we retry
# the network soon instead of serving stale data for a full hour.
if not _models_dev_cache:
_models_dev_cache = _load_disk_cache()
if _models_dev_cache:
_models_dev_cache_time = time.time() - _MODELS_DEV_CACHE_TTL + 300
logger.debug("Loaded models.dev from disk cache (%d providers)", len(_models_dev_cache))
# Stage 5: network failed — return any stale memory/disk cache. Cache
# freshness remains expired; the retry-after timestamp controls when
# the next automatic request is allowed.
if not _models_dev_cache:
_models_dev_cache = _load_disk_cache()
_models_dev_cache_time = 0
if _models_dev_cache:
logger.debug(
"Loaded stale models.dev disk cache (%d providers)",
len(_models_dev_cache),
)
return _models_dev_cache
return _models_dev_cache
def lookup_models_dev_context(provider: str, model: str) -> Optional[int]:
@@ -671,7 +840,9 @@ def _parse_provider_info(provider_id: str, raw: Dict[str, Any]) -> ProviderInfo:
# Provider-level queries
# ---------------------------------------------------------------------------
def get_provider_info(provider_id: str) -> Optional[ProviderInfo]:
def get_provider_info(
provider_id: str, *, allow_network: bool = True
) -> Optional[ProviderInfo]:
"""Get full provider metadata from models.dev.
Accepts either a Hermes provider ID (e.g. "kilocode") or a models.dev
@@ -680,7 +851,14 @@ def get_provider_info(provider_id: str) -> Optional[ProviderInfo]:
# Resolve Hermes ID → models.dev ID
mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id)
data = fetch_models_dev()
# NOTE: keep the zero-argument call on the default path. Dozens of test
# sites monkeypatch fetch_models_dev with zero-arg lambdas; passing the
# kwarg unconditionally would break them all (they raise TypeError).
data = (
fetch_models_dev()
if allow_network
else fetch_models_dev(allow_network=False)
)
raw = data.get(mdev_id)
if not isinstance(raw, dict):
return None
+29
View File
@@ -0,0 +1,29 @@
"""Hermes gateway monitoring.
Service health monitoring plus redacted operational diagnostics for the
gateway daemon, exported over OTLP to an operator-configured endpoint.
``emitter`` is the in-process event bus: producers (gateway status hooks,
the diagnostic log handler) hand typed events to a fire-and-forget queue,
and subscribers (the OTLP streamers) consume them off the hot path. The
emitter never blocks or raises into gateway code (the hot-path invariant),
and nothing is persisted locally — monitoring is an egress path, not a store.
Deliberately out of scope here: run/model/tool trajectory capture, usage
analytics, and any content-bearing signal. Those planes are served by the
NeMo Relay integration and its Hermes-owned subscribers.
"""
from __future__ import annotations
from . import emitter, events
emit = emitter.emit
get_emitter = emitter.get_emitter
__all__ = [
"emitter",
"events",
"emit",
"get_emitter",
]
+201
View File
@@ -0,0 +1,201 @@
"""Content-free cron service-health and execution telemetry projection."""
from __future__ import annotations
import hashlib
import logging
import re
from dataclasses import dataclass
from datetime import datetime
from typing import Any, Optional
from agent.monitoring.events import CronExecutionEvent
from agent.monitoring.gateway_health import GatewayHealthSnapshot, GatewayMetric
from cron.jobs import (
_compute_grace_seconds,
get_catch_up_occurrence_count,
get_ticker_heartbeat_age,
get_ticker_success_age,
load_jobs,
)
from cron.scheduler import get_running_job_ids
from hermes_time import now as _hermes_now
logger = logging.getLogger(__name__)
_KNOWN_STATUSES = {"claimed", "running", "completed", "failed", "unknown"}
_KNOWN_SOURCES = {"builtin", "direct", "external"}
_KNOWN_DELIVERY_OUTCOMES = {"delivered", "failed", "suppressed", "not_configured"}
@dataclass(frozen=True, slots=True)
class CronHealthSnapshot:
metrics: list[GatewayMetric]
events: list[CronExecutionEvent]
def _now() -> datetime:
return _hermes_now()
def _job_key(raw: Any) -> str:
value = str(raw or "unknown").encode("utf-8", errors="replace")
return f"sha256:{hashlib.sha256(value).hexdigest()[:24]}"
def classify_cron_error(raw: Any) -> str:
text = str(raw or "").lower()
if (
re.search(r"\b(?:authentication|authenticated|authenticate|authorization|authorized|authorize|unauthorized|forbidden)\b", text)
or re.search(r"\bbearer\b", text)
or re.search(r"\b(?:access|api|refresh) token\b", text)
or re.search(r"\b(?:401|403)\b", text)
):
return "auth_failed"
if "rate limit" in text or "429" in text or "quota" in text:
return "rate_limited"
if "timeout" in text or "timed out" in text:
return "timeout"
if any(value in text for value in ("network", "connection", "dns", "socket", "unreachable")):
return "network_error"
if "dispatch" in text or "executor" in text:
return "dispatch_failed"
if "interrupt" in text or "owner exited" in text or "restarted" in text:
return "interrupted"
if "empty response" in text:
return "empty_response"
if any(value in text for value in ("config", "missing", "invalid")):
return "invalid_config"
return "unknown"
def _parse_time(raw: Any) -> Optional[datetime]:
try:
return datetime.fromisoformat(str(raw)) if raw else None
except (TypeError, ValueError):
return None
def _duration_ms(record: dict[str, Any]) -> Optional[int]:
start = _parse_time(record.get("started_at")) or _parse_time(record.get("claimed_at"))
finish = _parse_time(record.get("finished_at"))
if start is None or finish is None:
return None
try:
duration = int((finish - start).total_seconds() * 1000)
except (TypeError, ValueError):
return None
return max(0, duration)
def project_execution_event(
record: dict[str, Any], *, delivery_outcome: Optional[str] = None
) -> CronExecutionEvent:
status = str(record.get("status") or "unknown").lower()
source = str(record.get("source") or "unknown").lower()
if source not in _KNOWN_SOURCES and source != "unknown":
source = "external"
outcome = str(delivery_outcome).lower() if delivery_outcome is not None else None
return CronExecutionEvent(
status=status if status in _KNOWN_STATUSES else "unknown",
job_key=_job_key(record.get("job_id")),
source=source if source in _KNOWN_SOURCES else "unknown",
duration_ms=_duration_ms(record),
delivery_outcome=(
outcome if outcome in _KNOWN_DELIVERY_OUTCOMES else None
),
error_class=(
classify_cron_error(record.get("error"))
if status in {"failed", "unknown"}
else None
),
)
def emit_execution_state(
record: Optional[dict[str, Any]], *, delivery_outcome: Optional[str] = None
) -> None:
"""Best-effort lifecycle emit; terminal states synchronously cross the queue barrier."""
if not record:
return
try:
from agent.monitoring import emitter
event = project_execution_event(record, delivery_outcome=delivery_outcome)
target = emitter.get_emitter()
target.emit(event)
if event.status in {"completed", "failed", "unknown"}:
target.flush(timeout=1.0)
except Exception:
logger.debug("cron execution telemetry emit failed", exc_info=True)
def _is_overdue(job: dict[str, Any], now: datetime) -> bool:
if not job.get("enabled", True):
return False
next_run = _parse_time(job.get("next_run_at"))
schedule = job.get("schedule")
if next_run is None or not isinstance(schedule, dict):
return False
try:
if next_run.tzinfo is None and now.tzinfo is not None:
next_run = next_run.replace(tzinfo=now.tzinfo)
lateness = (now - next_run).total_seconds()
return lateness > _compute_grace_seconds(schedule)
except (TypeError, ValueError):
return False
def build_cron_health_snapshot() -> CronHealthSnapshot:
metrics: list[GatewayMetric] = []
for name, reader in (
("hermes.cron.scheduler.heartbeat_age_seconds", get_ticker_heartbeat_age),
("hermes.cron.scheduler.last_success_age_seconds", get_ticker_success_age),
):
try:
value = reader()
if value is not None:
metrics.append(GatewayMetric(name, max(0.0, float(value)), {}))
except Exception:
logger.debug("cron freshness metric unavailable", exc_info=True)
try:
metrics.append(
GatewayMetric(
"hermes.cron.scheduler.catch_up_occurrences",
get_catch_up_occurrence_count(),
{},
)
)
except Exception:
logger.debug("cron catch-up metric unavailable", exc_info=True)
try:
jobs = load_jobs()
enabled = [job for job in jobs if job.get("enabled", True)]
metrics.append(GatewayMetric("hermes.cron.jobs.enabled", len(enabled), {}))
metrics.append(
GatewayMetric(
"hermes.cron.jobs.overdue",
sum(1 for job in enabled if _is_overdue(job, _now())),
{},
)
)
except Exception:
logger.debug("cron job metrics unavailable", exc_info=True)
try:
metrics.append(
GatewayMetric("hermes.cron.jobs.running", len(get_running_job_ids()), {})
)
except Exception:
logger.debug("cron running-job metric unavailable", exc_info=True)
return CronHealthSnapshot(metrics=metrics, events=[])
__all__ = [
"CronHealthSnapshot",
"build_cron_health_snapshot",
"classify_cron_error",
"emit_execution_state",
"project_execution_event",
]
+211
View File
@@ -0,0 +1,211 @@
"""Monitoring emitter: fire-and-forget queue + background dispatcher.
The emitter is the single seam between producers (gateway status hooks, the
diagnostic log handler) and consumers (the OTLP streamers). Its contract is
the hot-path invariant:
``emit()`` MUST return in O(microseconds), MUST NOT block on disk/network,
and MUST NEVER raise into the caller. A monitoring failure is logged
locally and dropped — it can never affect the gateway or a session.
Mechanism:
* ``emit(event)`` does a non-blocking ``queue.put_nowait`` wrapped in a bare
except. On a full queue it drops the *oldest* event and counts the drop.
* A daemon thread drains the queue and fans each batch out to subscribers
(the OTLP metric/span/log streamers). Each subscriber is fail-isolated —
a slow or raising subscriber never affects the hot path or its peers.
Nothing is persisted here. Monitoring is an egress path, not a local store;
if no subscriber is attached, events simply age out of the ring buffer.
"""
from __future__ import annotations
import logging
import queue
import threading
import time
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
_MAX_QUEUE = 10_000 # ring-buffer depth; oldest dropped when full
_DRAIN_BATCH = 256
class MonitoringEmitter:
"""Owns the queue, the dispatcher thread, and the subscriber list."""
def __init__(self, *, enabled: bool = True) -> None:
self._enabled = enabled
self._q: "queue.Queue[Dict[str, Any]]" = queue.Queue(maxsize=_MAX_QUEUE)
self._dropped = 0
self._dispatched = 0
self._stop = threading.Event()
self._started = False
self._lock = threading.Lock()
self._thread: Optional[threading.Thread] = None
# Live subscribers (the OTLP streamers). Called from the dispatcher
# thread, fully fail-isolated. Each subscriber is callable(batch: list[dict]).
self._subscribers: list = []
# ── public API (hot path) ───────────────────────────────────────────────
def emit(self, event: Any) -> None:
"""Enqueue an event. Never blocks, never raises.
``event`` may be a dataclass with ``to_dict()`` or a plain dict.
"""
if not self._enabled:
return
try:
payload = event.to_dict() if hasattr(event, "to_dict") else dict(event)
payload.setdefault("ts_ns", time.time_ns())
self._ensure_started()
try:
self._q.put_nowait(payload)
except queue.Full:
# Drop oldest to make room — bounded memory, newest-wins.
try:
self._q.get_nowait()
self._q.task_done()
self._dropped += 1
self._q.put_nowait(payload)
except Exception:
self._dropped += 1
except Exception: # the hot-path invariant: never propagate
logger.debug("monitoring emit failed", exc_info=True)
# ── lifecycle ───────────────────────────────────────────────────────────
def _ensure_started(self) -> None:
if self._started:
return
with self._lock:
if self._started:
return
self._thread = threading.Thread(
target=self._run, name="hermes-monitoring-dispatch", daemon=True
)
self._thread.start()
self._started = True
def _run(self) -> None:
while not self._stop.is_set():
try:
first = self._q.get(timeout=0.5)
except queue.Empty:
continue
batch = [first]
while len(batch) < _DRAIN_BATCH:
try:
batch.append(self._q.get_nowait())
except queue.Empty:
break
try:
self._dispatch(batch)
finally:
for _ in batch:
self._q.task_done()
def _dispatch(self, batch) -> None:
# Fan-out to subscribers (OTLP streamers) — fully fail-isolated.
for sub in list(self._subscribers):
try:
sub(batch)
except Exception:
logger.debug("monitoring subscriber failed", exc_info=True)
self._dispatched += len(batch)
def subscribe(self, callback) -> None:
"""Register a live batch subscriber (callable(batch: list[dict]))."""
if callback not in self._subscribers:
self._subscribers.append(callback)
self._enabled = True
def unsubscribe(self, callback) -> None:
try:
self._subscribers.remove(callback)
except ValueError:
pass
if not self._subscribers:
self._enabled = False
# ── introspection / shutdown (tests, CLI) ───────────────────────────────
def flush(self, timeout: float = 2.0) -> None:
"""Wait boundedly for queued and in-flight batches to finish dispatch."""
if timeout <= 0:
return
finished = threading.Event()
def _wait_for_completion() -> None:
self._q.join()
finished.set()
waiter = threading.Thread(
target=_wait_for_completion,
name="hermes-monitoring-flush",
daemon=True,
)
waiter.start()
finished.wait(timeout=timeout)
def stats(self) -> Dict[str, int]:
return {
"queued": self._q.qsize(),
"dispatched": self._dispatched,
"dropped": self._dropped,
"subscribers": len(self._subscribers),
}
def close(self) -> None:
self._stop.set()
if self._thread is not None:
self._thread.join(timeout=2.0)
self._started = False
# ── process-wide singleton ──────────────────────────────────────────────────
_EMITTER: Optional[MonitoringEmitter] = None
_EMITTER_LOCK = threading.Lock()
def get_emitter() -> MonitoringEmitter:
"""Return the process-wide monitoring emitter."""
global _EMITTER
if _EMITTER is not None:
return _EMITTER
with _EMITTER_LOCK:
if _EMITTER is None:
# Collection is opt-in. A plane exporter enables the singleton by
# attaching its first subscriber; until then producers are no-ops.
_EMITTER = MonitoringEmitter(enabled=False)
return _EMITTER
def emit(event: Any) -> None:
"""Module-level convenience: emit via the singleton."""
get_emitter().emit(event)
def reset_emitter_for_tests(emitter: Optional[MonitoringEmitter] = None) -> None:
"""Swap the singleton (tests only)."""
global _EMITTER
with _EMITTER_LOCK:
if _EMITTER is not None and emitter is not _EMITTER:
try:
_EMITTER.close()
except Exception:
pass
_EMITTER = emitter
# Back-compat alias for the salvaged class name used in emozilla's tests.
TelemetryEmitter = MonitoringEmitter
__all__ = [
"MonitoringEmitter",
"TelemetryEmitter",
"get_emitter",
"emit",
"reset_emitter_for_tests",
]
+86
View File
@@ -0,0 +1,86 @@
"""Typed gateway monitoring events.
Content-free service-health and redacted diagnostic events for the gateway
daemon. These are the only event shapes the monitoring plane emits: no
prompts, messages, tool args/results, session history, or usage analytics.
"""
from __future__ import annotations
import time
from dataclasses import dataclass, field, asdict
from typing import Any, Dict, Optional
def _now_ns() -> int:
return time.time_ns()
@dataclass(slots=True)
class GatewayHealthEvent:
"""Content-free gateway health snapshot or lifecycle event."""
name: str
gateway_state: Optional[str] = None
old_state: Optional[str] = None
new_state: Optional[str] = None
exit_reason: Optional[str] = None
restart_requested: Optional[bool] = None
active_agents: int = 0
gateway_busy: bool = False
gateway_drainable: bool = False
platform_count: int = 0
fatal_platform_count: int = 0
profile: Optional[str] = None
install_id: Optional[str] = None
version: Optional[str] = None
supervision_mode: Optional[str] = None
pid: Optional[int] = None
ts_ns: int = field(default_factory=_now_ns)
def to_dict(self) -> Dict[str, Any]:
return {"event": "gateway_health", **asdict(self)}
@dataclass(slots=True)
class GatewayDiagnosticEvent:
"""Redacted gateway diagnostic event for operator-owned observability."""
name: str
subsystem: str
error_class: str = "unknown"
error_code: Optional[str] = None
platform: Optional[str] = None
old_state: Optional[str] = None
new_state: Optional[str] = None
profile: Optional[str] = None
version: Optional[str] = None
severity: str = "warning"
ts_ns: int = field(default_factory=_now_ns)
source_logger: Optional[str] = None
def to_dict(self) -> Dict[str, Any]:
return {"event": "gateway_diagnostic", **asdict(self)}
@dataclass(slots=True)
class CronExecutionEvent:
"""Content-free durable cron execution lifecycle projection."""
status: str
job_key: str
source: str = "unknown"
duration_ms: Optional[int] = None
delivery_outcome: Optional[str] = None
error_class: Optional[str] = None
ts_ns: int = field(default_factory=_now_ns)
def to_dict(self) -> Dict[str, Any]:
return {"event": "cron_execution", **asdict(self)}
__all__ = [
"GatewayHealthEvent",
"GatewayDiagnosticEvent",
"CronExecutionEvent",
]
+469
View File
@@ -0,0 +1,469 @@
"""Gateway health and diagnostics signal producer.
This module keeps the plane narrow: service health monitoring plus
redacted operational diagnostics. It reuses the existing gateway runtime-status
contract and emits content-free metrics/events. No prompts, messages, tool args,
session history, audit records, or product analytics belong here.
"""
from __future__ import annotations
import hashlib
import logging
import re
from dataclasses import dataclass
from typing import Any, Dict, List, Optional
from agent.monitoring.events import GatewayDiagnosticEvent, GatewayHealthEvent
@dataclass(frozen=True, slots=True)
class GatewayMetric:
name: str
value: int | float
attributes: Dict[str, str]
@dataclass(frozen=True, slots=True)
class GatewayHealthSnapshot:
metrics: List[GatewayMetric]
events: List[GatewayHealthEvent | GatewayDiagnosticEvent]
_RUNNING_PLATFORM_STATES = {"running", "connected", "ok", "ready"}
_FATAL_PLATFORM_STATES = {"fatal", "degraded", "error", "failed"}
_KNOWN_GATEWAY_STATES = {
"starting", "draining", "stopping", "stopped", "startup_failed", "unknown"
} | _RUNNING_PLATFORM_STATES | _FATAL_PLATFORM_STATES
_KNOWN_PLATFORM_STATES = _RUNNING_PLATFORM_STATES | _FATAL_PLATFORM_STATES | {
"connecting", "disconnected", "disabled", "paused", "retrying", "unknown"
}
_SUPERVISION_MODES = {"systemd", "s6", "container", "launchd", "manual", "unknown"}
_SOURCE_LOGGER_RE = re.compile(r"^gateway(?:\.[A-Za-z_][A-Za-z0-9_]*)*$")
def _allowed_logger(name: str) -> bool:
return name == "gateway" or name.startswith("gateway.")
def source_logger_for_export(name: Any) -> Optional[str]:
"""Return a bounded source-controlled gateway logger name for OTLP scope."""
value = str(name or "")
return value if len(value) <= 128 and _SOURCE_LOGGER_RE.fullmatch(value) else None
def redact_gateway_message(message: Any) -> str:
"""Redact gateway diagnostic free text for operator-owned export.
Single scrub path: everything goes through
``agent.monitoring.redaction.redact_for_export`` (unconditional
secrets + PII), then is length-bounded.
"""
try:
from agent.monitoring.redaction import redact_for_export
redacted = redact_for_export(str(message or "")) or ""
except Exception:
redacted = "[redaction-unavailable]"
return redacted[:500]
def classify_gateway_error(raw: Any) -> str:
s = str(raw or "").lower()
if any(k in s for k in ("auth", "token", "unauthorized", "forbidden", "401", "403")):
return "auth_failed"
if "rate" in s and "limit" in s:
return "rate_limited"
if "timeout" in s or "timed out" in s:
return "timeout"
if any(
k in s
for k in (
"network",
"connection",
"dns",
"socket",
"connect call failed",
"failed to connect",
"cannot connect",
"unreachable",
"name resolution",
)
):
return "network_error"
if any(k in s for k in ("config", "missing", "invalid")):
return "invalid_config"
if "startup" in s:
return "startup_failed"
if "fatal" in s:
return "platform_fatal"
return "unknown"
def classify_exit_reason(
raw: Any, *, state: Any, restart_requested: bool
) -> Optional[str]:
"""Reduce free-form shutdown text to a bounded operational class."""
if restart_requested:
return "restart_requested"
state_name = str(state or "").lower()
if raw is None and state_name != "startup_failed":
return None
classified = classify_gateway_error(raw)
if state_name == "startup_failed":
return classified if classified != "unknown" else "startup_failed"
text = str(raw or "").lower()
if "signal" in text or "sigterm" in text or "sigint" in text:
return "signal"
if state_name == "stopped" and any(word in text for word in ("shutdown", "stop")):
return "planned_stop"
return classified
def _bounded_state(raw: Any, *, allowed: set[str]) -> str:
state = str(raw or "unknown").lower()
return state if state in allowed else "unknown"
def _safe_metric_value(raw: Any, *, limit: int = 128) -> str:
try:
from agent.monitoring.redaction import redact_for_export
value = redact_for_export(str(raw or "")) or "unknown"
except Exception:
return "unknown"
return value[:limit]
def _safe_instance_id(raw: Any) -> str:
"""Return a stable opaque instance key without exporting the source ID."""
value = str(raw or "unknown").encode("utf-8", errors="replace")
return f"sha256:{hashlib.sha256(value).hexdigest()[:24]}"
def subsystem_for_logger(logger_name: str) -> str:
if logger_name == "gateway.relay" or logger_name.startswith("gateway.relay."):
return "platform.relay"
if logger_name.startswith("gateway.platforms."):
parts = logger_name.split(".")
if len(parts) >= 3 and parts[2]:
return f"platform.{parts[2]}"
if logger_name.startswith("gateway.platforms"):
return "platform"
if logger_name.startswith("gateway"):
return "gateway"
return "gateway"
def platform_for_subsystem(subsystem: str) -> Optional[str]:
if subsystem.startswith("platform."):
return subsystem.split(".", 1)[1] or None
return None
def _parse_active_agents(raw: Any) -> int:
try:
from gateway.status import parse_active_agents
return parse_active_agents(raw)
except Exception:
try:
return max(0, int(raw))
except (TypeError, ValueError):
return 0
def _derive_busy(gateway_running: bool, gateway_state: Any, active_agents: Any) -> bool:
try:
from gateway.status import derive_gateway_busy
return derive_gateway_busy(
gateway_running=gateway_running,
gateway_state=gateway_state,
active_agents=active_agents,
)
except Exception:
return bool(gateway_running and gateway_state == "running" and _parse_active_agents(active_agents) > 0)
def _derive_drainable(gateway_running: bool, gateway_state: Any) -> bool:
try:
from gateway.status import derive_gateway_drainable
return derive_gateway_drainable(gateway_running=gateway_running, gateway_state=gateway_state)
except Exception:
return bool(gateway_running and gateway_state == "running")
def _base_attrs(*, profile: str, install_id: str, version: str, supervision_mode: str) -> Dict[str, str]:
mode = str(supervision_mode or "unknown").lower()
return {
"service.instance.id": _safe_instance_id(install_id),
"service.version": _safe_metric_value(version, limit=64),
"hermes.supervision_mode": mode if mode in _SUPERVISION_MODES else "unknown",
}
def _metric(name: str, value: int | float, attrs: Dict[str, str], **extra: str) -> GatewayMetric:
out = dict(attrs)
for key, val in extra.items():
if val is not None:
out[key] = _safe_metric_value(val)
return GatewayMetric(name=name, value=value, attributes=out)
def build_gateway_health_snapshot(
runtime: Optional[dict[str, Any]],
*,
gateway_running: bool,
profile: str,
install_id: str,
version: str,
supervision_mode: str = "unknown",
) -> GatewayHealthSnapshot:
"""Convert gateway_state.json-compatible runtime state into P0 signals."""
runtime = runtime or {}
gateway_state = _bounded_state(
runtime.get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
)
active_agents = _parse_active_agents(runtime.get("active_agents", 0))
busy = _derive_busy(gateway_running, gateway_state, active_agents)
drainable = _derive_drainable(gateway_running, gateway_state)
platforms = runtime.get("platforms") if isinstance(runtime.get("platforms"), dict) else {}
base = _base_attrs(profile=profile, install_id=install_id, version=version, supervision_mode=supervision_mode)
metrics: list[GatewayMetric] = [
_metric("hermes.gateway.up", 1 if gateway_running else 0, base),
_metric("hermes.gateway.active_agents", active_agents, base),
_metric("hermes.gateway.busy", 1 if busy else 0, base),
_metric("hermes.gateway.drainable", 1 if drainable else 0, base),
_metric("hermes.gateway.restart_requested", 1 if runtime.get("restart_requested") else 0, base),
]
if gateway_state:
metrics.append(_metric("hermes.gateway.state", 1, base, **{"hermes.gateway.state": str(gateway_state)}))
fatal_count = 0
events: list[GatewayHealthEvent | GatewayDiagnosticEvent] = []
for platform, pdata in platforms.items():
pdata = pdata if isinstance(pdata, dict) else {}
state = _bounded_state(
pdata.get("state"), allowed=_KNOWN_PLATFORM_STATES
)
raw_error = pdata.get("error_code") or pdata.get("error_message")
error_code = classify_gateway_error(raw_error)
is_up = state in _RUNNING_PLATFORM_STATES
is_degraded = state in _FATAL_PLATFORM_STATES
if is_degraded:
fatal_count += 1
metrics.append(_metric(
"hermes.platform.up",
1 if is_up else 0,
base,
**{"hermes.platform": str(platform), "hermes.platform.state": state},
))
metrics.append(_metric(
"hermes.platform.degraded",
1 if is_degraded else 0,
base,
**{"hermes.platform": str(platform), "hermes.platform.state": state, "hermes.error_code": error_code},
))
if is_degraded:
events.append(GatewayDiagnosticEvent(
name="platform.fatal",
subsystem=f"platform.{platform}",
platform=str(platform),
error_code=error_code,
error_class=classify_gateway_error(error_code or pdata.get("error_message")),
profile=profile,
version=version,
severity="error" if state == "fatal" else "warning",
))
events.insert(0, GatewayHealthEvent(
name="gateway.health_snapshot",
gateway_state=str(gateway_state) if gateway_state is not None else None,
active_agents=active_agents,
gateway_busy=busy,
gateway_drainable=drainable,
platform_count=len(platforms),
fatal_platform_count=fatal_count,
profile=profile,
install_id=install_id,
version=version,
supervision_mode=supervision_mode,
pid=_coerce_pid(runtime.get("pid")),
))
return GatewayHealthSnapshot(metrics=metrics, events=events)
def _safe_profile() -> str:
try:
from hermes_cli.profiles import get_active_profile_name
return str(get_active_profile_name() or "default")
except Exception:
return "default"
def _safe_version() -> str:
try:
from hermes_cli import __version__
return str(__version__)
except Exception:
return "unknown"
def emit_runtime_status_transition(previous: Optional[dict[str, Any]], current: dict[str, Any]) -> None:
"""Emit immediate content-free gateway events for runtime status changes.
Called by gateway.status.write_runtime_status after persisting the new status.
Fully fail-open: failures never affect gateway status writes.
"""
try:
from agent.monitoring import emitter
out: list[GatewayHealthEvent | GatewayDiagnosticEvent] = []
profile = _safe_profile()
version = _safe_version()
old_gateway_state = _bounded_state(
(previous or {}).get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
) if (previous or {}).get("gateway_state") is not None else None
new_gateway_state = _bounded_state(
current.get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
) if current.get("gateway_state") is not None else None
if old_gateway_state != new_gateway_state and new_gateway_state:
out.append(GatewayHealthEvent(
name="gateway.lifecycle",
gateway_state=new_gateway_state,
old_state=old_gateway_state,
new_state=new_gateway_state,
exit_reason=classify_exit_reason(
current.get("exit_reason"),
state=new_gateway_state,
restart_requested=bool(current.get("restart_requested")),
),
restart_requested=bool(current.get("restart_requested")),
active_agents=_parse_active_agents(current.get("active_agents", 0)),
profile=profile,
version=version,
pid=_coerce_pid(current.get("pid")),
))
if new_gateway_state == "startup_failed":
out.append(GatewayDiagnosticEvent(
name="gateway.startup_failed",
subsystem="gateway",
error_class=classify_gateway_error(current.get("exit_reason") or "startup_failed"),
error_code=classify_gateway_error(current.get("exit_reason") or "startup_failed"),
profile=profile,
version=version,
severity="error",
))
if new_gateway_state == "stopped":
out.append(GatewayHealthEvent(
name="gateway.exit",
gateway_state=new_gateway_state,
old_state=old_gateway_state,
new_state=new_gateway_state,
exit_reason=classify_exit_reason(
current.get("exit_reason"),
state=new_gateway_state,
restart_requested=bool(current.get("restart_requested")),
),
restart_requested=bool(current.get("restart_requested")),
active_agents=_parse_active_agents(current.get("active_agents", 0)),
profile=profile,
version=version,
pid=_coerce_pid(current.get("pid")),
))
old_platforms_raw = (previous or {}).get("platforms")
new_platforms_raw = current.get("platforms")
old_platforms = old_platforms_raw if isinstance(old_platforms_raw, dict) else {}
new_platforms = new_platforms_raw if isinstance(new_platforms_raw, dict) else {}
for platform, pdata in new_platforms.items():
pdata = pdata if isinstance(pdata, dict) else {}
prev_raw = old_platforms.get(platform, {})
prev = prev_raw if isinstance(prev_raw, dict) else {}
old_state = _bounded_state(
prev.get("state"), allowed=_KNOWN_PLATFORM_STATES
) if prev.get("state") is not None else None
new_state = _bounded_state(
pdata.get("state"), allowed=_KNOWN_PLATFORM_STATES
) if pdata.get("state") is not None else None
if old_state == new_state or not new_state:
continue
error_code = classify_gateway_error(pdata.get("error_code") or pdata.get("error_message"))
severity = "error" if new_state.lower() in {"fatal", "failed", "error"} else "warning"
out.append(GatewayDiagnosticEvent(
name="platform.state_change",
subsystem=f"platform.{platform}",
platform=str(platform),
old_state=old_state,
new_state=new_state,
error_code=error_code,
error_class=error_code,
profile=profile,
version=version,
severity=severity,
))
if new_state.lower() in _FATAL_PLATFORM_STATES:
out.append(GatewayDiagnosticEvent(
name="platform.fatal",
subsystem=f"platform.{platform}",
platform=str(platform),
error_code=error_code,
error_class=error_code,
profile=profile,
version=version,
severity=severity,
))
for ev in out:
emitter.emit(ev)
except Exception:
logging.getLogger(__name__).debug("gateway runtime status transition emit failed", exc_info=True)
def _coerce_pid(raw: Any) -> Optional[int]:
try:
pid = int(raw)
except (TypeError, ValueError):
return None
return pid if pid > 0 else None
class GatewayDiagnosticLogHandler(logging.Handler):
"""Allowlisted warning/error bridge for gateway-owned diagnostics."""
def __init__(self, *, profile: str = "default", version: str = "unknown") -> None:
super().__init__(level=logging.WARNING)
self.profile = profile
self.version = version
def emit(self, record: logging.LogRecord) -> None:
try:
if record.levelno < logging.WARNING:
return
if not _allowed_logger(record.name):
return
subsystem = subsystem_for_logger(record.name)
message = record.getMessage()
error_class = classify_gateway_error(message)
event = GatewayDiagnosticEvent(
name=f"gateway.log.{record.levelname.lower()}",
subsystem=subsystem,
source_logger=source_logger_for_export(record.name),
platform=platform_for_subsystem(subsystem),
error_class=error_class,
error_code=error_class,
profile=self.profile,
version=self.version,
severity=record.levelname.lower(),
)
from agent.monitoring import emitter
emitter.get_emitter().emit(event)
except Exception:
logging.getLogger(__name__).debug("gateway diagnostic emit failed", exc_info=True)
__all__ = [
"GatewayMetric",
"GatewayHealthSnapshot",
"GatewayDiagnosticLogHandler",
"build_gateway_health_snapshot",
"classify_gateway_error",
"source_logger_for_export",
"redact_gateway_message",
]
+643
View File
@@ -0,0 +1,643 @@
"""Gateway Health & Diagnostics OTLP export runtime.
This exporter emits operator-owned gateway service-health metrics plus
narrow redacted diagnostic events. It is deliberately in-process and fail-open so
it works under systemd, launchd, s6, containers, tmux, nohup, or a simple shell
without a sidecar/watchdog dependency.
"""
from __future__ import annotations
import logging
import os
import re
import threading
from dataclasses import dataclass
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
_DEFAULT_DIAGNOSTIC_SCOPE = "hermes.gateway.diagnostics"
_RESOURCE_ATTRIBUTE_KEYS = frozenset({
"service.name",
"service.namespace",
"service.version",
"service.instance.id",
"deployment.environment.name",
"cloud.provider",
"cloud.platform",
"cloud.region",
"telemetry.scope",
})
_DIAGNOSTIC_ATTRIBUTE_KEYS = frozenset({
"name",
"subsystem",
"error_class",
"error_code",
"platform",
"old_state",
"new_state",
"version",
"severity",
})
_SAFE_RESOURCE_VALUE = re.compile(r"^[A-Za-z0-9._:/-]{1,128}$")
def _redact_string(raw: Any, *, limit: int = 500) -> str:
try:
from agent.monitoring.redaction import redact_for_export
return (redact_for_export(str(raw or "")) or "[redacted]")[:limit]
except Exception:
return "[redaction-unavailable]"
def _safe_resource_attributes(raw: Any) -> Dict[str, str]:
"""Allowlist bounded resource labels and reject values changed by redaction."""
attrs: Dict[str, str] = {}
if not isinstance(raw, dict):
return attrs
for key, value in raw.items():
key = str(key)
if key not in _RESOURCE_ATTRIBUTE_KEYS or value is None:
continue
if key == "service.instance.id":
from agent.monitoring.gateway_health import _safe_instance_id
attrs[key] = _safe_instance_id(value)
continue
text = str(value)
if not _SAFE_RESOURCE_VALUE.fullmatch(text):
continue
if _redact_string(text, limit=128) != text:
continue
attrs[key] = text
return attrs
def _runtime_resource_attributes(
config: Dict[str, Any], *, telemetry_scope: str
) -> Dict[str, str]:
"""Build the safe OTLP resource shared by metrics and diagnostic logs."""
gh = _gateway_health_config(config)
attrs = _safe_resource_attributes(gh.get("resource_attributes"))
from agent.monitoring.gateway_health import _safe_instance_id
attrs["service.name"] = "hermes-gateway"
attrs["service.instance.id"] = _safe_instance_id(_install_id(config))
attrs["telemetry.scope"] = telemetry_scope
return attrs
def _diagnostic_log_attributes(event: Dict[str, Any]) -> Dict[str, Any]:
attrs: Dict[str, Any] = {}
for key in _DIAGNOSTIC_ATTRIBUTE_KEYS:
value = event.get(key)
if value is None:
continue
attrs[f"hermes.{key}"] = _redact_string(value) if isinstance(value, str) else value
return attrs
@dataclass(slots=True)
class GatewayHealthExportRuntime:
enabled: bool
reason: str = "disabled"
streamer: Any = None
metric_provider: Any = None
log_handler: Any = None
log_streamer: Any = None
thread: Optional[threading.Thread] = None
stop_event: Optional[threading.Event] = None
def shutdown(self) -> None:
if self.stop_event is not None:
self.stop_event.set()
if self.thread is not None:
self.thread.join(timeout=0.25)
if self.log_handler is not None:
try:
logging.getLogger().removeHandler(self.log_handler)
except Exception:
pass
# All producers above are now stopped. Drain queued and in-flight
# events before detaching subscribers so the terminal lifecycle event
# cannot race exporter shutdown. The barrier is bounded and fail-open.
try:
from agent.monitoring.emitter import get_emitter
emitter = get_emitter()
emitter.flush(timeout=1.0)
if self.streamer is not None:
emitter.unsubscribe(self.streamer)
if self.log_streamer is not None:
emitter.unsubscribe(self.log_streamer)
except Exception:
pass
# Network flush/close runs under one bounded daemon-thread deadline and
# can never delay gateway teardown indefinitely.
closeables = [
item for item in (self.streamer, self.log_streamer, self.metric_provider)
if item is not None
]
def _close() -> None:
for item in closeables:
try:
item.shutdown()
except Exception:
pass
if closeables:
worker = threading.Thread(
target=_close,
name="hermes-gateway-health-export-shutdown",
daemon=True,
)
worker.start()
worker.join(timeout=2.0)
self.streamer = None
self.log_streamer = None
self.metric_provider = None
self.thread = None
self.stop_event = None
def _gateway_health_config(config: Dict[str, Any]) -> Dict[str, Any]:
mon = (config or {}).get("monitoring") or {}
return mon.get("gateway_health_export") or {}
def _otlp_config(config: Dict[str, Any]) -> Dict[str, Any]:
mon = (config or {}).get("monitoring") or {}
export = mon.get("export") or {}
return export.get("otlp") or {}
def _enabled(config: Dict[str, Any]) -> bool:
gh = _gateway_health_config(config)
otlp = _otlp_config(config)
return bool(gh.get("enabled") and otlp.get("enabled") and otlp.get("endpoint"))
def _require_metrics_sdk(*, auto_install: bool = True, prompt: bool = False) -> Dict[str, Any]:
if auto_install:
try:
from tools.lazy_deps import ensure as _lazy_ensure
_lazy_ensure("export.otlp", prompt=prompt)
except Exception:
pass
try:
from opentelemetry.exporter.otlp.proto.http._log_exporter import OTLPLogExporter
from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter
from opentelemetry.metrics import Observation
from opentelemetry.trace import INVALID_SPAN_ID, INVALID_TRACE_ID, TraceFlags
from opentelemetry._logs import LogRecord
from opentelemetry._logs.severity import SeverityNumber
from opentelemetry.sdk._logs import LoggerProvider
from opentelemetry.sdk._logs.export import BatchLogRecordProcessor
from opentelemetry.sdk.metrics import MeterProvider
from opentelemetry.sdk.metrics.export import PeriodicExportingMetricReader
from opentelemetry.sdk.resources import Resource
return {
"OTLPLogExporter": OTLPLogExporter,
"OTLPMetricExporter": OTLPMetricExporter,
"Observation": Observation,
"LogRecord": LogRecord,
"LoggerProvider": LoggerProvider,
"INVALID_SPAN_ID": INVALID_SPAN_ID,
"INVALID_TRACE_ID": INVALID_TRACE_ID,
"TraceFlags": TraceFlags,
"SeverityNumber": SeverityNumber,
"BatchLogRecordProcessor": BatchLogRecordProcessor,
"MeterProvider": MeterProvider,
"PeriodicExportingMetricReader": PeriodicExportingMetricReader,
"Resource": Resource,
}
except Exception as exc:
raise RuntimeError(f"OTLP metrics SDK unavailable: {exc}") from exc
def _resolve_headers(headers_env: Optional[Dict[str, str]]) -> Dict[str, str]:
resolved: Dict[str, str] = {}
for header_name, env_name in (headers_env or {}).items():
val = os.environ.get(str(env_name))
if val:
resolved[str(header_name)] = val
return resolved
def _metric_endpoint(endpoint: str) -> str:
if endpoint.endswith("/v1/traces"):
return endpoint[: -len("/v1/traces")] + "/v1/metrics"
return endpoint
def _logs_endpoint(endpoint: str) -> str:
if endpoint.endswith("/v1/traces"):
return endpoint[: -len("/v1/traces")] + "/v1/logs"
if endpoint.endswith("/v1/metrics"):
return endpoint[: -len("/v1/metrics")] + "/v1/logs"
return endpoint
def _version() -> str:
try:
from hermes_cli import __version__
return str(__version__)
except Exception:
return "unknown"
def _profile() -> str:
try:
from hermes_cli.profiles import get_active_profile_name
return str(get_active_profile_name() or "default")
except Exception:
return "default"
def _install_id(config: Dict[str, Any]) -> str:
try:
from agent.monitoring.policy import ensure_install_id
return str(ensure_install_id(config))
except Exception:
return "unknown"
def _supervision_mode() -> str:
if os.environ.get("INVOCATION_ID"):
return "systemd"
if os.environ.get("S6_CMD_ARG0") or os.environ.get("S6_VERSION"):
return "s6"
if os.environ.get("container") or os.path.exists("/.dockerenv"):
return "container"
if os.environ.get("LAUNCHD_SOCKET"):
return "launchd"
return "manual"
def _read_gateway_snapshot(config: Dict[str, Any]):
from agent.monitoring.gateway_health import build_gateway_health_snapshot
try:
from gateway.status import read_runtime_status
runtime = read_runtime_status() or {}
except Exception:
runtime = {}
return build_gateway_health_snapshot(
runtime,
gateway_running=True,
profile=_profile(),
install_id=_install_id(config),
version=_version(),
supervision_mode=_supervision_mode(),
)
def _read_cron_snapshot():
from agent.monitoring.cron_health import build_cron_health_snapshot
return build_cron_health_snapshot()
def _read_background_work_count() -> int:
"""Count live background/subagent work that ``active_agents`` does NOT include.
``hermes.gateway.active_agents`` counts foreground turns + in-flight cron
jobs + API runs, but deliberately excludes backgrounded ``delegate_task``
subagents, ``terminal(background=true)`` processes, kanban workers, and the
runner's own background tasks (they are tracked only for the scale-to-zero
suspend guard, ``_scale_to_zero_has_live_background_work``). Without this
metric a peer churning through delegated subagents shows ``active_agents=0``
on the fleet dashboard. Best-effort and content-free: a single integer,
no job/task identity. Returns 0 if a source can't be imported.
Delegation is counted TASK-granular (``active_task_count``): a fan-out batch
of N subagents contributes N, not 1, so the metric reflects real concurrent
subagent load rather than dispatch-unit/pool-slot count. This intentionally
differs from the async pool's capacity accounting (one batch = one slot).
"""
total = 0
try:
from tools.async_delegation import active_task_count
total += max(0, int(active_task_count()))
except Exception:
logger.debug("background-work async-delegation count failed", exc_info=True)
try:
from tools.process_registry import process_registry
total += max(0, int(process_registry.count_running()))
except Exception:
logger.debug("background-work process-registry count failed", exc_info=True)
return total
def _read_background_delegations_count() -> int:
"""Count live async delegation UNITS (dispatch/pool slots).
Complements ``_read_background_work_count`` (which is task-granular): this
counts each ``delegate_task`` dispatch as ONE regardless of fan-out width,
matching the async pool's capacity accounting (a batch = one slot). Together
the two metrics let an operator see both slot pressure
(``background_delegations``, alert vs ``max_concurrent_children``) and real
concurrent subagent load (``background_work``). Delegations only — it does
not include ``terminal(background)`` / kanban work, which are already folded
into ``background_work``. Best-effort; 0 if the source can't be imported.
"""
try:
from tools.async_delegation import active_count
return max(0, int(active_count()))
except Exception:
logger.debug("background-delegations count failed", exc_info=True)
return 0
def _read_runtime_snapshot(config: Dict[str, Any]):
gateway_snapshot = _read_gateway_snapshot(config)
# Background/subagent work — a distinct metric from active_agents (which
# never counts it). Appended to the gateway snapshot so it rides the same
# base resource attributes (service.instance.id etc.).
try:
from agent.monitoring.gateway_health import GatewayMetric
base = dict(gateway_snapshot.metrics[0].attributes) if gateway_snapshot.metrics else {}
gateway_snapshot.metrics.append(
GatewayMetric(
name="hermes.gateway.background_work",
value=_read_background_work_count(),
attributes=base,
)
)
gateway_snapshot.metrics.append(
GatewayMetric(
name="hermes.gateway.background_delegations",
value=_read_background_delegations_count(),
attributes=base,
)
)
except Exception as exc:
logger.warning(
"background-work snapshot unavailable; metric not exported (error_type=%s)",
type(exc).__name__,
)
logger.debug("background-work snapshot traceback", exc_info=True)
try:
cron_snapshot = _read_cron_snapshot()
except Exception as exc:
# Content-free visibility: cron telemetry silently dropping out is a
# release-relevant regression, so surface it at WARNING with only the
# exception *type* name (never the message, which could carry paths or
# other environment detail). exc_info stays on the DEBUG record.
logger.warning(
"cron health snapshot unavailable; cron telemetry not exported (error_type=%s)",
type(exc).__name__,
)
logger.debug("cron health snapshot traceback", exc_info=True)
return gateway_snapshot
gateway_snapshot.metrics.extend(cron_snapshot.metrics)
return gateway_snapshot
def _emit_snapshot_events(config: Dict[str, Any]) -> None:
gh = _gateway_health_config(config)
if not gh.get("diagnostic_events_enabled", True):
return
try:
from agent.monitoring import emitter
snapshot = _read_runtime_snapshot(config)
for event in snapshot.events:
emitter.emit(event)
except Exception:
logger.debug("gateway health snapshot emit failed", exc_info=True)
def _start_metric_provider(config: Dict[str, Any], sdk: Dict[str, Any]) -> Any:
gh = _gateway_health_config(config)
if not gh.get("metrics_enabled", True):
return None
otlp = _otlp_config(config)
endpoint = _metric_endpoint(str(otlp.get("endpoint")))
headers = _resolve_headers(otlp.get("headers_env"))
exporter = sdk["OTLPMetricExporter"](endpoint=endpoint, headers=headers or None)
interval_ms = max(5, int(gh.get("export_interval_seconds", 60))) * 1000
reader = sdk["PeriodicExportingMetricReader"](exporter, export_interval_millis=interval_ms)
resource_attrs = _runtime_resource_attributes(
config, telemetry_scope="gateway_health"
)
provider = sdk["MeterProvider"](
metric_readers=[reader],
resource=sdk["Resource"].create(resource_attrs),
)
meter = provider.get_meter("hermes.gateway.health")
Observation = sdk["Observation"]
metric_names = [
"hermes.gateway.up",
"hermes.gateway.state",
"hermes.gateway.active_agents",
"hermes.gateway.busy",
"hermes.gateway.drainable",
"hermes.gateway.restart_requested",
"hermes.gateway.background_work",
"hermes.gateway.background_delegations",
"hermes.platform.up",
"hermes.platform.degraded",
"hermes.cron.scheduler.heartbeat_age_seconds",
"hermes.cron.scheduler.last_success_age_seconds",
"hermes.cron.scheduler.catch_up_occurrences",
"hermes.cron.jobs.enabled",
"hermes.cron.jobs.running",
"hermes.cron.jobs.overdue",
]
def callback(name: str):
def _cb(_options=None):
try:
snapshot = _read_runtime_snapshot(config)
return [Observation(m.value, m.attributes) for m in snapshot.metrics if m.name == name]
except Exception:
logger.debug("gateway metric callback failed", exc_info=True)
return []
return _cb
for metric_name in metric_names:
meter.create_observable_gauge(metric_name, callbacks=[callback(metric_name)])
return provider
def _severity_number(sdk: Dict[str, Any], severity: Any) -> Any:
SeverityNumber = sdk["SeverityNumber"]
sev = str(severity or "warning").lower()
if sev in {"critical", "fatal"}:
return SeverityNumber.FATAL
if sev == "error":
return SeverityNumber.ERROR
if sev in {"info", "information"}:
return SeverityNumber.INFO
if sev == "debug":
return SeverityNumber.DEBUG
return SeverityNumber.WARN
class GatewayDiagnosticLogStreamer:
"""Emitter subscriber that sends gateway diagnostic events as OTLP logs."""
def __init__(self, config: Dict[str, Any], sdk: Dict[str, Any]):
otlp = _otlp_config(config)
headers = _resolve_headers(otlp.get("headers_env"))
endpoint = _logs_endpoint(str(otlp.get("endpoint")))
resource_attrs = _runtime_resource_attributes(
config, telemetry_scope="gateway_diagnostics"
)
self._provider = sdk["LoggerProvider"](resource=sdk["Resource"].create(resource_attrs))
self._processor = sdk["BatchLogRecordProcessor"](
sdk["OTLPLogExporter"](endpoint=endpoint, headers=headers or None)
)
self._provider.add_log_record_processor(self._processor)
self._logger = self._provider.get_logger(_DEFAULT_DIAGNOSTIC_SCOPE)
self._LogRecord = sdk["LogRecord"]
self._sdk = sdk
self.exported = 0
def __call__(self, batch: list[Dict[str, Any]]) -> None:
from agent.monitoring.gateway_health import source_logger_for_export
for ev in batch:
if ev.get("event") != "gateway_diagnostic":
continue
attrs = _diagnostic_log_attributes(ev)
# Preserve the source-controlled Python logger as the OTel
# instrumentation scope. This adds precise code attribution without
# turning a fluid module layout into a maintained subsystem enum.
# Rendered messages stay out because they may contain arbitrary IDs,
# names, paths, or configured strings. A future, separately gated
# ``diagnostic_detail: redacted_message`` mode may add best-effort
# free text when an observability plane defines that privacy policy.
source_logger = source_logger_for_export(ev.get("source_logger"))
otel_logger = (
self._provider.get_logger(source_logger)
if source_logger is not None
else self._logger
)
body = "gateway diagnostic"
record = self._LogRecord(
timestamp=ev.get("ts_ns"),
trace_id=self._sdk["INVALID_TRACE_ID"],
span_id=self._sdk["INVALID_SPAN_ID"],
trace_flags=self._sdk["TraceFlags"].DEFAULT,
severity_text=str(ev.get("severity") or "warning").upper(),
severity_number=_severity_number(self._sdk, ev.get("severity")),
body=_redact_string(body),
attributes=attrs,
)
otel_logger.emit(record)
self.exported += 1
def shutdown(self) -> None:
try:
from agent.monitoring.emitter import get_emitter
get_emitter().unsubscribe(self)
except Exception:
pass
try:
self._processor.force_flush()
self._provider.shutdown()
except Exception:
pass
def _start_diagnostic_log_streamer(config: Dict[str, Any], sdk: Dict[str, Any]) -> GatewayDiagnosticLogStreamer:
from agent.monitoring.emitter import get_emitter
streamer = GatewayDiagnosticLogStreamer(config, sdk)
get_emitter().subscribe(streamer)
return streamer
def _start_snapshot_thread(config: Dict[str, Any], stop_event: threading.Event) -> threading.Thread:
interval = max(5, int(_gateway_health_config(config).get("logs_export_interval_seconds", 5)))
def _run() -> None:
while not stop_event.wait(interval):
_emit_snapshot_events(config)
thread = threading.Thread(target=_run, name="hermes-gateway-health-export", daemon=True)
thread.start()
return thread
def _attach_log_handler(config: Dict[str, Any]) -> Any:
gh = _gateway_health_config(config)
if not gh.get("diagnostic_events_enabled", True) or not gh.get("warning_error_events_enabled", True):
return None
from agent.monitoring.gateway_health import GatewayDiagnosticLogHandler
handler = GatewayDiagnosticLogHandler(profile=_profile(), version=_version())
root = logging.getLogger()
if handler not in root.handlers:
root.addHandler(handler)
return handler
def _gateway_health_event(ev: Dict[str, Any]) -> bool:
return ev.get("event") in {"gateway_health", "cron_execution"}
def start_gateway_health_export(config: Dict[str, Any]) -> GatewayHealthExportRuntime:
"""Start P0 gateway health export if configured. Never raises."""
if not _enabled(config):
return GatewayHealthExportRuntime(enabled=False, reason="disabled")
gh = _gateway_health_config(config)
runtime = GatewayHealthExportRuntime(enabled=True, reason="enabled")
sdk: Optional[Dict[str, Any]] = None
if gh.get("metrics_enabled", True) or gh.get("diagnostic_events_enabled", True):
try:
sdk = _require_metrics_sdk(prompt=False)
except Exception:
logger.warning(
"monitoring.gateway_health_export.enabled but OTLP SDK is unavailable; "
"install 'hermes-agent[otlp]'",
exc_info=True,
)
return GatewayHealthExportRuntime(enabled=False, reason="otlp_unavailable")
if gh.get("metrics_enabled", True) and sdk is not None:
try:
runtime.metric_provider = _start_metric_provider(config, sdk)
except Exception:
logger.warning("gateway health OTLP metrics failed to start", exc_info=True)
runtime.shutdown()
return GatewayHealthExportRuntime(enabled=False, reason="metrics_start_failed")
if gh.get("diagnostic_events_enabled", True) and sdk is not None:
try:
from agent.monitoring import otlp_exporter
runtime.streamer = otlp_exporter.start_streaming(config, event_filter=_gateway_health_event)
if runtime.streamer is None:
raise RuntimeError("gateway health span streamer did not start")
runtime.log_streamer = _start_diagnostic_log_streamer(config, sdk)
except Exception:
logger.debug("gateway diagnostic OTLP export failed to start", exc_info=True)
runtime.shutdown()
return GatewayHealthExportRuntime(enabled=False, reason="diagnostics_start_failed")
try:
runtime.log_handler = _attach_log_handler(config)
except Exception:
logger.debug("gateway diagnostic log handler failed to attach", exc_info=True)
if gh.get("diagnostic_events_enabled", True):
try:
_emit_snapshot_events(config)
runtime.stop_event = threading.Event()
runtime.thread = _start_snapshot_thread(config, runtime.stop_event)
except Exception:
logger.debug("gateway health snapshot thread failed to start", exc_info=True)
return runtime
__all__ = [
"GatewayHealthExportRuntime",
"start_gateway_health_export",
]
+272
View File
@@ -0,0 +1,272 @@
"""Export monitoring events to an OpenTelemetry Collector over OTLP/HTTP.
Maps gateway monitoring events to OTel spans and sends them to the endpoint
configured under ``monitoring.export.otlp``. Lets an operator stream Hermes
gateway health into their own observability stack (OTEL Collector, DataDog,
and similar).
Notes:
* The destination is operator-configured; this module only sends to that
endpoint. No default destination ships.
* ``opentelemetry-sdk`` + ``opentelemetry-exporter-otlp-proto-http`` are an
optional extra (``pip install hermes-agent[otlp]``), imported lazily so the
dependency is only required when OTLP export is actually used.
* ``headers_env`` maps a header name to an environment variable name; values
are read from the environment at export time and never logged or stored.
* The continuous subscriber runs in the emitter's dispatcher thread and is
fail-isolated, so an export error cannot affect the gateway.
Only monitoring events (gateway_health / gateway_diagnostic) exist on this
plane; the ``event_filter`` seam is kept so future planes sharing the emitter
cannot silently ride along on this exporter.
"""
from __future__ import annotations
import logging
import os
from typing import Any, Callable, Dict, List, Optional
logger = logging.getLogger(__name__)
class OTLPUnavailable(RuntimeError):
"""Raised when the optional OpenTelemetry SDK isn't installed."""
def _require_sdk(*, auto_install: bool = True, prompt: bool = True):
"""Import the OTel SDK, lazily installing it on first use if needed.
Routes through tools.lazy_deps (feature 'export.otlp') so a missing SDK
triggers the standard venv install flow — same as every other optional
backend — gated by security.allow_lazy_installs and TTY-prompted. Falls back
to OTLPUnavailable (with a manual install hint) when the SDK can't be made
importable (lazy installs disabled, install failed, or auto_install=False).
``auto_install``: attempt the lazy install when missing (default True).
``prompt``: ask before installing when interactive (default True); pass
False from non-interactive contexts like the continuous streamer.
"""
if auto_install:
try:
from tools.lazy_deps import ensure as _lazy_ensure
_lazy_ensure("export.otlp", prompt=prompt)
except ImportError:
pass # lazy_deps unavailable — fall through to the import attempt
except Exception:
# FeatureUnavailable (lazy installs disabled / declined / failed) —
# fall through; the import below raises OTLPUnavailable with the hint.
pass
try:
from opentelemetry.sdk.trace import TracerProvider
from opentelemetry.sdk.trace.export import BatchSpanProcessor
from opentelemetry.sdk.resources import Resource
from opentelemetry.exporter.otlp.proto.http.trace_exporter import (
OTLPSpanExporter,
)
from opentelemetry.trace import SpanKind
return {
"TracerProvider": TracerProvider,
"BatchSpanProcessor": BatchSpanProcessor,
"Resource": Resource,
"OTLPSpanExporter": OTLPSpanExporter,
"SpanKind": SpanKind,
}
except Exception as e: # ImportError or partial install
raise OTLPUnavailable(
"OTLP export requires the optional dependency. Install with:\n"
" pip install 'hermes-agent[otlp]'\n"
f"(import error: {e})"
)
def _resolve_headers(headers_env: Optional[Dict[str, str]]) -> Dict[str, str]:
"""Resolve {header_name: ENV_VAR_NAME} -> {header_name: value} from env.
The config stores environment variable names, not secret values; values are
read from the environment here. Missing variables are skipped (and noted at
debug level without the value).
"""
resolved: Dict[str, str] = {}
for header_name, env_name in (headers_env or {}).items():
val = os.environ.get(str(env_name))
if val:
resolved[str(header_name)] = val
else:
logger.debug("OTLP header %s: env var %s not set; skipping",
header_name, env_name)
return resolved
def _otlp_config(config: Dict[str, Any]) -> Dict[str, Any]:
mon = (config or {}).get("monitoring") or {}
export = mon.get("export") or {}
return export.get("otlp") or {}
def build_exporter(config: Dict[str, Any]):
"""Construct an OTLP span exporter from config. Raises OTLPUnavailable if no SDK."""
sdk = _require_sdk()
otlp = _otlp_config(config)
endpoint = otlp.get("endpoint")
if not endpoint:
raise ValueError("monitoring.export.otlp.endpoint is not set")
headers = _resolve_headers(otlp.get("headers_env"))
return sdk["OTLPSpanExporter"](endpoint=endpoint, headers=headers or None)
def _resource_attributes(config: Dict[str, Any]) -> Dict[str, str]:
from agent.monitoring.gateway_health import _safe_instance_id
from agent.monitoring.policy import ensure_install_id
return {
"service.name": "hermes-gateway",
"service.instance.id": _safe_instance_id(ensure_install_id(config)),
"telemetry.scope": "gateway_monitoring",
}
def _make_provider(config: Dict[str, Any]):
sdk = _require_sdk()
resource = sdk["Resource"].create(_resource_attributes(config))
provider = sdk["TracerProvider"](resource=resource)
processor = sdk["BatchSpanProcessor"](build_exporter(config))
provider.add_span_processor(processor)
return provider, processor
# ── event -> span attribute mapping ──────────────────────────────────────────
def _span_attrs(ev: Dict[str, Any]) -> Dict[str, Any]:
"""Span attributes for a monitoring event (content-free by construction)."""
kind = ev.get("event")
attrs: Dict[str, Any] = {"hermes.event": kind or "unknown"}
keep_by_kind = {
"gateway_health": ("name", "gateway_state", "old_state", "new_state",
"exit_reason", "restart_requested", "active_agents",
"gateway_busy", "gateway_drainable", "platform_count",
"fatal_platform_count", "version",
"supervision_mode", "pid"),
"gateway_diagnostic": ("name", "subsystem", "error_class", "error_code",
"platform", "old_state", "new_state",
"version", "severity"),
"cron_execution": ("status", "job_key", "source", "duration_ms",
"delivery_outcome", "error_class"),
}
for col in keep_by_kind.get(kind, ()): # type: ignore[arg-type]
v = ev.get(col)
if v is not None:
if isinstance(v, str):
try:
from agent.monitoring.redaction import redact_for_export
v = (redact_for_export(v) or "[redacted]")[:500]
except Exception:
v = "[redaction-unavailable]"
attrs[f"hermes.{col}"] = v
return attrs
def export_batch(provider, batch: List[Dict[str, Any]]) -> int:
"""Map a batch of events to OTel spans. Returns spans created."""
tracer = provider.get_tracer("hermes.monitoring")
n = 0
for ev in batch:
try:
name = f"hermes.{ev.get('event', 'event')}"
span = tracer.start_span(name, attributes=_span_attrs(ev))
span.end()
n += 1
except Exception:
logger.debug("OTLP span map failed", exc_info=True)
return n
# ── continuous streaming subscriber ─────────────────────────────────────────
class OTLPStreamer:
"""A live subscriber that pushes each emitter batch to OTLP as it lands.
Register with ``emitter.subscribe(streamer)``. Fail-isolated by the emitter.
"""
def __init__(
self,
config: Dict[str, Any],
*,
event_filter: Optional[Callable[[Dict[str, Any]], bool]] = None,
):
self._provider, self._processor = _make_provider(config)
self._event_filter = event_filter
self.exported = 0
def __call__(self, batch: List[Dict[str, Any]]) -> None:
if self._event_filter is not None:
batch = [ev for ev in batch if self._event_filter(ev)]
if not batch:
return
self.exported += export_batch(self._provider, batch)
def shutdown(self) -> None:
try:
from agent.monitoring.emitter import get_emitter
get_emitter().unsubscribe(self)
except Exception:
pass
try:
self._processor.force_flush()
self._provider.shutdown()
except Exception:
pass
def is_available() -> bool:
"""True when the OTel SDK is already importable. Does NOT auto-install —
this is a pure check (e.g. for status display)."""
try:
_require_sdk(auto_install=False)
return True
except OTLPUnavailable:
return False
def is_enabled(config: Dict[str, Any]) -> bool:
otlp = _otlp_config(config)
return bool(otlp.get("enabled") and otlp.get("endpoint"))
def start_streaming(
config: Dict[str, Any],
*,
event_filter: Optional[Callable[[Dict[str, Any]], bool]] = None,
) -> Optional[OTLPStreamer]:
"""If OTLP is enabled, attach a streamer to the singleton emitter.
``event_filter`` scopes the exporter to its plane, e.g. gateway-health
export, so enabling one plane cannot silently export unrelated events.
Non-interactive context (startup): attempts a lazy install with prompt=False
so a configured-but-missing SDK is installed once (gated by
security.allow_lazy_installs), then streams. If it still can't load, logs and
no-ops — never blocks or raises into startup.
"""
if not is_enabled(config):
return None
try:
_require_sdk(prompt=False)
except OTLPUnavailable:
logger.warning("monitoring.export.otlp.enabled but the OTel SDK could not "
"be installed/imported; install 'hermes-agent[otlp]'")
return None
from agent.monitoring.emitter import get_emitter
streamer = OTLPStreamer(config, event_filter=event_filter)
get_emitter().subscribe(streamer)
return streamer
__all__ = [
"OTLPUnavailable",
"OTLPStreamer",
"build_exporter",
"export_batch",
"is_available",
"is_enabled",
"start_streaming",
]
+57
View File
@@ -0,0 +1,57 @@
"""Install identity for gateway monitoring.
The install id is a stable, resettable pseudonymous identifier attached to
exported health signals so an operator can tell instances apart in their
collector. It carries no account identity and can be rotated by clearing
``monitoring.install_id`` in config.
"""
from __future__ import annotations
import logging
import uuid
from typing import Any, Dict
logger = logging.getLogger(__name__)
def ensure_install_id(config: Dict[str, Any]) -> str:
"""Return a stable install id, minting and persisting one when empty.
The id must survive gateway restarts (it becomes ``service.instance.id``
on exported signals), so a freshly minted UUID is written back to
config.yaml immediately. The write is fail-open: if persisting fails
(read-only home, managed scope), the ephemeral id is still returned and
a new one is minted next start.
Clearing ``monitoring.install_id`` (e.g. ``hermes config set
monitoring.install_id ""``) rotates the id on the next gateway start.
"""
mon = config.get("monitoring") if isinstance(config, dict) else None
existing = (mon or {}).get("install_id") if isinstance(mon, dict) else None
if isinstance(existing, str) and existing.strip():
return existing
minted = str(uuid.uuid4())
try:
from hermes_cli.config import load_config, save_config
fresh = load_config()
if isinstance(fresh, dict):
slot = fresh.setdefault("monitoring", {})
if isinstance(slot, dict) and not str(slot.get("install_id") or "").strip():
slot["install_id"] = minted
save_config(fresh)
except Exception:
logger.debug("install_id persist failed; using ephemeral id", exc_info=True)
# Keep the in-memory config consistent for this process either way.
if isinstance(config, dict):
config.setdefault("monitoring", {})
if isinstance(config["monitoring"], dict):
config["monitoring"]["install_id"] = minted
return minted
__all__ = [
"ensure_install_id",
]
+71
View File
@@ -0,0 +1,71 @@
"""Redaction applied to monitoring data before egress.
One unconditional scrub, no modes, no knobs. Every string that leaves the
process passes through ``redact_for_export``:
* Secrets first — wraps ``agent/redact.py::redact_sensitive_text(force=True)``
plus bearer/token-shape patterns, and fails CLOSED: if the redactor cannot
run, the raw string is never emitted.
* PII second — e-mail addresses, phone numbers, and UUID-shaped identifiers
are rewritten to ``[email]`` / ``[phone]`` / ``[id]``.
There is deliberately no setting to weaken this. The monitoring plane is
content-free by design: rendered log messages are not exported, and bounded
structured strings are still scrubbed as defense-in-depth. This redactor also
remains available for a future, explicitly gated redacted-message detail mode.
"""
from __future__ import annotations
import re
from typing import Optional
# ── secret shapes (belt-and-suspenders on top of agent/redact.py) ───────────
_BEARER_RE = re.compile(r"\bBearer\s+[A-Za-z0-9._~+\-/]+=*", re.IGNORECASE)
_TOKEN_RE = re.compile(
r"\b(xox[baprs]-[A-Za-z0-9-]+|sk-[A-Za-z0-9_-]{8,}|gh[pousr]_[A-Za-z0-9_]{8,})\b"
)
_SECRET_LITERAL_RE = re.compile(r"\*{3,}")
_BEARER_RESIDUE_RE = re.compile(r"\bBearer\s+\[[^\]]+\]", re.IGNORECASE)
# ── PII shapes ───────────────────────────────────────────────────────────────
_EMAIL_RE = re.compile(r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}")
# E.164-ish and common separators; conservative to avoid nuking code/IDs.
_PHONE_RE = re.compile(
r"(?<!\w)(?:\+?\d{1,3}[\s.\-]?)?(?:\(\d{2,4}\)[\s.\-]?)?\d{3}[\s.\-]?\d{3,4}(?:[\s.\-]?\d{2,4})?(?!\w)"
)
# Long opaque hex/uuid-ish user identifiers.
_UUID_RE = re.compile(
r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b"
)
def _secret_redact(text: str) -> str:
"""Always-on secret redaction. force=True so user config can't disable it."""
try:
from agent.redact import redact_sensitive_text
out = redact_sensitive_text(text, force=True)
except Exception:
# Fail CLOSED: if the redactor can't run, do not emit the raw string.
return "[redaction-unavailable]"
out = _BEARER_RE.sub("[redacted]", out)
out = _TOKEN_RE.sub("[redacted]", out)
out = _SECRET_LITERAL_RE.sub("[redacted]", out)
out = _BEARER_RESIDUE_RE.sub("[redacted]", out)
return out
def redact_for_export(text: Optional[str]) -> Optional[str]:
"""Scrub a string for egress: secrets, then PII. Unconditional."""
if text is None:
return None
out = _secret_redact(str(text))
out = _EMAIL_RE.sub("[email]", out)
out = _UUID_RE.sub("[id]", out)
out = _PHONE_RE.sub("[phone]", out)
return out
__all__ = [
"redact_for_export",
]
+2 -2
View File
@@ -210,8 +210,8 @@ def _resolve_trust_policy(plugin_id: str) -> _TrustPolicy:
return _TrustPolicy(plugin_id="")
try:
from hermes_cli.config import load_config
config = load_config() or {}
from hermes_cli.config import load_config_readonly
config = load_config_readonly() or {}
except Exception: # pragma: no cover — config IO failure
return _TrustPolicy(plugin_id=plugin_id)
+122 -25
View File
@@ -19,13 +19,18 @@ from typing import Optional
from agent.runtime_cwd import resolve_agent_cwd
from agent.skill_utils import (
EXCLUDED_SKILL_DIRS,
ORG_ACTIVE_MARKER,
ORG_MIRROR_DIR_NAME,
ORG_PROVENANCE_FILE,
SKILL_SUPPORT_DIRS,
extract_skill_conditions,
extract_skill_description,
get_all_skills_dirs,
get_disabled_skill_names,
iter_skill_index_files,
org_id_of_path,
parse_frontmatter,
read_active_org_id,
skill_matches_environment,
skill_matches_platform,
skill_matches_platform_list,
@@ -567,16 +572,18 @@ def computer_use_guidance(platform_name: Optional[str] = None) -> str:
"Background delivery is the DEFAULT and the co-work path, but it is "
"the first rung, not the only one. Read each action's structured "
"result and climb only when the driver tells you to:\n"
"- `effect: 'confirmed'` + `verified: true` — the driver read the "
"result back. Done.\n"
"- `effect: 'confirmed'` (or `verified: true`) — done, even if an "
"advisory escalation is also present. Never repeat successful input.\n"
"- `effect: 'unverifiable'` — the input was delivered but the driver "
"can't confirm it. Re-capture and check the screenshot/tree yourself "
"before deciding it worked.\n"
"- `effect: 'suspected_noop'`, `code: 'background_unavailable'`, or an "
"`escalation.recommended` field — the action did NOT land. Follow "
"`escalation.recommended`:\n"
"can't confirm it. Get fresh state and check it before any retry; an "
"escalation recommendation does not override this rule.\n"
"- `effect: 'suspected_noop'` or a structured refusal such as "
"`code: 'background_unavailable'` — escalation is allowed. Follow "
"the recommended rung when present:\n"
" - `'px'` → re-issue addressing the target by `coordinate=[x,y]` "
"read off the screenshot instead of `element`.\n"
" - `'page'` → use the exact-bound typed browser page rung below "
"before native foreground escalation. Do not start a legacy page workflow.\n"
" - `'foreground'` (or a pixel click still didn't land) → re-issue "
"the SAME action with `delivery_mode='foreground'`. This briefly "
"raises the window; it needs its own approval and is only appropriate "
@@ -586,6 +593,21 @@ def computer_use_guidance(platform_name: Optional[str] = None) -> str:
"as a prediction from the app being Electron/Chromium/GTK. Do not "
"silently retry the same rung expecting a different result, and do "
"not conclude 'cua-driver can't drive this app' — climb the ladder.\n\n"
"## Typed browser page rung\n"
"For `recommended='page'` or supported browser PAGE content, use the namespaced "
"`cua_browser_*` actions: bind with `cua_browser_state` using the exact "
"native `(pid, window_id)`, require `binding_quality='exact'` and "
"`mutation_allowed=true`, select its opaque `tab_id`, then take a "
"fresh semantic snapshot before using a current `ref`. After every "
"typed mutation, call `cua_browser_state` again before another action. "
"Input defaults to trusted; `input_route='dom_event'` is an explicit "
"downgrade, never an automatic retry. Use native capture/input for "
"browser chrome, OS permission prompts, native dialogs, and unsupported "
"targets. Browser setup is a separately approved action; attaching an "
"existing profile is enforced by cua-driver's immutable permission "
"mode: standard requires a certified protected host and fails closed "
"when Hermes has none; explicit Hermes YOLO uses a private unrestricted "
"daemon after the user's launch/session risk acceptance.\n\n"
"## Background mode rules\n"
"- Do NOT use `raise_window=true` on `focus_app` unless the user "
"explicitly asked you to bring a window to front. Input routing to "
@@ -959,7 +981,7 @@ WSL_ENVIRONMENT_HINT = (
# misleading — the agent should only see the machine it can actually touch.
_REMOTE_TERMINAL_BACKENDS = frozenset({
"docker", "singularity", "modal", "daytona", "ssh",
"managed_modal",
"vercel_sandbox", "managed_modal",
})
@@ -973,6 +995,7 @@ _BACKEND_FALLBACK_DESCRIPTIONS: dict[str, str] = {
"modal": "a Modal sandbox (Linux)",
"managed_modal": "a managed Modal sandbox (Linux)",
"daytona": "a Daytona workspace (Linux)",
"vercel_sandbox": "a Vercel sandbox (Linux)",
"ssh": "a remote host reached over SSH (likely Linux)",
}
@@ -1047,7 +1070,7 @@ def _probe_remote_backend(env_type: str) -> str | None:
}
container_config = None
if env_type in {"docker", "singularity", "modal", "daytona"}:
if env_type in {"docker", "singularity", "modal", "daytona", "vercel_sandbox"}:
container_config = {
"container_cpu": config.get("container_cpu", 1),
"container_memory": config.get("container_memory", 5120),
@@ -1137,7 +1160,7 @@ def build_environment_hints() -> str:
and a Windows-only note that `terminal` shells out to bash, not
PowerShell).
- For **remote / sandbox** terminal backends (docker, singularity,
modal, daytona, ssh): host info is **suppressed**
modal, daytona, ssh, vercel_sandbox): host info is **suppressed**
because the agent's tools can't touch the host — only the backend
matters. A live probe inside the backend reports its OS, user, $HOME,
and cwd. Falls back to a static summary if the probe fails.
@@ -1224,10 +1247,10 @@ def build_environment_hints() -> str:
extra = (os.getenv("HERMES_ENVIRONMENT_HINT") or "").strip()
if not extra:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
extra = str(
(load_config().get("agent", {}) or {}).get("environment_hint", "")
(load_config_readonly().get("agent", {}) or {}).get("environment_hint", "")
).strip()
except Exception as e:
logger.debug("Could not read agent.environment_hint from config: %s", e)
@@ -1278,9 +1301,9 @@ def _get_context_file_max_chars(context_length: Optional[int] = None) -> int:
3. ``CONTEXT_FILE_MAX_CHARS`` (20K) as the upstream-compatible fallback.
"""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
val = load_config().get("context_file_max_chars")
val = load_config_readonly().get("context_file_max_chars")
if isinstance(val, (int, float)) and val > 0:
return int(val)
except Exception as e:
@@ -1323,7 +1346,9 @@ def drain_truncation_warnings() -> list:
_SKILLS_PROMPT_CACHE_MAX = 8
_SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict()
_SKILLS_PROMPT_CACHE_LOCK = threading.Lock()
_SKILLS_SNAPSHOT_VERSION = 1
# v2: entries gained org provenance fields (org_id/org_author/rel_dir) for M2
# org-shared skills; older snapshots are discarded and rebuilt.
_SKILLS_SNAPSHOT_VERSION = 2
def _skills_prompt_snapshot_path() -> Path:
@@ -1342,13 +1367,32 @@ def clear_skills_system_prompt_cache(*, clear_snapshot: bool = False) -> None:
def _build_skills_manifest(skills_dir: Path) -> dict[str, list[int]]:
"""Build an mtime/size manifest of all SKILL.md and DESCRIPTION.md files."""
"""Build an mtime/size manifest of all SKILL.md and DESCRIPTION.md files.
Org mirrors (M2): only the ACTIVE org's mirror participates, and the
``.active_org`` marker itself is included — so switching/leaving an org
invalidates the snapshot even when no SKILL.md changed.
"""
manifest: dict[str, list[int]] = {}
skills_dir_str = str(skills_dir)
base = os.path.join(skills_dir_str, "")
prefix_len = len(base)
active_org = read_active_org_id(skills_dir)
org_root = os.path.join(skills_dir_str, ORG_MIRROR_DIR_NAME)
marker_path = os.path.join(org_root, ORG_ACTIVE_MARKER)
try:
st = os.stat(marker_path)
manifest[ORG_MIRROR_DIR_NAME + "/" + ORG_ACTIVE_MARKER] = [
int(st.st_mtime), int(st.st_size),
]
except OSError:
pass
for root, dirs, files in os.walk(skills_dir_str, followlinks=True):
has_skill_md = "SKILL.md" in files
if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None:
dirs.remove(ORG_MIRROR_DIR_NAME)
elif root == org_root:
dirs[:] = [d for d in dirs if d == active_org]
dirs[:] = [
d
for d in dirs
@@ -1413,6 +1457,15 @@ def _build_snapshot_entry(
"""Build a serialisable metadata dict for one skill."""
rel_path = skill_file.relative_to(skills_dir)
parts = rel_path.parts
# M2 org mirror: strip the `_org/<org_id>/` prefix so category/name derive
# from the path WITHIN the mirror (same shape the org tree was built
# from), and record provenance for labeling + fail-loud collisions.
org_id: str | None = None
if len(parts) >= 3 and parts[0] == ORG_MIRROR_DIR_NAME:
org_id = parts[1]
parts = parts[2:]
if len(parts) >= 2:
skill_name = parts[-2]
category = "/".join(parts[:-2]) if len(parts) > 2 else parts[0]
@@ -1424,7 +1477,7 @@ def _build_snapshot_entry(
if isinstance(platforms, str):
platforms = [platforms]
return {
entry = {
"skill_name": skill_name,
"category": category,
"frontmatter_name": str(frontmatter.get("name", skill_name)),
@@ -1432,6 +1485,22 @@ def _build_snapshot_entry(
"platforms": [str(p).strip() for p in platforms if str(p).strip()],
"conditions": extract_skill_conditions(frontmatter),
}
if org_id:
entry["org_id"] = org_id
# Author from the pull-time provenance sidecar (token-verified at
# push by the plane's author_mismatch guard). Best-effort.
try:
import json as _json
prov_path = (
skills_dir / ORG_MIRROR_DIR_NAME / org_id / ORG_PROVENANCE_FILE
)
prov = _json.loads(prov_path.read_text(encoding="utf-8"))
device = str(prov.get("author_device") or "")
entry["org_author"] = device or str(prov.get("author_user_id") or "")
except Exception:
entry["org_author"] = ""
return entry
# =========================================================================
@@ -1567,6 +1636,10 @@ def build_skills_system_prompt(
skills_by_category: dict[str, list[tuple[str, str]]] = {}
category_descriptions: dict[str, str] = {}
# Unified visible-entry list (both paths) so the org labeling +
# fail-loud collision pass below runs identically for snapshot and scan.
visible_entries: list[dict] = []
skill_entries: list[dict] = []
if snapshot is not None:
# Fast path: use pre-parsed metadata from disk
@@ -1574,7 +1647,6 @@ def build_skills_system_prompt(
if not isinstance(entry, dict):
continue
skill_name = entry.get("skill_name") or ""
category = entry.get("category") or "general"
frontmatter_name = entry.get("frontmatter_name") or skill_name
platforms = entry.get("platforms") or []
if not skill_matches_platform_list(platforms):
@@ -1587,16 +1659,13 @@ def build_skills_system_prompt(
available_toolsets,
):
continue
skills_by_category.setdefault(category, []).append(
(frontmatter_name, entry.get("description", ""))
)
visible_entries.append(entry)
category_descriptions = {
str(k): str(v)
for k, v in (snapshot.get("category_descriptions") or {}).items()
}
else:
# Cold path: full filesystem scan + write snapshot for next time
skill_entries: list[dict] = []
for skill_file in iter_skill_index_files(skills_dir, "SKILL.md"):
is_compatible, frontmatter, desc = _parse_skill_file(skill_file)
entry = _build_snapshot_entry(skill_file, skills_dir, frontmatter, desc)
@@ -1612,10 +1681,38 @@ def build_skills_system_prompt(
available_toolsets,
):
continue
skills_by_category.setdefault(entry["category"], []).append(
(entry["frontmatter_name"], entry["description"])
)
visible_entries.append(entry)
# ── M2 org labeling + FAIL-LOUD collisions ─────────────────────────
# An org skill lists with an explicit provenance tag. When a personal and
# an org skill share a name, NEITHER silently wins: both list qualified
# (personal keeps the bare name is the wrong default — silent divergence
# from the org set; org winning silently shadows the user's own work) —
# so both entries carry a [name collision] flag and skill_view refuses
# the ambiguous bare name (its existing multi-candidate guard).
name_owners: dict[str, set[str]] = {}
for entry in visible_entries:
fm = entry.get("frontmatter_name") or entry.get("skill_name") or ""
kind = "org" if entry.get("org_id") else "personal"
name_owners.setdefault(fm, set()).add(kind)
for entry in visible_entries:
fm = entry.get("frontmatter_name") or entry.get("skill_name") or ""
desc = entry.get("description", "")
org_id = entry.get("org_id")
collided = len(name_owners.get(fm, set())) > 1
if org_id:
author = entry.get("org_author") or ""
tag = f"[org-shared{': by ' + author if author else ''}]"
desc = f"{tag} {desc}".strip()
category = f"org:{org_id}"
else:
category = entry.get("category") or "general"
if collided:
desc = f"[name collision — also exists {'personally' if org_id else 'in your org'}; load via category path] {desc}".strip()
skills_by_category.setdefault(category, []).append((fm, desc))
if snapshot is None:
# (continuation of the cold path below: category descriptions + write)
# Read category-level DESCRIPTION.md files
for desc_file in iter_skill_index_files(skills_dir, "DESCRIPTION.md"):
try:
+24 -2
View File
@@ -140,6 +140,11 @@ _ENV_ASSIGN_RE = re.compile(
# The colon-form URL guard (skip when ``://`` present) lives at the call site.
_SECRET_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential|auth)"
_CFG_VALUE = r"(['\"]?)([^\s&]+?)\2(?=[\s&]|$)"
# Linear pre-gate for the _CFG_*_RE subs below: a text with no secret keyword
# can never match either pattern, so the (potentially backtrack-heavy) subs
# are skipped entirely for such text. See the call site in
# redact_sensitive_text().
_CFG_SECRET_WORD_RE = re.compile(_SECRET_CFG_NAMES, re.IGNORECASE)
# Programmatic env lookups (``os.getenv(...)``, ``os.environ[...]``,
# ``os.environ.get(...)``, ``process.env.X``, ``$ENV{X}``) reference variable
@@ -377,8 +382,17 @@ _STRICT_URL_PARAM_RE = re.compile(
# Match userinfo in both absolute (``scheme://user:pass@host``) and
# network-path (``//user:pass@host``) references. The authority boundary stops
# at path/query/fragment delimiters so an ``@`` elsewhere in a URL is ignored.
#
# Anchored on the mandatory ``//`` rather than an optional scheme prefix: the
# scheme sits outside the match either way (replacement callbacks re-emit
# group(1), so ``https:`` stays untouched in the surrounding text), and the
# old optional-scheme prefix ``(?:[A-Za-z][A-Za-z0-9+.-]*:)?`` backtracked
# catastrophically (O(n²)) on long unbroken alphanumeric runs — a 320KB
# synthetic compaction payload spent ~55s inside this pattern per sub() call.
# Output-equivalence to the old pattern was fuzz-verified (20k random strings
# plus targeted URL forms).
_STRICT_URL_USERINFO_RE = re.compile(
r"((?:[A-Za-z][A-Za-z0-9+.-]*:)?//)([^/\s?#@]+)@"
r"(//)([^/\s?#@]+)@"
)
# HTTP access logs often use a relative request target rather than a full URL:
@@ -706,7 +720,15 @@ def redact_sensitive_text(
# web-URL query params are intentionally passed through (see note
# near the bottom of this function); _DB_CONNSTR_RE still guards
# connection-string passwords.
if "://" not in text:
#
# Extra gate: every _CFG_*_RE match requires a secret keyword in
# the key, so a text without any secret keyword cannot match —
# skipping is exact. This matters because _CFG_DOTTED_RE
# backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs
# (e.g. base64/hex blobs in compaction payloads); the linear
# keyword scan prevents that pathological path on secret-free
# text.
if "://" not in text and _CFG_SECRET_WORD_RE.search(text):
text = _CFG_DOTTED_RE.sub(_redact_env, text)
text = _CFG_ANCHORED_RE.sub(_redact_env, text)
+55 -1
View File
@@ -8,7 +8,9 @@ rate-limited provider concurrently.
import random
import threading
import time
from typing import Any
from datetime import datetime, timezone
from email.utils import parsedate_to_datetime
from typing import Any, Optional
# Monotonic counter for jitter seed uniqueness within the same process.
# Protected by a lock to avoid race conditions in concurrent retry paths
@@ -33,6 +35,58 @@ _ZAI_CODING_OVERLOAD_LONG_BACKOFF = (30.0, 60.0, 90.0, 120.0)
_ZAI_CODING_OVERLOAD_SHORT_ATTEMPTS = 3
def parse_retry_after_seconds(value_or_headers: Any) -> Optional[float]:
"""Parse a ``Retry-After`` value into non-negative seconds.
Accepts either a raw header value (numeric string / HTTP-date / number)
or a headers mapping, in which case the ``Retry-After`` key is looked up
case-insensitively (``.get`` on dict-like objects tries both common
casings; real HTTP header containers like httpx/requests are already
case-insensitive).
Returns:
Seconds as a ``float`` (negative deltas clamped to ``0.0``), or
``None`` when the header is absent or unparseable.
"""
raw = value_or_headers
if raw is not None and not isinstance(raw, (str, int, float)):
# Looks like a headers mapping — pull the header out of it.
getter = getattr(raw, "get", None)
if callable(getter):
try:
value = getter("Retry-After")
if value is None:
value = getter("retry-after")
except Exception:
return None
raw = value
else:
return None
if raw is None:
return None
if isinstance(raw, bool):
return None
if isinstance(raw, (int, float)):
return max(0.0, float(raw))
text = str(raw).strip()
if not text:
return None
try:
return max(0.0, float(text))
except (TypeError, ValueError):
pass
# HTTP-date form (RFC 7231): seconds until that instant, clamped at 0.
try:
when = parsedate_to_datetime(text)
except (TypeError, ValueError):
return None
if when is None:
return None
if when.tzinfo is None:
when = when.replace(tzinfo=timezone.utc)
return max(0.0, (when - datetime.now(timezone.utc)).total_seconds())
def jittered_backoff(
attempt: int,
*,
+3 -1
View File
@@ -82,7 +82,9 @@ def resolve_cache_home(home_path: Optional[Path] = None) -> Path:
(and tests that don't thread a home through) working.
"""
if home_path is None:
home_path = Path(os.getenv("HERMES_HOME", Path.home() / ".hermes"))
from hermes_constants import get_hermes_home
home_path = get_hermes_home()
return home_path
+4
View File
@@ -239,6 +239,10 @@ _ENV_NAME_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
# ANSI CSI/OSC escape sequences — helper-CLI stderr often carries color
# codes that must not reach Hermes' own startup output.
# NOTE: intentionally NOT migrated to tools.ansi_strip.strip_ansi — the
# optional terminator here (``(?:\x07|\x1b\\)?``) also strips *unterminated*
# OSC sequences (common when a CLI is killed mid-write), which strip_ansi
# leaves untouched. strip_ansi is not a superset of this regex.
_ANSI_RE = re.compile(r"\x1b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\x07\x1b]*(?:\x07|\x1b\\)?)")
+4 -1
View File
@@ -667,7 +667,10 @@ def _run_bws_list(
bws: Path, access_token: str, project_id: str, server_url: str = ""
) -> Tuple[Dict[str, str], List[str]]:
cmd = [str(bws), "secret", "list", project_id, "--output", "json"]
env = os.environ.copy()
# bws child intentionally receives the access token; exact preservation
# (BWS_SERVER_URL manual overrides etc. must survive untouched).
from tools.environments.local import build_subprocess_env
env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False)
env["BWS_ACCESS_TOKEN"] = access_token
# Make sure we're not echoing telemetry / colour codes into json.
env.setdefault("NO_COLOR", "1")
+4 -1
View File
@@ -179,7 +179,10 @@ def _run_helper(
)
return None
env = os.environ.copy()
# User-configured secret-helper command: runs with the user's full shell
# env by design (it may need any credential to resolve the secret).
from tools.environments.local import build_subprocess_env
env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False)
env["HERMES_SECRET_KEY"] = secret_key
try:
+8 -6
View File
@@ -42,7 +42,6 @@ from __future__ import annotations
import hashlib
import logging
import os
import re
import shutil
import subprocess
import time
@@ -73,10 +72,10 @@ _OP_RUN_TIMEOUT = 30
# looks for.
_DEFAULT_TOKEN_ENV = "OP_SERVICE_ACCOUNT_TOKEN"
# Strip whole ANSI CSI sequences (colour, cursor moves, line erases) from any
# `op` diagnostic we surface — not just the lone ESC byte — so a control
# sequence can't reposition the cursor or hide text after a redaction marker.
_ANSI_CSI_RE = re.compile(r"\x1b\[[0-?]*[ -/]*[@-~]")
# ANSI stripping for `op` diagnostics we surface uses the shared
# tools.ansi_strip.strip_ansi (full ECMA-48: CSI, OSC, DCS/SOS/PM/APC,
# C1) so a control sequence can't reposition the cursor or hide text
# after a redaction marker.
# Env vars the `op` child actually needs. We build a minimal allowlisted env
# rather than copying all of os.environ (which, post-dotenv, holds every
@@ -231,7 +230,10 @@ def find_op(binary_path: str = "") -> Optional[Path]:
def _scrub(text: str) -> str:
"""Remove ANSI control sequences and trim, for safe message surfacing."""
return _ANSI_CSI_RE.sub("", text).replace("\x1b", "").strip()
from tools.ansi_strip import strip_ansi
# strip_ansi removes well-formed sequences; drop any stray lone ESC too.
return strip_ansi(text).replace("\x1b", "").strip()
def _op_child_env(token_value: str) -> Dict[str, str]:
+2 -2
View File
@@ -25,9 +25,9 @@ _INLINE_SHELL_MAX_OUTPUT = 4000
def load_skills_config() -> dict:
"""Load the ``skills`` section of config.yaml (best-effort)."""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config() or {}
cfg = load_config_readonly() or {}
skills_cfg = cfg.get("skills")
if isinstance(skills_cfg, dict):
return skills_cfg
+62
View File
@@ -49,6 +49,55 @@ EXCLUDED_SKILL_DIRS = frozenset(
# archive workflow preserves a complete old skill package under references/.
SKILL_SUPPORT_DIRS = frozenset(("references", "templates", "assets", "scripts"))
# ── Org-shared skills (sync contract) ───────────────────────────
# Org mirrors live under ~/.hermes/skills/_org/<org_id>/. Resolution is
# TOKEN-GATED via a marker file the sync client writes after verifying the
# token (skills_sync_client.pull_org_skills): only the marked org's mirror is
# scanned. No marker ⇒ no org skills load. The marker is plain data (org_id
# string) so this module stays import-light; the VERIFICATION lives in the
# sync client, which is the only writer. Offline grace: the marker persists,
# so already-pulled org skills keep working without connectivity; a VERIFIED
# org change (or personal-org token) rewrites/removes it.
ORG_MIRROR_DIR_NAME = "_org"
ORG_ACTIVE_MARKER = ".active_org"
ORG_PROVENANCE_FILE = ".org-provenance.json"
# Records the fingerprint of each skill exactly as upstream sent it, so a
# later local edit is detectable and an org pull can refuse to clobber it.
ORG_BASELINE_FILE = ".org-baseline.json"
def read_active_org_id(skills_dir: Path) -> Optional[str]:
"""The org id whose mirror may resolve, or None (no org skills load)."""
try:
marker = skills_dir / ORG_MIRROR_DIR_NAME / ORG_ACTIVE_MARKER
if not marker.exists():
return None
val = marker.read_text(encoding="utf-8").strip()
return val or None
except OSError:
return None
def is_org_mirror_path(path, skills_dir: Path) -> bool:
"""True when *path* is inside the org mirror (``_org/``)."""
try:
rel = Path(path).resolve().relative_to(Path(skills_dir).resolve())
except (OSError, ValueError):
return False
return bool(rel.parts) and rel.parts[0] == ORG_MIRROR_DIR_NAME
def org_id_of_path(path, skills_dir: Path) -> Optional[str]:
"""The ``<org_id>`` segment for a path under ``_org/<org_id>/...``."""
try:
rel = Path(path).resolve().relative_to(Path(skills_dir).resolve())
except (OSError, ValueError):
return None
if len(rel.parts) >= 2 and rel.parts[0] == ORG_MIRROR_DIR_NAME:
return rel.parts[1]
return None
def is_excluded_skill_path(path, *, root: Optional[Path] = None) -> bool:
"""True if *path* should be skipped by active skill scanners.
@@ -817,11 +866,24 @@ def iter_skill_index_files(skills_dir: Path, filename: str):
scripts) can contain arbitrary markdown and even archived package
``SKILL.md`` files, but they are progressive-disclosure data loaded through
``skill_view(..., file_path=...)`` rather than active skill roots.
M2 org mirrors (``_org/``): TOKEN-GATED resolution. Only the active org's
subdir (per the sync-client-written ``.active_org`` marker) is walked;
every other ``_org/<id>/`` (stale mirror from a previous org, or no
marker at all) is pruned — leave an org and its skills stop resolving,
without any manual cleanup.
"""
skills_dir_str = str(skills_dir)
active_org = read_active_org_id(skills_dir)
org_root = os.path.join(skills_dir_str, ORG_MIRROR_DIR_NAME)
matches: list[str] = []
for root, dirs, files in os.walk(skills_dir_str, followlinks=True):
has_skill_md = "SKILL.md" in files
if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None:
dirs.remove(ORG_MIRROR_DIR_NAME)
elif root == org_root:
# Inside _org/: descend ONLY into the active org's mirror.
dirs[:] = [d for d in dirs if d == active_org]
dirs[:] = [
d
for d in dirs
+1 -1
View File
@@ -18,7 +18,7 @@ import secrets
import threading
import time
from contextlib import contextmanager
from concurrent.futures import Future, ThreadPoolExecutor, TimeoutError
from concurrent.futures import Future, TimeoutError
from typing import Any, Callable, Mapping, Optional
+2 -2
View File
@@ -43,10 +43,10 @@ _TITLE_PROMPT_PINNED_LANGUAGE = (
def _title_language() -> str:
"""Return configured title language, or empty string to match the user."""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
return str(
((load_config() or {}).get("auxiliary") or {})
((load_config_readonly() or {}).get("auxiliary") or {})
.get("title_generation", {})
.get("language", "")
).strip()
-4
View File
@@ -32,7 +32,6 @@ from agent.display import (
redact_tool_args_for_display as _redact_tool_args_for_display,
_detect_tool_failure,
)
from agent.tool_guardrails import ToolGuardrailDecision
from agent.tool_dispatch_helpers import (
_is_destructive_command,
_is_multimodal_tool_result,
@@ -1980,9 +1979,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
return
break
if agent.tool_delay > 0 and i < len(assistant_message.tool_calls):
time.sleep(agent.tool_delay)
# ── Per-turn aggregate budget enforcement ─────────────────────────
num_tools_seq = len(assistant_message.tool_calls)
if finalize and num_tools_seq > 0:
-2
View File
@@ -378,7 +378,6 @@ class ChatCompletionsTransport(ProviderTransport):
ephemeral = params.get("ephemeral_max_output_tokens")
max_tokens = params.get("max_tokens")
anthropic_max_out = params.get("anthropic_max_output")
is_nvidia_nim = params.get("is_nvidia_nim", False)
is_kimi = params.get("is_kimi", False)
is_tokenhub = params.get("is_tokenhub", False)
reasoning_config = _reasoning_config_for_model(model, params.get("reasoning_config"))
@@ -436,7 +435,6 @@ class ChatCompletionsTransport(ProviderTransport):
extra_body: dict[str, Any] = {}
is_openrouter = params.get("is_openrouter", False)
is_nous = params.get("is_nous", False)
is_github_models = params.get("is_github_models", False)
provider_name = str(params.get("provider_name") or "").strip().lower()
base_url = params.get("base_url")
@@ -1054,6 +1054,18 @@ class CodexAppServerSession:
)
def _decide_exec_approval(self, params: dict) -> str:
"""Decide a Codex exec approval request.
This is protocol-level routing only — it carries NO Hermes
approval-mode/timeout logic. The Hermes-side resolution happens
upstream: ``agent/codex_runtime.py`` derives
``auto_approve_exec`` from the canonical
``tools.approval.is_approval_bypass_active()`` (which reads
``approvals.mode`` via ``tools.approval._get_approval_mode``),
and ``self._approval_callback`` itself runs the shared approval
gate (mode + ``approvals.timeout``) in ``tools/approval.py``.
Keep it that way — do not re-read approval config here.
"""
if self._routing.auto_approve_exec:
return "accept"
command = params.get("command") or ""
@@ -1077,6 +1089,12 @@ class CodexAppServerSession:
return "decline" # fail-closed when no callback wired
def _decide_apply_patch_approval(self, params: dict) -> str:
"""Decide a Codex apply_patch approval request.
Protocol-level routing only; Hermes approval-mode/timeout
resolution is delegated to ``tools/approval.py`` upstream — see
the docstring on ``_decide_exec_approval``.
"""
if self._routing.auto_approve_apply_patch:
return "accept"
if self._approval_callback is not None:
@@ -1231,6 +1249,11 @@ def _approval_choice_to_codex_decision(choice: str) -> str:
Codex expects 'accept', 'acceptForSession', 'decline', or 'cancel'
(verified against codex-rs/app-server-protocol/src/protocol/v2/item.rs
on codex 0.130.0).
This mapping is Codex-protocol-semantic and intentionally lives here,
NOT in tools/approval.py: the Hermes approval mode/timeout resolution
and the choice itself come from the shared core (tools/approval.py);
only the wire-value translation is local.
"""
if choice in {"once",}:
return "accept"
+2 -2
View File
@@ -1244,8 +1244,8 @@ def normalize_usage(
output_tokens = _to_int(getattr(response_usage, "completion_tokens", 0))
details = getattr(response_usage, "prompt_tokens_details", None)
# Primary: OpenAI-style prompt_tokens_details. Fallback: Anthropic-style
# top-level fields that some OpenAI-compatible proxies (OpenRouter, Cline)
# expose when routing Claude models — without this
# top-level fields that some OpenAI-compatible proxies (OpenRouter, Vercel
# AI Gateway, Cline) expose when routing Claude models — without this
# fallback, cache writes are undercounted as 0 and cache reads can be
# missed when the proxy only surfaces them at the top level.
# Port of cline/cline#10266.
+14 -54
View File
@@ -72,64 +72,24 @@ def _filter_verifiable_paths(paths: Iterable[str]) -> list[str]:
return [p for p in paths if p and not _is_non_code_path(p)]
# Session identities (platform or source) that are NOT human conversational
# messaging surfaces: interactive coding surfaces (CLI, TUI, desktop, codex,
# local, gateway) and programmatic callers (API server, webhooks, tools).
# Verify-on-stop stays ON by default for these. Any other resolved gateway
# platform is a conversational messaging surface (Telegram, Discord, WhatsApp,
# Signal, Slack, etc.) where the verification narrative would reach a human as
# chat noise, so it defaults OFF. Mirrors LOCAL_SESSION_SOURCE_IDS in
# apps/desktop/src/lib/session-source.ts; keep roughly in sync when adding a
# local or programmatic surface. Default-deny by design: an unrecognized
# identity is treated as messaging (OFF) so a new chat platform never leaks the
# verification receipt before this set is updated.
_NON_MESSAGING_SESSION_SURFACES = frozenset(
{
"",
"cli",
"codex",
"desktop",
"gateway",
"local",
"tui",
"tool",
"api_server",
"webhook",
"msgraph_webhook",
}
)
def _session_is_messaging_surface() -> bool:
"""Return whether this turn is delivered over a human messaging channel.
"""Whether this turn is delivered over a human messaging channel.
The gateway binds the platform value (e.g. ``telegram``) to
``HERMES_SESSION_PLATFORM``; the CLI and TUI set ``HERMES_SESSION_SOURCE``
(e.g. ``cli``, ``tui``) instead. Both are consulted via the session-context
helper (with an ``os.environ`` fallback), alongside the ``HERMES_PLATFORM``
override, matching the sibling platform resolution in
``agent/skill_commands.py`` and ``agent/prompt_builder.py``. A turn is a
messaging surface when a resolved identity is present and is not a known
non-messaging surface.
Verify-on-stop defaults ON for the interactive coding surfaces and
programmatic callers, and OFF on a conversational platform (Telegram,
Discord, Slack, ...) where the verification narrative reaches a human as
chat noise. The surface classification itself is shared with the other
consumers of this distinction — see
``gateway.session_context.session_is_messaging_surface``.
"""
try:
from gateway.session_context import get_session_env
from gateway.session_context import session_is_messaging_surface
platform = (
os.getenv("HERMES_PLATFORM")
or get_session_env("HERMES_SESSION_PLATFORM", "")
)
source = get_session_env("HERMES_SESSION_SOURCE", "")
return session_is_messaging_surface()
except Exception:
platform = os.getenv("HERMES_PLATFORM", "") or os.environ.get(
"HERMES_SESSION_PLATFORM", ""
)
source = os.environ.get("HERMES_SESSION_SOURCE", "")
for identity in (platform, source):
identity = str(identity or "").strip().lower()
if identity and identity not in _NON_MESSAGING_SESSION_SURFACES:
return True
return False
# The gateway package is unreachable, so there is no messaging channel
# to be on. Reporting a local surface keeps verify-on-stop enabled.
return False
def verify_on_stop_enabled(config: dict[str, Any] | None = None) -> bool:
@@ -149,9 +109,9 @@ def verify_on_stop_enabled(config: dict[str, Any] | None = None) -> bool:
return env.strip().lower() not in {"0", "false", "no", "off"}
if config is None:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
config = load_config()
config = load_config_readonly()
except Exception:
config = {}
agent_cfg = (config or {}).get("agent") if isinstance(config, dict) else None
+2 -2
View File
@@ -84,9 +84,9 @@ def get_active_provider() -> Optional[VideoGenProvider]:
"""
configured: Optional[str] = None
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
section = cfg.get("video_gen") if isinstance(cfg, dict) else None
if isinstance(section, dict):
raw = section.get("provider")
+2 -2
View File
@@ -98,9 +98,9 @@ def get_provider(name: str) -> Optional[WebSearchProvider]:
def _read_config_key(*path: str) -> Optional[str]:
"""Resolve a dotted config key from ``config.yaml``. Returns None on miss."""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
cur = cfg
for segment in path:
if not isinstance(cur, dict):
@@ -66,6 +66,10 @@ windows-sys = { version = "0.59", features = [
"Win32_UI_WindowsAndMessaging",
] }
# Signal-0 liveness probe for the update-lock marker owner (update.rs).
[target.'cfg(unix)'.dependencies]
libc = "0.2"
[profile.release]
# A 5-10MB signed installer is the goal. LTO + size-opt + single codegen unit.
panic = "abort"
+362 -14
View File
@@ -107,16 +107,110 @@ pub async fn start_update(app: AppHandle) -> Result<(), String> {
/// future desktop launches. The marker payload is `{pid}\n{started_at_unix}`
/// so the desktop's launch gate can detect a stale marker (dead PID / past a
/// hard ceiling) and self-heal rather than wait forever.
///
/// The marker is also the cross-process update lock: `hermes update` claims
/// the same file (see `hermes_cli/update_lock.py`) so a dashboard-spawned
/// update and this updater can't mutate one checkout at the same time.
/// `acquire` therefore REFUSES when a live foreign owner holds it rather than
/// overwriting — the pre-fix clobber is what let a dashboard `hermes update`
/// keep running while install-mode bootstrap rewrote the tree underneath it.
struct UpdateMarkerGuard {
path: PathBuf,
/// False when a live foreign updater already owns the marker: we hold no
/// claim, so `Drop` must not delete their marker.
owned: bool,
}
/// Never treat a marker older than this as a live update. Mirrors
/// UPDATE_MARKER_MAX_AGE_MS in apps/desktop/electron/update-marker.ts and
/// UPDATE_MARKER_MAX_AGE_SECONDS in hermes_cli/update_lock.py — all three read
/// this one file, so a shorter ceiling in any of them would steal a lock the
/// others still consider live.
const UPDATE_MARKER_MAX_AGE_SECS: u64 = 20 * 60;
/// The pid + age of a confirmed-live update holding the marker.
struct MarkerOwner {
pid: u32,
age_secs: u64,
}
/// Read the marker and report a live *foreign* owner, if any. `None` for every
/// "no live update" case — absent, unreadable, malformed, dead pid, past the
/// ceiling, or a marker whose pid is **this** process — matching
/// `readLiveUpdateMarker` in the Electron gate. Never panics.
///
/// Self-PID is treated as non-ownership on purpose (#74761): since #50238 the
/// desktop pre-writes this marker with the spawned updater's pid before the
/// updater reaches `acquire`. Without the exclusion, `acquire` sees a live
/// owner that is itself and aborts ("Another Hermes update is already
/// running"), then the desktop relaunches and retries forever. A foreign live
/// pid (e.g. a dashboard-spawned `hermes update`) still blocks.
fn live_marker_owner(path: &Path) -> Option<MarkerOwner> {
let raw = std::fs::read_to_string(path).ok()?;
let mut lines = raw.lines();
let pid: u32 = lines.next()?.trim().parse().ok()?;
let started_at: u64 = lines.next().unwrap_or("").trim().parse().unwrap_or(0);
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0);
let age_secs = now.saturating_sub(started_at);
if age_secs > UPDATE_MARKER_MAX_AGE_SECS || !pid_is_alive(pid) {
return None;
}
// Desktop `writeUpdateMarker(hermesHome, child.pid)` races ahead of us;
// adopt that pre-claim rather than refusing our own marker.
if pid == std::process::id() {
return None;
}
Some(MarkerOwner { pid, age_secs })
}
/// True when a process with `pid` currently exists.
#[cfg(windows)]
fn pid_is_alive(pid: u32) -> bool {
use windows_sys::Win32::Foundation::{CloseHandle, STILL_ACTIVE};
use windows_sys::Win32::System::Threading::{
GetExitCodeProcess, OpenProcess, PROCESS_QUERY_LIMITED_INFORMATION,
};
unsafe {
let handle = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, 0, pid);
if handle.is_null() {
// Either the pid is gone or we lack rights to open it. A pid we
// can't inspect is treated as dead so an unopenable straggler
// can't wedge every future update.
return false;
}
let mut code: u32 = 0;
let ok = GetExitCodeProcess(handle, &mut code);
CloseHandle(handle);
ok != 0 && code == STILL_ACTIVE as u32
}
}
#[cfg(not(windows))]
fn pid_is_alive(pid: u32) -> bool {
// signal 0 delivers nothing; it only probes existence/permission.
// ESRCH => dead. EPERM => alive but owned by another user.
let rc = unsafe { libc::kill(pid as libc::pid_t, 0) };
if rc == 0 {
return true;
}
std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM)
}
impl UpdateMarkerGuard {
/// Write the marker. Best-effort: a write failure must NOT abort the
/// update (the gate degrades to "no marker => proceed", i.e. exactly the
/// pre-fix behavior), so we log and carry on with a guard that still
/// attempts cleanup of whatever may exist at the path.
fn acquire(path: PathBuf) -> Self {
/// Claim the marker, or report the live updater that already owns it.
///
/// Writing is best-effort: a write failure must NOT abort the update (the
/// gate degrades to "no marker => proceed", i.e. exactly the pre-marker
/// behavior), so we log and carry on with a guard that still attempts
/// cleanup of whatever may exist at the path.
fn acquire(path: PathBuf) -> Result<Self, MarkerOwner> {
if let Some(owner) = live_marker_owner(&path) {
return Err(owner);
}
let pid = std::process::id();
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
@@ -128,17 +222,32 @@ impl UpdateMarkerGuard {
if let Err(err) = std::fs::write(&path, format!("{pid}\n{started_at}")) {
tracing::warn!(?path, %err, "could not write update-in-progress marker");
}
Self { path }
Ok(Self { path, owned: true })
}
/// Release the marker as soon as every mutating stage has completed.
///
/// The updater still owns a Tauri/Cocoa event loop while it relaunches the
/// desktop, and that loop can outlive `app.exit(0)`. Relying on `Drop`
/// alone therefore leaves a *successful* update looking active — a live
/// pid holding a fresh marker — which blocks desktop startup and every
/// other updater for the full age ceiling. Idempotent: `Drop` still runs
/// and tolerates an already-removed marker.
fn complete(&self) {
if !self.owned {
return;
}
if let Err(err) = std::fs::remove_file(&self.path) {
if err.kind() != std::io::ErrorKind::NotFound {
tracing::warn!(path = ?self.path, %err, "could not remove completed update marker");
}
}
}
}
impl Drop for UpdateMarkerGuard {
fn drop(&mut self) {
if let Err(err) = std::fs::remove_file(&self.path) {
if err.kind() != std::io::ErrorKind::NotFound {
tracing::warn!(path = ?self.path, %err, "could not remove update-in-progress marker");
}
}
self.complete();
}
}
@@ -152,7 +261,39 @@ async fn run_update(app: AppHandle) -> Result<()> {
// it, that backend re-locks the venv shim, our `force_kill_other_hermes`
// straggler-cleanup kills it, and the relaunch/kill cycle loops. The guard
// removes the marker on every exit path (incl. early returns / panics).
let _update_marker = UpdateMarkerGuard::acquire(crate::paths::update_in_progress_marker());
//
// The same marker is the cross-process update lock (hermes_cli/
// update_lock.py claims it too), so a live foreign owner means another
// updater — most often a dashboard-spawned `hermes update` — is already
// mutating this checkout. Refuse instead of running a second one over it.
let _update_marker = match UpdateMarkerGuard::acquire(
crate::paths::update_in_progress_marker(),
) {
Ok(guard) => guard,
Err(owner) => {
let mins = owner.age_secs / 60;
let secs = owner.age_secs % 60;
let elapsed = if mins > 0 {
format!("{mins}m {secs}s")
} else {
format!("{secs}s")
};
let msg = format!(
"Another Hermes update is already running (PID {}, started {} ago). \
Wait for it to finish, or close the window or dashboard tab that \
started it, then try again.",
owner.pid, elapsed
);
emit(
&app,
BootstrapEvent::Failed {
stage: None,
error: msg.clone(),
},
);
return Err(anyhow!(msg));
}
};
let update_branch = update_branch_from_args(std::env::args().skip(1))
.or_else(|| option_env_string("BUILD_PIN_BRANCH"))
@@ -453,6 +594,12 @@ async fn run_update(app: AppHandle) -> Result<()> {
marker: None,
},
);
// Every install-tree mutation is finished. Release the lock BEFORE the
// relaunch: this process can stay wedged in its native event loop even
// after a successful app.exit(), and a live pid on a fresh marker would
// make a completed update look active — blocking desktop startup and
// every other updater until the age ceiling expires.
_update_marker.complete();
if let Some(target_app) = launch_target {
if let Err(err) = launch_macos_app_and_exit(&app, &target_app).await {
@@ -477,9 +624,26 @@ async fn run_update(app: AppHandle) -> Result<()> {
);
}
// The launch helpers normally request exit themselves, but their failure
// paths must still close a successful updater. A native event loop can
// ignore that graceful request, so arm a process-exit fallback now that
// all update state and the marker have been settled.
exit_after_success(&app);
Ok(())
}
/// Ask the app to exit, with a hard `process::exit` fallback for a native
/// event loop that ignores the graceful request. Without it a finished updater
/// can linger as a live pid forever.
fn exit_after_success(app: &AppHandle) {
std::thread::spawn(|| {
std::thread::sleep(std::time::Duration::from_secs(3));
tracing::warn!("graceful updater exit timed out; forcing process exit");
std::process::exit(0);
});
app.exit(0);
}
/// Poll until the venv shim AND packaged desktop app bundle are no longer locked
/// (Windows) or a bounded timeout elapses. On non-Windows this is a short fixed
/// grace since file locking isn't the failure mode there.
@@ -744,6 +908,17 @@ fn update_child_env(install_root: &Path) -> Vec<(String, OsString)> {
// a frozen stage, and users cancel a healthy update. Force line-by-line
// output instead.
envs.push(("PYTHONUNBUFFERED".to_string(), OsString::from("1")));
// We hold the update-in-progress marker for this whole run, and the
// `hermes update` child claims that SAME lock (hermes_cli/update_lock.py).
// Name our pid so the child recognizes the live holder as its own
// orchestrator and runs under our claim — without this every GUI update
// refuses its parent's marker with exit 2 ("Hermes is still running")
// and no number of retries can ever succeed. Keep the variable name in
// sync with HANDOFF_PID_ENV in hermes_cli/update_lock.py.
envs.push((
"HERMES_UPDATE_HANDOFF_PID".to_string(),
OsString::from(std::process::id().to_string()),
));
if let Some(path) = path_with_prepended_entries(&[
hermes_home.join("node").join("bin"),
venv_bin_dir(install_root),
@@ -1067,6 +1242,17 @@ mod tests {
);
}
#[test]
fn update_child_env_names_our_pid_for_the_lock_handoff() {
let envs = update_child_env(Path::new("/x/hermes-agent"));
assert!(
envs.iter().any(|(k, v)| k == "HERMES_UPDATE_HANDOFF_PID"
&& v.to_str() == Some(std::process::id().to_string().as_str())),
"the hermes update child claims the same marker we hold; without our pid \
it refuses its own parent's lock and every GUI update dead-ends on exit 2"
);
}
#[test]
fn lock_probe_paths_include_desktop_app_payload() {
let root = Path::new("/x/hermes-agent");
@@ -1102,7 +1288,8 @@ mod tests {
let marker = dir.join(".hermes-update-in-progress");
{
let _g = UpdateMarkerGuard::acquire(marker.clone());
let _g = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
assert!(marker.exists(), "marker must exist while the guard is held");
let body = std::fs::read_to_string(&marker).unwrap();
let pid_line = body.lines().next().unwrap();
@@ -1127,7 +1314,8 @@ mod tests {
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
let guard = UpdateMarkerGuard::acquire(marker.clone());
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
// Simulate an external cleanup (e.g. the desktop pruned a marker it
// judged stale) before our guard drops — Drop must not panic.
std::fs::remove_file(&marker).unwrap();
@@ -1137,6 +1325,166 @@ mod tests {
let _ = std::fs::remove_dir_all(&dir);
}
/// Spawn a short-lived sibling process whose pid stands in for a foreign
/// updater. Same-process double-acquire no longer models contention: since
/// #74761 `live_marker_owner` treats our own pid as adoptable (desktop
/// pre-writes it), so a second acquire in *this* process would succeed.
fn spawn_foreign_holder() -> std::process::Child {
#[cfg(windows)]
{
std::process::Command::new("timeout")
.args(["/t", "30", "/nobreak"])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.spawn()
.expect("spawn foreign marker holder")
}
#[cfg(not(windows))]
{
std::process::Command::new("sleep")
.arg("30")
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.spawn()
.expect("spawn foreign marker holder")
}
}
#[test]
fn acquire_refuses_while_a_live_updater_owns_the_marker() {
let dir = unique_tmp_dir("marker-contended");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
// A live *foreign* updater holds it. We must NOT clobber the marker and
// run concurrently over the same checkout — that race is what let a
// dashboard `hermes update` and install-mode bootstrap mutate one tree
// at once. Own-pid markers are adoptable (#74761), so the foreign pid
// must be a real sibling process.
let mut foreign = spawn_foreign_holder();
let foreign_pid = foreign.id();
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0);
std::fs::write(&marker, format!("{foreign_pid}\n{started_at}")).unwrap();
let owner = UpdateMarkerGuard::acquire(marker.clone())
.err()
.expect("acquire must be refused while a foreign updater is live");
assert_eq!(owner.pid, foreign_pid);
// The refused guard must not delete the live owner's marker.
assert!(marker.exists(), "refused acquire must leave the marker intact");
let _ = foreign.kill();
let _ = foreign.wait();
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn acquire_adopts_a_marker_prewritten_with_our_own_pid() {
// #74761: desktop writeUpdateMarker(hermesHome, child.pid) races ahead
// of UpdateMarkerGuard::acquire. The marker names US; refusing it made
// every in-app desktop update loop forever. Adopt and rewrite.
let dir = unique_tmp_dir("marker-own-pid");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0)
.saturating_sub(2);
std::fs::write(&marker, format!("{}\n{started_at}", std::process::id())).unwrap();
let guard = UpdateMarkerGuard::acquire(marker.clone()).unwrap_or_else(|owner| {
panic!(
"own-pid pre-write must be adoptable, got foreign owner pid={}",
owner.pid
)
});
assert!(marker.exists(), "adopted guard must own the marker");
let body = std::fs::read_to_string(&marker).unwrap();
assert_eq!(
body.lines().next().unwrap().trim().parse::<u32>().unwrap(),
std::process::id(),
"acquire rewrites the marker with our pid + fresh started_at"
);
drop(guard);
assert!(
!marker.exists(),
"Drop must still clear the marker we adopted"
);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn acquire_reclaims_a_marker_owned_by_a_dead_pid() {
let dir = unique_tmp_dir("marker-dead-pid");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
// pid 1 exists everywhere, so fabricate a dead one: a very large pid
// that no live process owns. A crashed updater must never wedge every
// future update.
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0);
std::fs::write(&marker, format!("4294967294\n{started_at}")).unwrap();
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("a dead owner must not block acquisition"));
let body = std::fs::read_to_string(&marker).unwrap();
assert_eq!(
body.lines().next().unwrap().trim().parse::<u32>().unwrap(),
std::process::id(),
"reclaiming rewrites the marker with our pid"
);
drop(guard);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn acquire_reclaims_a_marker_past_the_age_ceiling() {
let dir = unique_tmp_dir("marker-stale-age");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
// Our own (live) pid, but started well past the ceiling: a wedged
// updater must not hold the lock forever.
let long_ago = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0)
.saturating_sub(UPDATE_MARKER_MAX_AGE_SECS + 60);
std::fs::write(&marker, format!("{}\n{long_ago}", std::process::id())).unwrap();
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("a marker past the ceiling must be reclaimable"));
drop(guard);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn completed_update_releases_marker_before_guard_drop() {
let dir = unique_tmp_dir("marker-complete");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
guard.complete();
assert!(
!marker.exists(),
"a successful update must unblock desktop startup before relaunch/exit"
);
drop(guard);
assert!(!marker.exists(), "Drop stays idempotent after completion");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn parses_update_branch_from_space_or_equals_args() {
assert_eq!(
+10 -1
View File
@@ -91,10 +91,11 @@ for call-site shadow or border inventions.
| Token | Use |
| --- | --- |
| `--ui-stroke-primary…quaternary` | hairlines, in descending strength |
| `--ui-stroke-tertiary` | the default in-panel divider / list hairline |
| `--ui-stroke-tertiary` | the default in-panel divider / list hairline — and every bordered surface in the transcript |
| `--stroke-nous` | the overlay hairline (pairs with `shadow-nous`) |
| `--ui-text-primary / -secondary / -tertiary` | text hierarchy |
| `--ui-bg-quaternary` | soft control fill (secondary button) |
| `--ui-widget-surface-background` | fill for inline chat widgets (`WIDGET_SHELL_CLASS`) |
| `--chrome-action-hover` | hover fill for quiet controls |
| `--theme-primary`, `--ui-accent` | brand/accent |
@@ -196,6 +197,14 @@ Notes:
existing components under `src/components/assistant-ui` and
`src/app/chat/composer`; do not fork a second markdown, message, tool-call, or
approval renderer for one feature.
- **Inline widgets** — a tool result that renders as a panel the user reads or
acts on (clarify, artifact card) wears `WIDGET_SHELL_CLASS`
(`src/components/chat/widget-shell.ts`): shared radius, the
`--ui-widget-surface-background` fill, no border. Its actions sit *outside*
the panel, below it. Don't give one widget its own radius or fill.
- Bordered surfaces in the transcript (tables, fences, callouts, attachments)
use `--ui-stroke-tertiary`. Not `border-border` — that's the app-wide
default and reads too hot against the thread.
- A tool result may expose an inline action that opens a preview. It must not
open the rail automatically.
- Install, onboarding, connecting, boot failure, and reauthentication are
+20 -1
View File
@@ -14,7 +14,10 @@ import { test } from 'vitest'
import {
canImportHermesCli,
DEFAULT_PROBE_TIMEOUT_MS,
hermesRuntimeImportProbe,
PROBE_TIMEOUT_MS,
resolveProbeTimeoutMs,
shouldTrustHermesOverride,
verifyHermesCli
} from './backend-probes'
@@ -100,9 +103,25 @@ test('verifyHermesCli returns true when --version exits 0', () => {
})
test('verifyHermesCli swallows timeouts (does not throw)', () => {
// We can't easily provoke a real 5s hang in CI without slowing the
// We can't easily provoke a real hang in CI without slowing the
// suite, but we CAN confirm that an invocation that DOES throw
// (because the binary is missing) returns false rather than
// propagating. Same code path the timeout case takes.
assert.equal(verifyHermesCli('/definitely/not/a/real/binary/anywhere'), false)
})
test('default probe timeout is 15s (not the old 5s death-loop value)', () => {
assert.equal(DEFAULT_PROBE_TIMEOUT_MS, 15_000)
// Module constant uses process.env at load time; with no override it
// matches the default (tests run without HERMES_PROBE_TIMEOUT_MS).
assert.equal(PROBE_TIMEOUT_MS, DEFAULT_PROBE_TIMEOUT_MS)
})
test('resolveProbeTimeoutMs honours HERMES_PROBE_TIMEOUT_MS', () => {
assert.equal(resolveProbeTimeoutMs({}), DEFAULT_PROBE_TIMEOUT_MS)
assert.equal(resolveProbeTimeoutMs({ HERMES_PROBE_TIMEOUT_MS: '30000' }), 30_000)
assert.equal(resolveProbeTimeoutMs({ HERMES_PROBE_TIMEOUT_MS: '0' }), DEFAULT_PROBE_TIMEOUT_MS)
assert.equal(resolveProbeTimeoutMs({ HERMES_PROBE_TIMEOUT_MS: 'nope' }), DEFAULT_PROBE_TIMEOUT_MS)
// Cap runaway values
assert.equal(resolveProbeTimeoutMs({ HERMES_PROBE_TIMEOUT_MS: '999999' }), 120_000)
})
+91 -6
View File
@@ -20,8 +20,9 @@
* actually works.
*
* Both probes are deliberately fast and forgiving:
* - 5s timeout (a hung interpreter beats forever, but we still give
* slow disks / cold caches room to breathe)
* - default 15s timeout (5s was too short on cold Windows disks / AV;
* issue #61764 death-loop) with HERMES_PROBE_TIMEOUT_MS override
* - one automatic retry after a timeout before declaring the runtime dead
* - stdio ignored (we only care about exit code; stdout/stderr are
* not surfaced to the user, just to recentHermesLog for forensics
* via the caller's catch block if it chooses)
@@ -34,7 +35,82 @@
import { execFileSync } from 'node:child_process'
const PROBE_TIMEOUT_MS = 5000
/** Default probe budget. 5s false-negativeed healthy Windows cold starts (#61764). */
const DEFAULT_PROBE_TIMEOUT_MS = 15_000
/**
* Resolve the backend probe timeout (ms).
* Honours HERMES_PROBE_TIMEOUT_MS when it parses as a positive integer.
*/
function resolveProbeTimeoutMs(env: NodeJS.ProcessEnv = process.env): number {
const raw = env.HERMES_PROBE_TIMEOUT_MS
if (raw == null || raw === '') {
return DEFAULT_PROBE_TIMEOUT_MS
}
const n = Number.parseInt(String(raw), 10)
if (!Number.isFinite(n) || n <= 0) {
return DEFAULT_PROBE_TIMEOUT_MS
}
// Clamp absurd values (ms) so a typo can't hang startup forever.
return Math.min(n, 120_000)
}
const PROBE_TIMEOUT_MS = resolveProbeTimeoutMs()
function isTimeoutError(err: unknown): boolean {
if (!err || typeof err !== 'object') {
return false
}
const e = err as { code?: string; killed?: boolean; signal?: string }
if (e.killed === true) {
return true
}
if (e.code === 'ETIMEDOUT') {
return true
}
// Node marks timed-out execFileSync with SIGTERM on some platforms.
if (e.signal === 'SIGTERM') {
return true
}
return false
}
/**
* Run execFileSync; on timeout only, retry once before failing.
* Non-timeout failures (ENOENT, non-zero exit) fail immediately.
*/
function execProbeSync(
command: string,
args: string[],
options: {
cwd?: string
env?: NodeJS.ProcessEnv
stdio: 'ignore'
timeout: number
shell?: boolean
windowsHide?: boolean
}
): void {
try {
execFileSync(command, args, options)
} catch (err) {
if (!isTimeoutError(err)) {
throw err
}
// One cold-cache / AV miss should not force hermes-setup --update (#61764).
execFileSync(command, args, options)
}
}
/**
* Return the Python snippet used to verify Hermes can import far enough to
@@ -71,7 +147,7 @@ function canImportHermesCli(pythonPath: string, opts: { env?: Record<string, str
}
try {
execFileSync(pythonPath, ['-c', hermesRuntimeImportProbe()], {
execProbeSync(pythonPath, ['-c', hermesRuntimeImportProbe()], {
env: { ...process.env, ...(opts.env || {}) },
stdio: 'ignore',
timeout: PROBE_TIMEOUT_MS,
@@ -120,7 +196,7 @@ function verifyHermesCli(hermesCommand: string, opts?: { shell?: boolean }) {
}
try {
execFileSync(hermesCommand, ['--version'], {
execProbeSync(hermesCommand, ['--version'], {
stdio: 'ignore',
timeout: PROBE_TIMEOUT_MS,
shell: Boolean(opts?.shell),
@@ -133,4 +209,13 @@ function verifyHermesCli(hermesCommand: string, opts?: { shell?: boolean }) {
}
}
export { canImportHermesCli, hermesRuntimeImportProbe, PROBE_TIMEOUT_MS, shouldTrustHermesOverride, verifyHermesCli }
export {
canImportHermesCli,
DEFAULT_PROBE_TIMEOUT_MS,
execProbeSync,
hermesRuntimeImportProbe,
PROBE_TIMEOUT_MS,
resolveProbeTimeoutMs,
shouldTrustHermesOverride,
verifyHermesCli
}
+148 -33
View File
@@ -37,7 +37,13 @@ import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command'
import { createBackendConnectionState } from './backend-connection-state'
import { buildDesktopBackendEnv, normalizeHermesHomeRoot } from './backend-env'
import { isReauthRequiredError, waitForHermesReady } from './backend-health'
import { canImportHermesCli, shouldTrustHermesOverride, verifyHermesCli } from './backend-probes'
import {
canImportHermesCli,
execProbeSync,
PROBE_TIMEOUT_MS,
shouldTrustHermesOverride,
verifyHermesCli
} from './backend-probes'
import { waitForDashboardPortAnnouncement } from './backend-ready'
import { shouldLatchBackendStartFailure, shouldLatchRemoteReauthFailure } from './backend-start-failure'
import { detectRemoteDisplay, isWindowsBinaryPathInWsl, isWslEnvironment } from './bootstrap-platform'
@@ -179,6 +185,7 @@ import {
} from './ssh-connection'
import { nativeOverlayWidth as computeNativeOverlayWidth, macTitleBarOverlayHeight } from './titlebar-overlay-width'
import { resolveBehindCount, shouldCountCommits } from './update-count'
import { waitForUpdateClearance } from './update-gate'
import { readLiveUpdateMarker, writeUpdateMarker } from './update-marker'
import { runRebuildWithRetry } from './update-rebuild'
import {
@@ -1730,30 +1737,48 @@ const UPDATE_WAIT_POLL_MS = 1000
// updater's own progress window appears. (#50419)
const UPDATE_HANDOFF_DWELL_MS = 2500
// Gate deps shared by the primary-window boot path and the pool-backend
// spawn path. Consulting BOTH the on-disk marker and the in-process
// updateInFlight flag is load-bearing (#73822): applyUpdates kills its own
// backend BEFORE the Windows venv-blocker scan but only writes the marker
// AFTER it, so a marker-only gate lets the renderer's ~1s reconnect respawn
// a backend inside the update's own critical section — which the scan then
// reports as a blocker, aborting every update attempt.
function updateGateDeps() {
return {
hasLiveMarker: () => Boolean(readLiveUpdateMarker(HERMES_HOME)),
isUpdateInFlight: () => updateInFlight
}
}
// Block until no live update is in progress (or we hit the wait timeout).
// Emits a boot-progress phase so the renderer shows "Update in progress…"
// rather than a frozen splash. Returns true if it parked at all.
async function waitForUpdateToFinish() {
let marker = readLiveUpdateMarker(HERMES_HOME)
let announced = false
if (!marker) {
const outcome = await waitForUpdateClearance(updateGateDeps(), {
onWaitTick: async reason => {
if (!announced) {
announced = true
rememberLog(`[updates] update in progress (${reason}); deferring backend start until it finishes`)
}
await advanceBootProgress(
'backend.update-wait',
'An update is finishing — Hermes will start automatically when it completes…',
12
)
},
pollMs: UPDATE_WAIT_POLL_MS,
timeoutMs: UPDATE_WAIT_TIMEOUT_MS
})
if (outcome === 'clear') {
return false
}
rememberLog(`[updates] update in progress (pid=${marker.pid}); deferring backend start until it finishes`)
const deadline = Date.now() + UPDATE_WAIT_TIMEOUT_MS
while (marker && Date.now() < deadline) {
await advanceBootProgress(
'backend.update-wait',
'An update is finishing — Hermes will start automatically when it completes…',
12
)
await new Promise(r => setTimeout(r, UPDATE_WAIT_POLL_MS))
marker = readLiveUpdateMarker(HERMES_HOME)
}
if (marker) {
if (outcome === 'timeout') {
rememberLog('[updates] update still in progress after wait timeout; starting backend anyway')
} else {
rememberLog('[updates] update finished; proceeding with backend start')
@@ -1867,11 +1892,22 @@ function backendSupportsServe(backend) {
if (supported === null) {
try {
const prefix = backend.args && backend.args[0] === '-m' ? backend.args.slice(0, 2) : []
execFileSync(backend.command, [...prefix, 'serve', '--help'], {
// Same cold-Windows Python-startup class as the runtime probes
// (#61764/#72632/#72707): `serve --help` imports at least as much as
// `hermes --version` (~10.5s measured cold), and a false negative here
// is cached for the process lifetime, silently routing a modern
// runtime through the legacy `dashboard` form. Share the probe budget
// and its timeout-only retry instead of a thinner local bound.
execProbeSync(backend.command, [...prefix, 'serve', '--help'], {
cwd: backend.root || undefined,
env: { ...process.env, HERMES_HOME, ...(backend.env || {}) },
timeout: 15000,
timeout: PROBE_TIMEOUT_MS,
stdio: 'ignore',
// `.cmd`/`.bat` shim backends carry shell: true in their descriptor
// (see resolveHermesBackend step 4); execFileSync of a .cmd without
// shell throws EINVAL on modern Node, which the catch below would
// mis-cache as "serve unsupported" for the process lifetime.
shell: Boolean(backend.shell),
windowsHide: true
})
supported = true
@@ -2028,7 +2064,10 @@ function findSystemPython() {
const out = execFileSync(
'reg',
['query', `${hive}\\SOFTWARE\\Python\\PythonCore\\${version}\\InstallPath`, '/ve', '/reg:64'],
hiddenWindowsChildOptions({ encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] })
// Registry reads are near-instant; the bound only exists so a
// pathologically wedged reg.exe can't hang the synchronous boot
// resolver forever (this ran unbounded before).
hiddenWindowsChildOptions({ encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], timeout: 5_000 })
)
// Output format: " (Default) REG_SZ C:\Path\To\Python\"
@@ -2083,7 +2122,12 @@ function findSystemPython() {
[`-${version}`, '-c', 'import sys; print(sys.executable)'],
hiddenWindowsChildOptions({
encoding: 'utf8',
stdio: ['ignore', 'pipe', 'ignore']
stdio: ['ignore', 'pipe', 'ignore'],
// Bare interpreter startup — much lighter than the hermes-import
// probes, but still python.exe under cold cache / AV scan, so
// share the probe budget rather than running unbounded (this
// synchronous exec previously had no timeout at all).
timeout: PROBE_TIMEOUT_MS
})
)
@@ -3860,17 +3904,20 @@ function resolveHermesBackend(backendArgs) {
// through to the install-script bootstrap if the optional probe times
// out under load; the pinned backend is the only valid runtime there.
if (shouldTrustHermesOverride(hermesOverride) || verifyHermesCli(hermesCommand, { shell: shellForProbe })) {
return (
unwrapWindowsVenvHermesCommand(hermesCommand, backendArgs) || {
label: `existing Hermes CLI at ${hermesCommand}`,
command: hermesCommand,
args: backendArgs,
bootstrap: false,
env: {},
kind: 'command',
shell: shellForProbe
}
)
// `unwrapped` above already answered "is this a Windows venv shim?" —
// it was null (not a shim, or its import probe failed). Do NOT re-run
// unwrapWindowsVenvHermesCommand here: the second call repeats the
// same un-memoized import probe, costing up to another full probe
// timeout on the boot path for an answer we already have.
return {
label: `existing Hermes CLI at ${hermesCommand}`,
command: hermesCommand,
args: backendArgs,
bootstrap: false,
env: {},
kind: 'command',
shell: shellForProbe
}
}
rememberLog(
@@ -5108,6 +5155,20 @@ function sendClosePreviewRequested() {
webContents.send('hermes:close-preview-requested')
}
function sendOpenFolderRequested() {
if (!mainWindow || mainWindow.isDestroyed()) {
return
}
const webContents = mainWindow.webContents
if (!webContents || webContents.isDestroyed()) {
return
}
webContents.send('hermes:open-folder-requested')
}
// Tell the renderer the machine just woke. Sleep silently drops the
// renderer's WebSocket to the local backend; the renderer reconnects on this
// signal so the chat composer doesn't stay stuck on "Starting Hermes...".
@@ -5225,6 +5286,10 @@ function buildApplicationMenu() {
// a menu accelerator would fight the rebind panel and (on macOS) be
// swallowed before the renderer sees it. Here purely for discoverability.
{ click: () => createInstanceWindow(), label: 'New Window' },
// Same no-accelerator rationale: ⌘O is the rebindable renderer keybind
// (workspace.openFolder). Clicking runs the same open-folder-as-project
// flow through the renderer.
{ click: () => sendOpenFolderRequested(), label: 'Open Folder…' },
{ type: 'separator' },
IS_MAC
? {
@@ -5577,8 +5642,12 @@ function installContextMenu(window) {
}
}
// Bare right-click on non-editable, non-selected, non-media content (a pane
// body, the sidebar, chrome): the renderer's own context menus own those
// surfaces, and anywhere without one shows nothing — not a lone, useless
// "Select All" from the native fallback.
if (!template.length) {
template.push({ role: 'selectAll' })
return
}
Menu.buildFromTemplate(template).popup({ window })
@@ -8007,6 +8076,27 @@ async function spawnPoolBackend(profile, entry) {
}
const token = crypto.randomBytes(32).toString('base64url')
// Same update mutual exclusion as the primary window's waitForLocalStart
// (#73822): pool backends spawn from the same venv, so an ungated respawn
// during applyUpdates' critical section re-locks the venv and trips the
// venv-blocker preflight. No boot-progress UI here — pool backends boot
// silently for background profiles — so we only log while parked.
{
let poolAnnounced = false
await waitForUpdateClearance(updateGateDeps(), {
onWaitTick: reason => {
if (!poolAnnounced) {
poolAnnounced = true
rememberLog(`[updates] update in progress (${reason}); deferring pool backend start for profile "${profile}"`)
}
},
pollMs: UPDATE_WAIT_POLL_MS,
timeoutMs: UPDATE_WAIT_TIMEOUT_MS
})
}
// --profile wins over the inherited HERMES_HOME env (see _apply_profile_override
// step 3 in hermes_cli/main.py), so the child re-homes to this profile.
// --port 0: the OS assigns an ephemeral port; the child announces it on stdout.
@@ -10860,6 +10950,31 @@ ipcMain.handle('hermes:fs:openDir', async (_event, dirPath) => {
}
})
// The LOCAL Desktop runtime-plugin root: `<HERMES_HOME>/desktop-plugins`,
// resolved from the main-process HERMES_HOME (see resolveHermesHome) — NOT from
// the connected backend. A remote backend reports its own `hermes_home` over
// the gateway, which is a path on the REMOTE box; deriving the plugin dir from
// it yields `undefined/desktop-plugins` (or a non-existent remote path) and the
// on-disk plugin door silently breaks (#66899). Electron owns this resolution
// so it stays valid in every connection mode. Created on demand, like openDir.
ipcMain.handle('hermes:fs:desktopPluginsRoot', async () => {
// Profile-aware: a named Desktop profile gets its own plugin root under
// profiles/<name>/, matching the profile-scoped hermes_home the backend
// reported before this resolver existed. 'default'/unset pins the global root.
const profile = readActiveDesktopProfile()
const base = profile && profile !== 'default' ? path.join(HERMES_HOME, 'profiles', profile) : HERMES_HOME
const dir = path.join(base, 'desktop-plugins')
try {
await fs.promises.mkdir(dir, { recursive: true })
} catch {
// Best-effort create; return the path regardless so the reveal action can
// still surface a real openPath error and the scanner can retry later.
}
return dir
})
// Rename a file/folder in place. The renderer passes the existing path + a new
// base name; the destination is resolved in the SAME parent dir so a rename can
// never move the item elsewhere or traverse out. Rejects on a name collision.
+7
View File
@@ -155,6 +155,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
gitRoot: startPath => ipcRenderer.invoke('hermes:fs:gitRoot', startPath),
revealPath: targetPath => ipcRenderer.invoke('hermes:fs:reveal', targetPath),
openDir: dirPath => ipcRenderer.invoke('hermes:fs:openDir', dirPath),
desktopPluginsRoot: () => ipcRenderer.invoke('hermes:fs:desktopPluginsRoot'),
renamePath: (targetPath, newName) => ipcRenderer.invoke('hermes:fs:rename', targetPath, newName),
writeTextFile: (filePath, content) => ipcRenderer.invoke('hermes:fs:writeText', filePath, content),
trashPath: targetPath => ipcRenderer.invoke('hermes:fs:trash', targetPath),
@@ -211,6 +212,12 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
return () => ipcRenderer.removeListener('hermes:close-preview-requested', listener)
},
onOpenFolderRequested: callback => {
const listener = () => callback()
ipcRenderer.on('hermes:open-folder-requested', listener)
return () => ipcRenderer.removeListener('hermes:open-folder-requested', listener)
},
onOpenUpdatesRequested: callback => {
const listener = () => callback()
ipcRenderer.on('hermes:open-updates', listener)
+144
View File
@@ -0,0 +1,144 @@
/**
* Tests for electron/update-gate.ts — the update mutual-exclusion gate that
* parks local backend spawns while an in-app update is running.
*
* The regression this guards (#73822): applyUpdates kills its own backend
* BEFORE the Windows venv-blocker scan but writes the on-disk marker AFTER
* it. A marker-only gate therefore let the renderer's reconnect spawn a
* fresh backend inside the update's own critical section, which the scan
* reported as a blocker — aborting every Desktop update attempt on Windows.
* The gate must consult the in-process updateInFlight flag as well.
*/
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { updateGateReason, waitForUpdateClearance } from './update-gate'
function deps(marker: boolean, inFlight: boolean) {
return {
hasLiveMarker: () => marker,
isUpdateInFlight: () => inFlight
}
}
// ---------------------------------------------------------------------------
// updateGateReason
// ---------------------------------------------------------------------------
test('gate open when neither marker nor flag is set', () => {
assert.equal(updateGateReason(deps(false, false)), null)
})
test('marker alone closes the gate', () => {
assert.equal(updateGateReason(deps(true, false)), 'marker')
})
test('updateInFlight alone closes the gate (#73822 — the pre-marker window)', () => {
assert.equal(updateGateReason(deps(false, true)), 'update-in-flight')
})
test('marker wins as the reported reason when both are set', () => {
assert.equal(updateGateReason(deps(true, true)), 'marker')
})
// ---------------------------------------------------------------------------
// waitForUpdateClearance
// ---------------------------------------------------------------------------
test('returns clear immediately without sleeping when the gate is open', async () => {
let slept = 0
const outcome = await waitForUpdateClearance(deps(false, false), {
pollMs: 10,
sleep: async () => {
slept += 1
},
timeoutMs: 1000
})
assert.equal(outcome, 'clear')
assert.equal(slept, 0)
})
test('parks on the in-flight flag and finishes when it clears', async () => {
// Simulates the #73822 sequence: the reconnect arrives while updateInFlight
// is true and no marker exists yet; the flag clears (abort path finally)
// and the waiter proceeds.
let inFlight = true
let ticks = 0
const outcome = await waitForUpdateClearance(
{ hasLiveMarker: () => false, isUpdateInFlight: () => inFlight },
{
onWaitTick: reason => {
ticks += 1
assert.equal(reason, 'update-in-flight')
if (ticks >= 3) {
inFlight = false
}
},
pollMs: 1,
sleep: async () => {},
timeoutMs: 10_000
}
)
assert.equal(outcome, 'finished')
assert.equal(ticks, 3)
})
test('parks across the flag→marker handoff without a gap', async () => {
// Success path: the marker is written (main.ts:2936) BEFORE applyUpdates'
// finally clears the flag, so a waiter that arrived during the scan stays
// parked through the transition instead of slipping through.
let inFlight = true
let marker = false
let ticks = 0
const reasons: string[] = []
const outcome = await waitForUpdateClearance(
{ hasLiveMarker: () => marker, isUpdateInFlight: () => inFlight },
{
onWaitTick: reason => {
ticks += 1
reasons.push(reason)
if (ticks === 2) {
marker = true // updater hand-off: marker written first…
}
if (ticks === 3) {
inFlight = false // …then the flag clears; marker still holds the gate
}
if (ticks === 5) {
marker = false // updater finished
}
},
pollMs: 1,
sleep: async () => {},
timeoutMs: 10_000
}
)
assert.equal(outcome, 'finished')
assert.deepEqual(reasons, ['update-in-flight', 'update-in-flight', 'marker', 'marker', 'marker'])
})
test('returns timeout when the gate never opens', async () => {
let clock = 0
const outcome = await waitForUpdateClearance(deps(true, false), {
now: () => clock,
pollMs: 10,
sleep: async ms => {
clock += ms
},
timeoutMs: 50
})
assert.equal(outcome, 'timeout')
})
+95
View File
@@ -0,0 +1,95 @@
'use strict'
/**
* update-gate.ts
*
* Pure, dependency-injected gate that parks local backend spawns while an
* in-app update is running (#73822, #50238).
*
* Two independent signals mean "an update owns the venv right now":
*
* - the on-disk marker (`HERMES_HOME/.hermes-update-in-progress`), written
* by the updater — and by the desktop itself just before hand-off — and
* - the in-process `updateInFlight` flag, true for the whole
* `applyUpdates()` critical section.
*
* The marker alone is NOT enough (#73822): `applyUpdates` kills its own
* backend early (`releaseBackendLock`) but only writes the marker AFTER the
* Windows venv-blocker scan. Killing the backend drops the renderer's
* WebSocket, the renderer reconnects within ~1s, and a marker-only gate
* happily spawns a fresh backend inside the update's own critical section —
* which `scanVenvBlockers` then reports as a blocker, aborting every update
* attempt forever. Consulting the flag closes that window. On the success
* path the marker is written BEFORE the flag clears in `applyUpdates`'
* `finally`, so there is no instant where both signals are false and a
* waiter could slip through mid-update.
*/
export type UpdateGateReason = 'marker' | 'update-in-flight' | null
export interface UpdateGateDeps {
/** True when a live on-disk update marker exists (see update-marker.ts). */
hasLiveMarker: () => boolean
/** True while this process is inside applyUpdates()' critical section. */
isUpdateInFlight: () => boolean
}
/** Why the gate is closed right now, or null when it is open. */
export function updateGateReason(deps: UpdateGateDeps): UpdateGateReason {
if (deps.hasLiveMarker()) {
return 'marker'
}
if (deps.isUpdateInFlight()) {
return 'update-in-flight'
}
return null
}
export type UpdateClearanceOutcome = 'clear' | 'finished' | 'timeout'
export interface WaitForUpdateClearanceOptions {
timeoutMs: number
pollMs: number
/** Invoked once per poll while parked (boot progress / logging). */
onWaitTick?: (reason: Exclude<UpdateGateReason, null>) => void | Promise<void>
now?: () => number
sleep?: (ms: number) => Promise<void>
}
/**
* Park until no update signal remains, or the deadline passes.
*
* Returns 'clear' when the gate was already open (no wait happened),
* 'finished' when it opened during the wait, and 'timeout' when the deadline
* expired with the gate still closed (callers proceed anyway — matching the
* long-standing marker-gate behavior, since a wedged updater must not brick
* the app forever).
*/
export async function waitForUpdateClearance(
deps: UpdateGateDeps,
options: WaitForUpdateClearanceOptions
): Promise<UpdateClearanceOutcome> {
const now = options.now || Date.now
const sleep = options.sleep || (ms => new Promise<void>(r => setTimeout(r, ms)))
let reason = updateGateReason(deps)
if (!reason) {
return 'clear'
}
const deadline = now() + options.timeoutMs
while (reason && now() < deadline) {
if (options.onWaitTick) {
await options.onWaitTick(reason)
}
await sleep(options.pollMs)
reason = updateGateReason(deps)
}
return reason ? 'timeout' : 'finished'
}
+2
View File
@@ -101,7 +101,9 @@
"d3-force": "^3.0.0",
"dnd-core": "^14.0.1",
"dompurify": "^3.4.11",
"emojibase-data": "^16.0.3",
"fflate": "^0.8.3",
"frimousse": "^0.3.0",
"hast-util-from-html-isomorphic": "^2.0.0",
"hast-util-to-text": "^4.0.2",
"ignore": "^7.0.5",
+1
View File
@@ -57,6 +57,7 @@ directly via `window.__PERF_DRIVE__`, so no LLM credits are spent.
| `first-token` | backend | Enter → first assistant token painted (TTFT) | (new) |
| `submit` | backend | Enter → cleared → user msg painted, scroll jump | measure-submit, measure-jump |
| `session-switch` | backend | route → first-paint → settle | profile-session-switch |
| `session-load` | backend | how far a session's transcript moves after first paint | (new) |
| `profile-switch` | backend | rail click → sidebar settled | measure-profile-switch |
`ci` + `cold` scenarios need no backend/credits and are gated against
@@ -8,6 +8,7 @@ import keystroke from './keystroke.mjs'
import multitab from './multitab.mjs'
import profileSwitch from './profile-switch.mjs'
import renderChurn from './render-churn.mjs'
import sessionLoad from './session-load.mjs'
import sessionSwitch from './session-switch.mjs'
import stream from './stream.mjs'
import streamHistory from './stream-history.mjs'
@@ -25,6 +26,7 @@ export const SCENARIOS = {
[coldStart.name]: coldStart,
[firstToken.name]: firstToken,
[submit.name]: submit,
[sessionLoad.name]: sessionLoad,
[sessionSwitch.name]: sessionSwitch,
[profileSwitch.name]: profileSwitch
}
@@ -0,0 +1,136 @@
// Visual stability of a session LOAD. Sibling to `submit` (which measures the
// jump on Enter); this one measures the jump on opening a session — the
// prepend/settle path, not the append path.
//
// Clicks sidebar rows and tracks the bottom-most turn's on-screen top every
// frame. A clean load never moves it after first paint; a janky one strands it
// thousands of px away while the render-budget backfill and stick-to-bottom
// argue. Backend tier: needs real stored sessions in the sidebar.
//
// node scripts/perf/run.mjs session-load --rows 2,5,8 --rounds 2
import { SELECTORS, sleep } from '../lib/cdp.mjs'
import { summarize } from '../lib/stats.mjs'
// Below this a shift is sub-perceptual (sub-pixel rounding, a settling caret).
const SHIFT_PX = 4
const ARM = `
(() => {
const samples = []
const t0 = performance.now()
let running = true
const tick = () => {
if (!running) return
const v = document.querySelector(${JSON.stringify(SELECTORS.threadViewport)})
if (v) {
const turns = v.querySelectorAll(
${JSON.stringify(SELECTORS.turnPair)} + ',' + ${JSON.stringify(SELECTORS.assistantMessage)}
)
const last = turns[turns.length - 1]
const rect = last && last.getBoundingClientRect()
samples.push({
st: Math.round(v.scrollTop),
sh: v.scrollHeight,
ch: v.clientHeight,
bottomTop: rect ? Math.round(rect.top - v.getBoundingClientRect().top) : null,
turns: turns.length,
t: Math.round(performance.now() - t0)
})
}
requestAnimationFrame(tick)
}
requestAnimationFrame(tick)
window.__SL = { samples, stop() { running = false } }
})()
`
const CLICK = index => `
(() => {
const rows = [...document.querySelectorAll(${JSON.stringify(SELECTORS.rowButton)})].filter(el => el.offsetParent)
const row = rows[${index}]
if (!row) return null
row.click()
return (row.textContent ?? '').slice(0, 34)
})()
`
/** Total/max on-screen movement of the bottom turn after it first paints. */
function measureLoad(samples) {
const painted = samples.findIndex(s => s.turns > 0)
const after = painted === -1 ? [] : samples.slice(painted)
const load = { maxShiftPx: 0, offBottomFrames: 0, settledMs: 0, shiftedPx: 0, shifts: 0 }
let previous = null
for (const sample of after) {
if (sample.sh - (sample.st + sample.ch) > 2) {
load.offBottomFrames += 1
}
if (previous?.bottomTop != null && sample.bottomTop != null) {
const delta = Math.abs(sample.bottomTop - previous.bottomTop)
if (delta > SHIFT_PX) {
load.maxShiftPx = Math.max(load.maxShiftPx, delta)
load.settledMs = sample.t
load.shiftedPx += delta
load.shifts += 1
}
}
previous = sample
}
return load
}
export default {
name: 'session-load',
tier: 'backend',
description: 'Visual stability of opening a session: how far the transcript moves after first paint.',
async run(cdp, opts = {}) {
const rows = String(opts.rows ?? '2,5,8').split(',').map(Number)
const rounds = Number(opts.rounds ?? 2)
const watchMs = Number(opts.watchMs ?? 4500)
await cdp.send('Runtime.enable')
const loads = []
for (let round = 0; round < rounds; round++) {
for (const index of rows) {
await cdp.eval(ARM)
if (!(await cdp.eval(CLICK(index)))) {
continue
}
await sleep(watchMs)
const { samples } = await cdp.eval('(() => { window.__SL.stop(); return window.__SL })()')
loads.push(measureLoad(samples))
}
await sleep(500)
}
if (!loads.length) {
throw new Error('session-load found no sidebar rows to click')
}
const total = key => loads.reduce((sum, load) => sum + load[key], 0)
return {
metrics: {
load_max_shift_px: Math.max(...loads.map(load => load.maxShiftPx)),
load_off_bottom_frames: Math.round(total('offBottomFrames') / loads.length),
load_settled_p95_ms: summarize(loads.map(load => load.settledMs)).p95,
load_shifted_px: Math.round(total('shiftedPx') / loads.length)
},
detail: { loads: loads.length, shiftsPerLoad: total('shifts') / loads.length }
}
}
}
@@ -0,0 +1,146 @@
// ⌘K open latency, measured in-page (no CDP round-trip in the number).
//
// node scripts/probe-command-palette.mjs [--port 9222] [--rounds 8]
//
// Reports, per round, the time from the keydown the app actually receives to:
// frame_ms — the dialog frame + input in the DOM and painted (what "instant"
// means: the overlay owes you a frame immediately)
// rows_ms — the row list painted (may lag frame_ms; rows are deferred)
// plus any long tasks in the window, so a slow open is attributable.
import { CDP, sleep } from './perf/lib/cdp.mjs'
const args = process.argv.slice(2)
const flag = name => {
const i = args.indexOf(`--${name}`)
return i >= 0 ? args[i + 1] : undefined
}
const port = Number(flag('port') ?? 9222)
const rounds = Number(flag('rounds') ?? 8)
const cdp = await CDP.connect({ port })
await cdp.send('Runtime.enable')
const INSTALL = `
(() => {
if (window.__CMDK__) window.__CMDK__.stop()
const state = { t0: null, frame: null, rows: 0, rowsAt: null, tasks: [], armed: false }
// Time from the keydown the APP receives — excludes CDP transport, so the
// number is what a user's finger actually experiences.
const onKey = e => {
if (state.armed && (e.metaKey || e.ctrlKey) && e.key.toLowerCase() === 'k') {
state.t0 = performance.now()
state.armed = false
}
}
window.addEventListener('keydown', onKey, true)
const obs = new MutationObserver(() => {
if (state.t0 === null) return
if (state.frame === null && document.querySelector('[cmdk-input]')) {
state.frame = performance.now() - state.t0
}
const n = document.querySelectorAll('[cmdk-item]').length
if (n > state.rows) { state.rows = n; state.rowsAt = performance.now() - state.t0 }
})
obs.observe(document.body, { childList: true, subtree: true })
const po = new PerformanceObserver(list => {
for (const e of list.getEntries()) state.tasks.push({ start: e.startTime, dur: Math.round(e.duration) })
})
try { po.observe({ entryTypes: ['longtask'] }) } catch {}
window.__CMDK__ = {
arm: () => { state.t0 = null; state.frame = null; state.rows = 0; state.rowsAt = null; state.tasks = []; state.armed = true },
read: () => ({
frame_ms: state.frame === null ? -1 : Math.round(state.frame),
rows_ms: state.rowsAt === null ? -1 : Math.round(state.rowsAt),
rows: state.rows,
longtask_ms: state.t0 === null ? 0 : state.tasks.filter(t => t.start >= state.t0).reduce((s, t) => s + t.dur, 0)
}),
stop: () => { window.removeEventListener('keydown', onKey, true); obs.disconnect(); po.disconnect() }
}
return true
})()
`
// Settle: frame painted AND rows stopped growing for two frames.
const WAIT = `
new Promise(resolve => {
let stable = 0
let last = -1
const started = performance.now()
const tick = () => {
const r = window.__CMDK__.read()
if (r.frame_ms >= 0 && r.rows === last && r.rows > 0) {
if (++stable >= 2) { resolve(r); return }
} else { stable = 0 }
last = r.rows
if (performance.now() - started > 8000) { resolve(window.__CMDK__.read()); return }
requestAnimationFrame(tick)
}
requestAnimationFrame(tick)
})
`
const key = async type =>
cdp.send('Input.dispatchKeyEvent', {
type,
key: 'k',
code: 'KeyK',
windowsVirtualKeyCode: 75,
nativeVirtualKeyCode: 75,
modifiers: 4
})
const esc = async () => {
for (const type of ['keyDown', 'keyUp']) {
await cdp.send('Input.dispatchKeyEvent', { type, key: 'Escape', code: 'Escape', windowsVirtualKeyCode: 27 })
}
await sleep(400)
}
await cdp.eval(INSTALL)
await esc()
const samples = []
for (let i = 0; i < rounds; i++) {
await sleep(250)
await cdp.eval('window.__CMDK__.arm()')
await key('rawKeyDown')
await key('keyUp')
const r = await cdp.eval(WAIT)
samples.push(r)
console.log(`round ${i}:`, r)
await esc()
}
await cdp.eval('window.__CMDK__.stop()')
const stat = k => {
const v = samples.map(s => s[k]).filter(n => n >= 0).sort((a, b) => a - b)
if (!v.length) return null
return {
min: v[0],
median: v[Math.floor(v.length / 2)],
max: v[v.length - 1]
}
}
console.log('\nkeydown → dialog frame painted (ms):', stat('frame_ms'))
console.log('keydown → rows painted (ms):', stat('rows_ms'))
console.log('long-task time in window (ms):', stat('longtask_ms'))
cdp.close()
+99 -13
View File
@@ -1,9 +1,30 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const closeFocusedSessionTab = vi.fn(() => false)
const nextSessionTileForWorkspace = vi.fn<() => null | string>(() => null)
const closeSessionTile = vi.fn()
const requestFreshSession = vi.fn()
vi.mock('@/components/pane-shell/tree/store', () => ({
closeFocusedSessionTab: () => closeFocusedSessionTab()
}))
vi.mock('@/store/session-states', () => ({
closeSessionTile: (...args: unknown[]) => closeSessionTile(...args),
nextSessionTileForWorkspace: () => nextSessionTileForWorkspace()
}))
vi.mock('@/store/profile', () => ({
requestFreshSession: () => requestFreshSession()
}))
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs, closeRightRail, openPreview, type PreviewTarget } from '@/store/preview'
import { $activeSessionId, $selectedStoredSessionId } from '@/store/session'
import { closeActiveTab } from './close-tab'
import { $workspaceIsPage } from '../routes'
import { closeActiveTab, closeWorkspaceTab } from './close-tab'
function fileTarget(path: string): PreviewTarget {
return {
@@ -16,19 +37,31 @@ function fileTarget(path: string): PreviewTarget {
}
}
/** Main is holding a loaded chat and nothing else is stacked with it. */
function loadedMainOnly() {
$selectedStoredSessionId.set('stored-a')
$activeSessionId.set('runtime-a')
}
beforeEach(() => {
vi.stubGlobal('document', { activeElement: null })
closeRightRail()
window.localStorage.clear()
$selectedStoredSessionId.set(null)
$activeSessionId.set(null)
$workspaceIsPage.set(false)
closeFocusedSessionTab.mockReturnValue(false)
nextSessionTileForWorkspace.mockReturnValue(null)
vi.clearAllMocks()
})
afterEach(() => {
vi.unstubAllGlobals()
closeRightRail()
window.localStorage.clear()
})
describe('closeActiveTab', () => {
beforeEach(() => {
vi.stubGlobal('document', { activeElement: null })
closeRightRail()
window.localStorage.clear()
})
afterEach(() => {
vi.unstubAllGlobals()
closeRightRail()
window.localStorage.clear()
})
it('closes the active file preview tab (⌘W happy path)', () => {
openPreview(fileTarget('/work/notes.md'), 'manual')
@@ -50,3 +83,56 @@ describe('closeActiveTab', () => {
expect($previewTabs.get()).toHaveLength(0)
})
})
/**
* The main tab's own close. The workspace pane can never leave the tree, so
* every answer here is about what FILLS it — a stacked session, or an empty
* draft. The gesture used to dead-end whenever main was the only tab.
*/
describe('closeWorkspaceTab', () => {
it('shifts the next stacked session into main', () => {
loadedMainOnly()
nextSessionTileForWorkspace.mockReturnValue('stored-b')
const load = vi.fn()
expect(closeWorkspaceTab(load)).toBe(true)
expect(closeSessionTile).toHaveBeenCalledWith('stored-b')
expect(load).toHaveBeenCalledWith('stored-b')
// Promotion refills main — it must not ALSO blank it.
expect(requestFreshSession).not.toHaveBeenCalled()
})
it('drops a lone loaded main to a fresh draft', () => {
loadedMainOnly()
expect(closeWorkspaceTab(vi.fn())).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
it('empties main even with no session loader wired', () => {
loadedMainOnly()
expect(closeWorkspaceTab()).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
it('is a no-op on a blank draft — that IS the post-close state', () => {
expect(closeWorkspaceTab(vi.fn())).toBe(false)
expect(requestFreshSession).not.toHaveBeenCalled()
})
it('is a no-op over a full-page view, which owns no chat tab', () => {
loadedMainOnly()
$workspaceIsPage.set(true)
expect(closeWorkspaceTab(vi.fn())).toBe(false)
expect(requestFreshSession).not.toHaveBeenCalled()
})
it('⌘W reaches it once the terminal, rail and zone tabs pass', () => {
loadedMainOnly()
expect(closeActiveTab(vi.fn())).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
})
+51 -26
View File
@@ -1,27 +1,68 @@
import { mainChatOccupied } from '@/app/open-session'
import { closeActiveTerminal } from '@/app/right-sidebar/terminal/terminals'
import { $workspaceIsPage } from '@/app/routes'
import { closeFocusedSessionTab } from '@/components/pane-shell/tree/store'
import { isFocusWithin } from '@/lib/keybinds/combo'
import { $previewTabs, closeActiveRightRailTab } from '@/store/preview'
import { requestFreshSession } from '@/store/profile'
import { $activeSessionId, $selectedStoredSessionId } from '@/store/session'
import { closeSessionTile, nextSessionTileForWorkspace } from '@/store/session-states'
/**
* Close the MAIN tab. The workspace pane itself can't leave the tree, so
* "closing" it means emptying it, and what fills the hole depends on what's
* stacked beside it:
*
* - session tabs stacked with it → the next one shifts INTO main (drop its
* tile, load it as the primary — the session stays alive, no busy prompt),
* - nothing stacked → main drops to a fresh "New session" draft.
*
* The second half is what makes the gesture honest when main is the ONLY tab:
* ⌘W / ⌘-click / middle-click used to be a dead key there, since the only
* available answer was "remove the pane", which this app never does.
*
* Returns false when there is nothing to close — a blank draft (already the
* post-close state) or a full-page view (skills / artifacts, which isn't a
* chat and owns no tab). ⌘W then stays a no-op; it never closes the window.
*
* `loadSessionIntoWorkspace` carries the app's route-based "load this session
* into main"; omitting it disables the promotion half.
*/
export function closeWorkspaceTab(loadSessionIntoWorkspace?: (storedSessionId: string) => void): boolean {
// Order matters — close the tile FIRST so the selection homes to the
// workspace instead of re-fronting the tile.
if (loadSessionIntoWorkspace) {
const next = nextSessionTileForWorkspace()
if (next) {
closeSessionTile(next)
loadSessionIntoWorkspace(next)
return true
}
}
if ($workspaceIsPage.get() || !mainChatOccupied($activeSessionId.get(), $selectedStoredSessionId.get())) {
return false
}
requestFreshSession()
return true
}
/**
* ⌘W — close the tab of the context you're in, by precedence:
* 1. a focused terminal → its active terminal tab,
* 2. right-rail tabs (live preview and/or file peeks),
* 3. the FOCUSED chat zone → its active tab (a session tile stacked into it).
* 4. the workspace tab itself, when session tabs are stacked with it:
* the workspace can't close, so ⌘W shifts the NEXT session tab into main
* (loads it as the primary + drops its now-redundant tile).
* 4. the workspace tab itself — see `closeWorkspaceTab`.
* Returns false when nothing closes, so ⌘W is a no-op — it never closes the
* window (a bare workspace stays put). Shared by the keyboard path (Win/Linux)
* and the macOS menu-accelerator IPC.
* window. Shared by the keyboard path (Win/Linux) and the macOS
* menu-accelerator IPC.
*
* Steps 3-4 follow the same focused zone ⌘1…⌘9 indexes, so a second chat zone
* with its own tab strip closes ITS tab instead of main's.
*
* `loadSessionIntoWorkspace` carries the app's route-based "load this session
* into main" (the two call sites have router access); omitting it disables the
* step-4 promotion (⌘W stays the pre-existing no-op on the main tab).
*/
export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: string) => void): boolean {
if (isFocusWithin('[data-terminal]')) {
@@ -43,21 +84,5 @@ export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: stri
return true
}
// The main (workspace) tab is active and can't be closed — but if session
// tabs are stacked with it, ⌘W shifts the next one into the main tab: drop
// its tile (the session stays alive, no busy-close prompt) and load it into
// main. Order matters — close the tile FIRST so the selection homes to the
// workspace instead of re-fronting the tile.
if (loadSessionIntoWorkspace) {
const next = nextSessionTileForWorkspace()
if (next) {
closeSessionTile(next)
loadSessionIntoWorkspace(next)
return true
}
}
return false
return closeWorkspaceTab(loadSessionIntoWorkspace)
}
@@ -0,0 +1,131 @@
import type { Unstable_TriggerItem } from '@assistant-ui/core'
import { act, renderHook } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { hermesDirectiveFormatter } from '@/components/assistant-ui/directive-text'
import { classify } from './hooks/use-at-completions'
import { useComposerTrigger } from './hooks/use-composer-trigger'
import { composerPlainText, RICH_INPUT_SLOT } from './rich-editor'
/** A row exactly as tui_gateway's complete.path emits it, run through the
* real classify() the popover uses. */
function backendRow(text: string, display: string, meta: string): Unstable_TriggerItem {
const c = classify({ text, display, meta })
return {
id: `${text}|0`,
type: c.type,
label: c.display,
metadata: { icon: c.type, display: c.display, meta: c.meta, rawText: text, insertId: c.insertId }
}
}
function typed(text: string) {
const editor = document.createElement('div')
editor.contentEditable = 'true'
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
editor.append(document.createTextNode(text))
const range = document.createRange()
range.selectNodeContents(editor)
range.collapse(false)
const sel = window.getSelection()
sel?.removeAllRanges()
sel?.addRange(range)
const editorRef = { current: editor as HTMLDivElement | null }
const { result } = renderHook(() =>
useComposerTrigger({
at: { adapter: null, loading: false },
draftRef: { current: text },
editorRef,
requestMainFocus: vi.fn(),
setComposerText: vi.fn(),
slash: { adapter: null, loading: false }
})
)
act(() => result.current.refreshTrigger())
return { editor, result }
}
/** The label the sent message renders for a committed draft. */
function sentLabel(draft: string) {
return hermesDirectiveFormatter
.parse(draft)
.filter((s): s is Extract<typeof s, { kind: 'mention' }> => s.kind === 'mention')
.map(s => s.label)
.join(',')
}
describe('one label per reference, on every surface', () => {
it('the popover row, the committed chip, and the sent chip all read the same', () => {
const cases = [
{ text: '@folder:apps/desktop/', display: 'desktop/', meta: 'dir' },
{ text: '@file:apps/desktop/src/main.tsx', display: 'main.tsx', meta: 'apps/desktop/src' },
{ text: '@folder:apps/desktop/src/', display: 'src/', meta: 'dir' }
]
for (const entry of cases) {
const item = backendRow(entry.text, entry.display, entry.meta)
const { editor, result } = typed('@desk')
act(() => result.current.replaceTriggerWithChip(item))
const row = String((item.metadata as { display: string }).display)
const chip = editor.querySelector('[data-ref-text]')?.textContent ?? ''
expect(chip).toBe(row)
expect(sentLabel(composerPlainText(editor))).toBe(row)
}
})
it('a folder pick reads as its path, not a bare basename', () => {
// `src` and `desktop` repeat all over a repo — the row you picked said
// where it was, and the chip has to keep saying it.
const item = backendRow('@folder:apps/desktop/', 'desktop/', 'dir')
expect(item.label).toBe('apps/desktop/')
const { editor, result } = typed('@desk')
act(() => result.current.replaceTriggerWithChip(item))
expect(editor.querySelector('[data-ref-text]')?.textContent).toBe('apps/desktop/')
})
it('Tab-descend leaves the live query, and the scope when there is one', () => {
const { editor, result } = typed('@folder:desk')
act(() =>
result.current.replaceTriggerWithChip(backendRow('@folder:apps/desktop/', 'desktop/', 'dir'), {
descend: true
})
)
// Mid-browse the editor holds the live query, scope included — that's the
// path being typed, not a label, and it's what the next completion reads.
expect(composerPlainText(editor)).toBe('@folder:apps/desktop/')
})
it('a url still reads host + path on every surface', () => {
const item = backendRow('@url:https://github.com/NousResearch/hermes-agent/pull/74533', '', '')
const { editor, result } = typed('@gith')
act(() => result.current.replaceTriggerWithChip(item))
const expected = 'github.com/NousResearch/hermes-agent/pull/74533'
expect(item.label).toBe(expected)
expect(editor.querySelector('[data-ref-text]')?.textContent).toBe(expected)
expect(sentLabel(composerPlainText(editor))).toBe(expected)
})
})
@@ -0,0 +1,164 @@
import type { Unstable_TriggerItem } from '@assistant-ui/core'
import { act, renderHook } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { useComposerTrigger } from './hooks/use-composer-trigger'
import { pathifyRefs } from './path-refs'
import { composerPlainText, insertComposerContentsAtCaret, RICH_INPUT_SLOT } from './rich-editor'
import { detectTrigger, openDirectiveScope, textBeforeCaret } from './text-utils'
import { linkifyUrls } from './url-refs'
function folderItem(rel: string): Unstable_TriggerItem {
const rawText = `@folder:${rel}/`
return {
id: `${rawText}|0`,
type: 'folder',
label: rel.split('/').filter(Boolean).pop() ?? rel,
metadata: { icon: 'folder', display: `${rel}/`, meta: 'dir', rawText, insertId: `${rel}/` }
}
}
/** Literally-typed text, caret `fromEnd` characters before the end. */
function typed(text: string, fromEnd = 0) {
const editor = document.createElement('div')
editor.contentEditable = 'true'
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
const node = document.createTextNode(text)
editor.append(node)
const range = document.createRange()
range.setStart(node, text.length - fromEnd)
range.collapse(true)
const sel = window.getSelection()
sel?.removeAllRanges()
sel?.addRange(range)
return editor
}
function withTrigger(editor: HTMLDivElement, draft: string) {
const editorRef = { current: editor as HTMLDivElement | null }
const { result } = renderHook(() =>
useComposerTrigger({
at: { adapter: null, loading: false },
draftRef: { current: draft },
editorRef,
requestMainFocus: vi.fn(),
setComposerText: vi.fn(),
slash: { adapter: null, loading: false }
})
)
act(() => result.current.refreshTrigger())
return result
}
/** The composer's paste handler, minus the clipboard plumbing. */
function paste(editor: HTMLDivElement, text: string) {
insertComposerContentsAtCaret(editor, pathifyRefs(linkifyUrls(text)), openDirectiveScope(editor))
}
describe('directive scope is a browse mode, not text to maintain', () => {
it('Tab-descend carries the scope down instead of dropping to a bare path', () => {
const editor = typed('@folder:apps/deskt')
const result = withTrigger(editor, '@folder:apps/deskt')
expect(result.current.trigger).toMatchObject({ kind: '@', scope: 'folder', value: 'apps/deskt' })
act(() => result.current.replaceTriggerWithChip(folderItem('apps/desktop'), { descend: true }))
expect(composerPlainText(editor)).toBe('@folder:apps/desktop/')
})
it('Backspace climbs the path, then drops the whole scope', () => {
const editor = typed('@folder:apps/desktop/')
const result = withTrigger(editor, '@folder:apps/desktop/')
act(() => result.current.ascendTriggerPath())
expect(composerPlainText(editor)).toBe('@folder:apps/')
act(() => result.current.refreshTrigger())
act(() => result.current.ascendTriggerPath())
expect(composerPlainText(editor)).toBe('@folder:')
// The scope is one unit: Backspace drops it whole rather than nibbling
// back through `:`, `r`, `e`, `d`, `l`, `o`, `f`.
act(() => result.current.refreshTrigger())
act(() => result.current.ascendTriggerPath())
expect(composerPlainText(editor)).toBe('@')
})
it('leaves Backspace alone when there is no scope and no path', () => {
const editor = typed('@apps')
const result = withTrigger(editor, '@apps')
let handled = true
act(() => {
handled = result.current.ascendTriggerPath()
})
expect(handled).toBe(false)
})
it('a pick mid-message keeps the trailing prose and consumes the whole token', () => {
const editor = typed('@folder:apps/deskt and some trailing words', 24)
const result = withTrigger(editor, '@folder:apps/deskt and some trailing words')
act(() => result.current.replaceTriggerWithChip(folderItem('apps/desktop')))
expect(composerPlainText(editor)).toBe('@folder:`apps/desktop/` and some trailing words')
expect(editor.querySelector('[data-ref-kind="folder"]')).not.toBeNull()
})
it('pasting into an open @url: scope consumes it instead of stacking', () => {
const editor = typed('refer to @url:')
paste(editor, 'https://github.com/NousResearch/hermes-agent/pull/74533')
expect(composerPlainText(editor)).toBe('refer to @url:`https://github.com/NousResearch/hermes-agent/pull/74533`')
expect(editor.textContent).not.toContain('@url:@url:')
})
it('a normal paste with no open scope is untouched', () => {
const editor = typed('look at ')
paste(editor, 'https://example.com/x')
expect(composerPlainText(editor)).toBe('look at @url:`https://example.com/x`')
})
it('scope parsing leaves an unscoped @ query alone', () => {
expect(detectTrigger('@apps/desk')).toMatchObject({ kind: '@', value: 'apps/desk' })
expect(detectTrigger('@apps/desk')?.scope).toBeUndefined()
})
it('openDirectiveScope only fires on an EMPTY scope', () => {
// The count is what a paste consumes: `@url:` is 5 characters of syntax
// the user never typed and shouldn't be left holding.
expect(openDirectiveScope(typed('@url:'))).toBe(5)
expect(openDirectiveScope(typed('@url:https://x.com'))).toBe(0)
expect(openDirectiveScope(typed('plain text'))).toBe(0)
})
it('chips stay atomic to scope detection', () => {
const editor = typed('@folder:apps/desktop/')
const result = withTrigger(editor, '@folder:apps/desktop/')
act(() => result.current.replaceTriggerWithChip(folderItem('apps/desktop')))
// A committed chip is one object-replacement char, so a fresh `@` typed
// after it opens an unscoped browse rather than inheriting the old scope.
expect(detectTrigger(`${textBeforeCaret(editor)}@`)?.scope).toBeUndefined()
})
})
@@ -0,0 +1,139 @@
import { describe, expect, it } from 'vitest'
import {
composerPlainText,
normalizeComposerEditorDom,
renderComposerContents,
RICH_INPUT_SLOT
} from './rich-editor'
function editor(): HTMLDivElement {
const el = document.createElement('div')
el.dataset.slot = RICH_INPUT_SLOT
el.contentEditable = 'true'
document.body.append(el)
return el
}
/** Whatever emptied it — Delete, cut, Chromium's own selection-delete — the
* normalizer lands on the same DOM. */
function emptied(): HTMLDivElement {
const el = editor()
el.append(document.createTextNode('hello'))
el.replaceChildren()
normalizeComposerEditorDom(el)
return el
}
describe('an emptied composer reads as empty', () => {
it('keeps the placeholder <br> so the contenteditable holds its height', () => {
// The scaffolding is deliberate: a childless contenteditable collapses to a
// sliver in Chromium. It just must not read as content.
expect(emptied().innerHTML).toBe('<br>')
})
it('reads that editor as empty, not as a newline', () => {
expect(composerPlainText(emptied())).toBe('')
})
it('reads a truly childless editor as empty', () => {
expect(composerPlainText(editor())).toBe('')
})
it('still reads a real Shift+Enter line break as a newline', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'), document.createTextNode('two'))
expect(composerPlainText(el)).toBe('one\ntwo')
})
it('still reads a trailing break after text as a newline', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'))
expect(composerPlainText(el)).toBe('one\n')
})
it('only treats the EDITOR\u2019s lone <br> as scaffolding, not a nested one', () => {
// A lone <br> inside some other element is a real line break; the exemption
// is scoped to the editor root by its slot marker. (The block wrapper adds
// its own trailing newline — unchanged behavior, asserted so the exemption
// can't quietly widen to nested nodes.)
const el = editor()
const inner = document.createElement('div')
inner.append(document.createElement('br'))
el.append(document.createTextNode('one'), inner)
expect(composerPlainText(el)).toBe('one\n\n')
})
})
/** The rule the stylesheet paints the placeholder with. `:empty` alone goes
* false the instant the scaffolding <br> lands. */
const PLACEHOLDER_SHOWS = ':is(:empty, [data-empty])'
describe('an emptied composer shows its placeholder again', () => {
it('advertises emptiness once the scaffolding break is in place', () => {
expect(emptied().matches(PLACEHOLDER_SHOWS)).toBe(true)
})
it('advertises emptiness for a truly childless editor', () => {
expect(editor().matches(PLACEHOLDER_SHOWS)).toBe(true)
})
it('stops advertising it once something is typed', () => {
const el = emptied()
el.replaceChildren(document.createTextNode('hi'))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
// A text node is invisible to selectors, so `one<br>` and `<br>` are the same
// shape to any pure-CSS rule (`:has(> br:only-child)` matches both and paints
// the placeholder straight over the user's text). The DOM writer has to say.
it('does not advertise emptiness for a trailing break after text', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
it('does not advertise emptiness for a Shift+Enter break between text', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'), document.createTextNode('two'))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
// Repainting from text (restored draft, undo, completion rebuild) is the
// other writer that reshapes the editor root — it must not strand the marker.
it('drops the marker when a draft is painted back in', () => {
const el = emptied()
renderComposerContents(el, 'restored draft')
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
it('re-advertises emptiness when a draft is painted back out', () => {
const el = editor()
renderComposerContents(el, 'temporary')
renderComposerContents(el, '')
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true)
})
})
@@ -1,6 +1,17 @@
import { afterEach, describe, expect, it } from 'vitest'
import { blurComposerInput } from './focus'
import { $hoveredTreeGroup } from '@/components/pane-shell/tree/store'
import {
blurComposerInput,
getActiveComposer,
markActiveComposer,
onComposerFocusRequest,
onComposerModelMenuRequest,
releaseActiveComposer,
requestComposerFocus,
requestModelMenuToggle
} from './focus'
import { RICH_INPUT_SLOT } from './rich-editor'
/**
@@ -21,8 +32,24 @@ function mountInput(hidden = false) {
return input
}
/** A chat surface stamp — the same `data-composer-target` ChatView hangs. */
function mountSurface(target: string, hidden = false) {
const layer = document.createElement('div')
layer.toggleAttribute('data-pane-hidden', hidden)
const surface = document.createElement('div')
surface.dataset.composerTarget = target
layer.append(surface)
document.body.append(layer)
return surface
}
afterEach(() => {
document.body.innerHTML = ''
// `activeTarget` is module-level — a case that leaves a stale claim behind
// would otherwise decide the next one.
markActiveComposer('main')
$hoveredTreeGroup.set(null)
})
describe('blurComposerInput', () => {
@@ -48,3 +75,211 @@ describe('blurComposerInput', () => {
expect(document.activeElement).toBe(outside)
})
})
/**
* `markActiveComposer` has four call sites and, unguarded, no counterpart: an
* unmounting or keep-alive-buried composer left `activeTarget` pointing at
* itself, so every `'active'`-routed request was delivered to a target with no
* on-screen subscriber. Type-to-focus preventDefaults the keystroke BEFORE the
* request, so a dead target swallows the character and focuses nothing.
*/
describe('releaseActiveComposer', () => {
it('falls back to the main composer when the claimant releases', () => {
const root = document.createElement('div')
root.dataset.slot = 'aui_edit-composer-root'
document.body.append(root)
markActiveComposer('edit')
expect(getActiveComposer()).toBe('edit')
root.remove()
releaseActiveComposer('edit')
expect(getActiveComposer()).toBe('main')
})
it('leaves the key with the live claimant when a stale composer releases late', () => {
markActiveComposer('edit')
markActiveComposer('tile:abc')
releaseActiveComposer('edit')
expect(getActiveComposer()).toBe('tile:abc')
})
it('prefers the visible chat surface over a hard main default', () => {
const root = document.createElement('div')
root.dataset.slot = 'aui_edit-composer-root'
document.body.append(root)
mountSurface('tile:visible')
markActiveComposer('edit')
root.remove()
releaseActiveComposer('edit')
expect(getActiveComposer()).toBe('tile:visible')
})
it('routes an active-target request to the main composer once the edit composer closes', async () => {
// Mirrors the per-composer filter in use-composer-draft / user-edit-composer:
// a composer ignores any request not addressed to its own target.
const mainComposerSaw: string[] = []
const off = onComposerFocusRequest(({ target }) => {
if (target === 'main') {
mainComposerSaw.push(target)
}
})
const root = document.createElement('div')
root.dataset.slot = 'aui_edit-composer-root'
document.body.append(root)
markActiveComposer('edit')
root.remove()
releaseActiveComposer('edit')
requestComposerFocus('active')
// `dispatch` defers to a macrotask so click/keydown handlers settle first.
await new Promise(resolve => window.setTimeout(resolve, 0))
off()
expect(mainComposerSaw).toEqual(['main'])
})
})
describe('resolveActive / keep-alive tab heal', () => {
it('heals type-to-focus onto the visible main tab when a tile is buried', async () => {
// Repro for the reported main-tab miss: user typed in a session tile, then
// clicked the main/workspace tab without focusing its input. The tile stays
// mounted under data-pane-hidden, so activeTarget still reads tile:… and
// every type-to-focus request is dropped by the visible main composer.
mountSurface('tile:buried', true)
mountSurface('main')
markActiveComposer('tile:buried')
expect(getActiveComposer()).toBe('main')
const mainSaw: string[] = []
const tileSaw: string[] = []
const off = onComposerFocusRequest(({ target }) => {
if (target === 'main') {
mainSaw.push(target)
}
if (target === 'tile:buried') {
tileSaw.push(target)
}
})
requestComposerFocus('active', { typeChar: 'h' })
await new Promise(resolve => window.setTimeout(resolve, 0))
off()
expect(mainSaw).toEqual(['main'])
expect(tileSaw).toEqual([])
// Cache stays honest so dict/insert/Esc path all agree thereafter.
expect(getActiveComposer()).toBe('main')
})
it('keeps a live tile claim while that tile is the visible surface', () => {
mountSurface('main', true)
mountSurface('tile:front')
markActiveComposer('tile:front')
expect(getActiveComposer()).toBe('tile:front')
})
it('heals an edit claim once the edit root is gone (no release site needed)', async () => {
mountSurface('main')
markActiveComposer('edit')
// No edit root in the document → claim is dead. getActiveComposer heals.
expect(getActiveComposer()).toBe('main')
const mainSaw: string[] = []
const off = onComposerFocusRequest(({ target }) => {
if (target === 'main') {
mainSaw.push(target)
}
})
requestComposerFocus('active', { typeChar: 'a' })
await new Promise(resolve => window.setTimeout(resolve, 0))
off()
expect(mainSaw).toEqual(['main'])
})
it('holds an edit claim while the edit composer root is mounted', () => {
const root = document.createElement('div')
root.dataset.slot = 'aui_edit-composer-root'
document.body.append(root)
mountSurface('main')
markActiveComposer('edit')
expect(getActiveComposer()).toBe('edit')
})
})
/** A chat surface inside a layout zone, mirroring ChatView-in-tree-group. */
function mountZonedSurface(target: string, zone: string, hidden = false) {
const group = document.createElement('div')
group.dataset.treeGroup = zone
const layer = document.createElement('div')
layer.toggleAttribute('data-pane-hidden', hidden)
const surface = document.createElement('div')
surface.dataset.composerTarget = target
layer.append(surface)
group.append(layer)
document.body.append(group)
return surface
}
const collectModelMenuTargets = async (): Promise<string[]> => {
const saw: string[] = []
const off = onComposerModelMenuRequest(target => saw.push(target))
await new Promise(resolve => window.setTimeout(resolve, 0))
off()
return saw
}
describe('requestModelMenuToggle', () => {
it('targets the pane under the pointer over the focused one (#74447 convention)', async () => {
mountZonedSurface('main', 'zone-a')
mountZonedSurface('tile:hovered', 'zone-b')
markActiveComposer('main')
$hoveredTreeGroup.set('zone-b')
expect(requestModelMenuToggle()).toBe(true)
expect(await collectModelMenuTargets()).toEqual(['tile:hovered'])
})
it('falls back to the active composer when the pointer is off every zone', async () => {
mountZonedSurface('main', 'zone-a')
mountZonedSurface('tile:other', 'zone-b')
markActiveComposer('tile:other')
expect(requestModelMenuToggle()).toBe(true)
expect(await collectModelMenuTargets()).toEqual(['tile:other'])
})
it('skips a hidden keep-alive tab in the hovered zone (targets its visible sibling)', async () => {
mountZonedSurface('main', 'zone-a', true)
mountZonedSurface('tile:front', 'zone-a')
markActiveComposer('main')
$hoveredTreeGroup.set('zone-a')
expect(requestModelMenuToggle()).toBe(true)
expect(await collectModelMenuTargets()).toEqual(['tile:front'])
})
it('returns false with no chat surface on screen so the caller can open the dialog', async () => {
// Settings/profiles routes: no [data-composer-target] anywhere.
expect(requestModelMenuToggle()).toBe(false)
expect(await collectModelMenuTargets()).toEqual([])
})
})
+152 -4
View File
@@ -10,7 +10,8 @@
* steal focus from the composer effect.
*/
import { queryVisible } from '@/components/pane-shell/pane-visibility'
import { queryAllVisible, queryVisible } from '@/components/pane-shell/pane-visibility'
import { $hoveredTreeGroup } from '@/components/pane-shell/tree/store'
import type { InlineRefInput } from './inline-refs'
import { RICH_INPUT_SLOT } from './rich-editor'
@@ -42,6 +43,22 @@ const INSERT_EVENT = 'hermes:composer-insert'
const INSERT_REFS_EVENT = 'hermes:composer-insert-refs'
const SUBMIT_EVENT = 'hermes:composer-submit'
const VOICE_TOGGLE_EVENT = 'hermes:composer-voice-toggle'
const MODEL_MENU_EVENT = 'hermes:composer-model-menu'
/** Inline edit composer root — mounted only while a user bubble is being edited. */
const EDIT_COMPOSER_ROOT = '[data-slot="aui_edit-composer-root"]'
/** Attribute-safe selector fragment. jsdom (vitest) does not ship `CSS.escape`. */
const cssEscape = (value: string): string => {
if (typeof CSS !== 'undefined' && typeof CSS.escape === 'function') {
return CSS.escape(value)
}
// Our targets are `'main'` / `'edit'` / `'tile:<id>'` — alphanumerics plus `:`
// and `-`. Escape anything outside that set so a weird id cannot break the
// attribute selector.
return value.replace(/[^a-zA-Z0-9_:-]/g, ch => `\\${ch}`)
}
interface SubmitDetail {
target: ComposerTarget
@@ -50,7 +67,76 @@ interface SubmitDetail {
let activeTarget: ComposerTarget = 'main'
const resolve = (target: ComposerTarget | 'active') => (target === 'active' ? activeTarget : target)
/**
* The chat surface currently on screen (`data-composer-target` hung off each
* ChatView). Inactive tabs stay mounted with `data-pane-hidden`, so this uses
* the same visibility policy as every other document-wide surface lookup.
*/
const visibleChatTarget = (): ComposerTarget | null => {
if (typeof document === 'undefined') {
return null
}
const surface = queryVisible<HTMLElement>('[data-composer-target]')
const target = surface?.dataset.composerTarget
return target ? (target as ComposerTarget) : null
}
/** True when `target` still has a live, on-screen subscriber. */
const targetIsReachable = (target: ComposerTarget): boolean => {
if (typeof document === 'undefined') {
return true
}
// The edit composer is an in-thread overlay, not a chat surface — it never
// stamps `data-composer-target`. While its root is mounted it still owns the
// bus; once it tears down the claim is dead.
if (target === 'edit') {
return Boolean(document.querySelector(EDIT_COMPOSER_ROOT))
}
// Exact match on a VISIBLE surface. Background keep-alive tabs carry the same
// `data-composer-target` but sit under `data-pane-hidden`, so queryVisible
// filters them out.
if (queryVisible(`[data-composer-target="${cssEscape(target)}"]`)) {
return true
}
// A different chat surface is on screen → this claim is buried or gone.
// (A claim with zero stamped surfaces yet — first paint, pure-unit tests —
// keeps the marked key until the DOM contradicts it.)
if (queryVisible('[data-composer-target]')) {
return false
}
return true
}
/**
* The composer `'active'` should route to right now.
*
* The cached claim (`activeTarget`) wins while its surface is still on screen.
* Tab stacks keep inactive panes mounted, so focusing a tile then clicking the
* main tab leaves `activeTarget` pointing at a buried composer — with no
* subscriber on the visible surface, every type-to-focus keystroke is
* preventDefault'd and dropped. Heal to the visible chat surface (or main)
* whenever the claim is off-screen or gone, and keep the cache honest so Esc /
* voice / soft `/` agree with the keyboard path.
*/
const resolveActive = (): ComposerTarget => {
if (targetIsReachable(activeTarget)) {
return activeTarget
}
const visible = visibleChatTarget() ?? 'main'
activeTarget = visible
return visible
}
const resolve = (target: ComposerTarget | 'active') => (target === 'active' ? resolveActive() : target)
const dispatch = <T>(name: string, detail: T) => {
if (typeof window === 'undefined') {
@@ -82,9 +168,33 @@ export const markActiveComposer = (target: ComposerTarget) => {
activeTarget = target
}
/** Hand the routing key back when a composer unmounts, so `'active'` can never
* resolve to a composer that no longer has a subscriber — such a request is
* dispatched and then dropped by every mounted composer's target filter, and
* nothing re-marks the active composer on its own.
*
* Guarded on identity: a composer unmounting AFTER another one claimed the key
* (closing a background tile, a deferred edit-close cleanup) must not steal it
* from the live claimant. Falls through to {@link resolveActive} when the
* caller's surface is buried rather than gone, so closing on a tab switch that
* already re-fronted another chat surfaces there immediately. */
export const releaseActiveComposer = (target: ComposerTarget) => {
if (activeTarget !== target) {
return
}
// Prefer the visible chat surface over a hard `'main'` default — releasing a
// closed tile while another tile is fronted should land there, not the
// (possibly buried) workspace tab.
activeTarget = visibleChatTarget() ?? 'main'
}
/** The composer that last held focus — the target `'active'` resolves to.
* Used by broadcast listeners (voice, Esc-to-stop) to act on exactly one. */
export const getActiveComposer = (): ComposerTarget => activeTarget
* Used by broadcast listeners (voice, Esc-to-stop) to act on exactly one.
* Heals a stale claim the same way {@link requestComposerFocus} does, so Esc
* and type-to-focus never disagree after a tab switch left the bus pointing at
* a keep-alive-mounted background composer. */
export const getActiveComposer = (): ComposerTarget => resolveActive()
export const requestComposerFocus = (
target: ComposerTarget | 'active' = 'active',
@@ -150,6 +260,44 @@ export const requestVoiceToggle = (target: ComposerTarget | 'active' = 'active')
export const onComposerVoiceToggleRequest = (handler: (target: ComposerTarget) => void) =>
subscribe<{ target: ComposerTarget }>(VOICE_TOGGLE_EVENT, ({ target }) => handler(target))
/** The chat surface inside the zone the pointer is over, if any. Mirrors the
* tab verbs' hover-first targeting (`tabTargetGroupId`, #74447): the model
* hotkey lands in the pane you're pointing at without clicking into it first.
* Hidden keep-alive tabs are skipped like every document-wide lookup. */
const composerTargetInHoveredZone = (): ComposerTarget | null => {
const zone = $hoveredTreeGroup.get()
if (!zone || typeof document === 'undefined') {
return null
}
const surface = queryAllVisible<HTMLElement>('[data-composer-target]').find(
el => el.closest<HTMLElement>('[data-tree-group]')?.dataset.treeGroup === zone
)
return (surface?.dataset.composerTarget as ComposerTarget | undefined) ?? null
}
/** Toggle ONE composer's model menu — the `composer.modelPicker` hotkey.
* Targets the pane under the pointer first (the tab-verb convention), then
* the active composer. Returns false when no chat surface is on screen at
* all (settings, profiles…), so the caller can fall back to the full
* model-picker dialog instead of dispatching into the void. */
export const requestModelMenuToggle = (): boolean => {
if (typeof document !== 'undefined' && !queryVisible('[data-composer-target]')) {
return false
}
dispatch<{ target: ComposerTarget }>(MODEL_MENU_EVENT, {
target: composerTargetInHoveredZone() ?? resolveActive()
})
return true
}
export const onComposerModelMenuRequest = (handler: (target: ComposerTarget) => void) =>
subscribe<{ target: ComposerTarget }>(MODEL_MENU_EVENT, ({ target }) => handler(target))
/**
* Focus a composer input across React commit + browser focus restore.
*
@@ -0,0 +1,107 @@
import { act, renderHook } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { queryClient } from '@/lib/query-client'
import { useAtCompletions } from './use-at-completions'
function gatewayStub(latencyMs = 40) {
const calls: string[] = []
const gateway = {
request: vi.fn(async (_method: string, params: { word: string }) => {
calls.push(params.word)
await new Promise(r => setTimeout(r, latencyMs))
return { items: [{ text: `@folder:${params.word.slice(1)}x/`, display: 'x/', meta: 'dir' }] }
})
}
return { calls, gateway }
}
function setup(latencyMs = 40) {
const { calls, gateway } = gatewayStub(latencyMs)
const { result } = renderHook(() => useAtCompletions({ gateway: gateway as never, sessionId: 's1', cwd: '/repo' }))
return { calls, result }
}
/** Type a burst of keystrokes `gapMs` apart, like a person. */
async function type(
result: { current: { adapter: { search?: (q: string) => unknown } } },
queries: string[],
gapMs: number
) {
for (const q of queries) {
act(() => {
result.current.adapter.search?.(q)
})
await act(async () => {
await vi.advanceTimersByTimeAsync(gapMs)
})
}
}
describe('PERF: @ path completions are cached and skip the debounce', () => {
it('serves a repeated query with no round trip and no spinner', async () => {
vi.useFakeTimers()
queryClient.clear()
const { calls, result } = setup()
// First visit to `apps/` pays the round trip.
await type(result, ['apps/'], 0)
await act(async () => {
await vi.advanceTimersByTimeAsync(200)
})
const afterFirst = calls.length
expect(afterFirst).toBe(1)
// Walk away and come back — Tab in, Backspace out, retype. Every one of
// these used to be a fresh git ls-files + rank on the backend.
await type(result, ['apps/desktop/', 'apps/', 'apps/desktop/', 'apps/'], 0)
await act(async () => {
await vi.advanceTimersByTimeAsync(200)
})
// Two distinct paths, so exactly two round trips total — the repeats are free.
expect(calls.length).toBe(2)
expect(result.current.loading).toBe(false)
vi.useRealTimers()
})
it('a cached query paints without waiting out the debounce', async () => {
vi.useFakeTimers()
queryClient.clear()
const { calls, result } = setup()
await type(result, ['apps/'], 0)
await act(async () => {
await vi.advanceTimersByTimeAsync(200)
})
expect(calls.length).toBe(1)
// Re-ask for the cached query and advance by far less than the 60ms
// debounce. A cached answer resolves in a microtask, so it must paint
// without the timer and without ever flipping the spinner on.
act(() => {
result.current.adapter.search?.('apps/')
})
await act(async () => {
await vi.advanceTimersByTimeAsync(1)
})
expect(result.current.loading).toBe(false)
expect(calls.length).toBe(1)
vi.useRealTimers()
})
})
@@ -1,7 +1,9 @@
import type { Unstable_TriggerAdapter, Unstable_TriggerItem } from '@assistant-ui/core'
import { useCallback } from 'react'
import { refChipLabel } from '@/components/assistant-ui/directive-text'
import type { HermesGateway } from '@/hermes'
import { cachedPathCompletion, hasCachedPathCompletion } from '@/lib/slash-completion-cache'
import { normalize } from '@/lib/text'
import type { CompletionEntry, CompletionPayload } from './use-live-completion-adapter'
@@ -60,7 +62,14 @@ function classify(entry: CompletionEntry): {
return {
type: kind,
insertId: rest,
display: textValue(entry.display, rest || `@${kind}:`),
// The row shows exactly what picking it produces. Upstream keeps one
// label per item and hands it to the chip verbatim (DirectiveNode's
// `__label = item.label`); our wire format is `@kind:value`, which can't
// carry a label the way their `:type[label]{name=id}` does, so the same
// invariant is held by deriving both ends from refChipLabel. Without
// this the list said `desktop/`, the editor said `apps/desktop/`, and
// the chip said `desktop` — three names for one folder.
display: rest ? refChipLabel(kind, rest) : textValue(entry.display, `@${kind}:`),
meta: textValue(entry.meta)
}
}
@@ -82,6 +91,11 @@ export function useAtCompletions(options: {
const { gateway, sessionId, cwd } = options
const enabled = Boolean(gateway)
// Cache key: the completion depends on the query AND the directory it's
// resolved against, so a cwd or session change can't serve another tree's
// listing.
const cacheKey = useCallback((query: string) => `${cwd ?? ''}|${sessionId ?? ''}|${query}`, [cwd, sessionId])
const fetcher = useCallback(
async (query: string): Promise<CompletionPayload> => {
const starters = starterEntries(query)
@@ -102,7 +116,15 @@ export function useAtCompletions(options: {
}
try {
const result = await gateway.request<{ items?: CompletionEntry[] }>('complete.path', params)
// De-duplicated the same way `/` completions are. Walking a path is
// inherently repetitive — Tab into a folder, Backspace out, retype a
// segment — and every one of those steps used to be a fresh
// `git ls-files` + rank on the backend (~40ms of the ~50ms round trip
// measured on this repo's 8k files).
const result = await cachedPathCompletion(cacheKey(query), () =>
gateway.request<{ items?: CompletionEntry[] }>('complete.path', params)
)
const items = result.items ?? []
return { items: items.length > 0 ? items : starters, query }
@@ -110,7 +132,7 @@ export function useAtCompletions(options: {
return { items: starters, query }
}
},
[gateway, sessionId, cwd]
[cacheKey, gateway, sessionId, cwd]
)
const toItem = useCallback((entry: CompletionEntry, index: number): Unstable_TriggerItem => {
@@ -135,7 +157,13 @@ export function useAtCompletions(options: {
}
}, [])
return useLiveCompletionAdapter({ enabled, fetcher, toItem })
// A query already in cache skips both the debounce and the loading state.
// This is what makes walking a tree feel instant rather than merely fast:
// the 60ms debounce exists to avoid a request per keystroke, and it buys
// nothing when the answer is already in hand.
const isCached = useCallback((query: string) => hasCachedPathCompletion(cacheKey(query)), [cacheKey])
return useLiveCompletionAdapter({ enabled, fetcher, isCached, toItem })
}
/** Re-export `classify` for use by the formatter (insertion side). */
@@ -5,6 +5,8 @@ import { afterEach, describe, expect, it, vi } from 'vitest'
import { type ComposerAttachment, mainComposerScope, stashSessionDraft } from '@/store/composer'
import type { QueueEditState } from '../composer-utils'
import { type ComposerTarget, getActiveComposer, markActiveComposer } from '../focus'
import { type ComposerScope, ComposerScopeProvider, MAIN_COMPOSER_SCOPE } from '../scope'
import { useComposerDraft } from './use-composer-draft'
@@ -128,3 +130,48 @@ describe('useComposerDraft — rehydrate diagnostic log stays redacted', () => {
})
})
})
describe('useComposerDraft — a closing composer hands the focus-bus key back', () => {
afterEach(() => {
cleanup()
mainComposerScope.clear()
markActiveComposer('main')
})
function renderScoped(target: ComposerTarget) {
const scope: ComposerScope = { ...MAIN_COMPOSER_SCOPE, target }
return render(
<ComposerScopeProvider value={scope}>
<ProbeHarness
activeQueueSessionKey="session-tile"
onLayoutSnapshot={() => undefined}
sessionId="session-tile"
/>
</ComposerScopeProvider>
)
}
it('stops `active` resolving to a session tile once the tile unmounts', () => {
const { unmount } = renderScoped('tile:abc')
// Mounting claims the bus for this tile — the leak precondition.
expect(getActiveComposer()).toBe('tile:abc')
unmount()
expect(getActiveComposer()).toBe('main')
})
it('leaves the key alone when another composer claimed it before this one unmounted', () => {
const { unmount } = renderScoped('tile:abc')
expect(getActiveComposer()).toBe('tile:abc')
// The user clicks into a second tile, which claims the bus.
markActiveComposer('tile:other')
unmount()
expect(getActiveComposer()).toBe('tile:other')
})
})
@@ -2,6 +2,7 @@ import { useAui, useAuiState, useComposerRuntime } from '@assistant-ui/react'
import { type RefObject, useCallback, useEffect, useLayoutEffect, useRef, useState } from 'react'
import { SLASH_COMMAND_RE } from '@/lib/chat-runtime'
import { sanitizeComposerInput } from '@/lib/composer-input-sanitize'
import { type ComposerAttachment, stashSessionDraft, takeSessionDraft } from '@/store/composer'
import { isBrowsingHistory } from '@/store/composer-input-history'
@@ -17,10 +18,17 @@ import {
markActiveComposer,
onComposerFocusRequest,
onComposerInsertRefsRequest,
onComposerInsertRequest
onComposerInsertRequest,
releaseActiveComposer
} from '../focus'
import { type InlineRefInput, insertInlineRefsIntoEditor } from '../inline-refs'
import { composerPlainText, placeCaretEnd, REF_RE, renderComposerContents } from '../rich-editor'
import {
composerPlainText,
normalizeComposerEditorDom,
placeCaretEnd,
REF_RE,
renderComposerContents
} from '../rich-editor'
import { useComposerScope } from '../scope'
import type { ChatBarProps } from '../types'
@@ -120,7 +128,7 @@ export function useComposerDraft({
const editor = editorRef.current
if (editor) {
renderComposerContents(editor, next)
renderComposerContents(editor, next, { trailingCommitted: true })
placeCaretEnd(editor)
}
@@ -153,6 +161,15 @@ export function useComposerDraft({
}
}, [focusInput, focusKey, focusRequestId, inputDisabled])
// The mirror of the `markActiveComposer` above: give the key back when this
// composer goes away (a session tile closing, a pane unmounting). Covers both
// claim sites for this composer — `focusInput` here and ChatBar's `onFocus` —
// since they mark the same scope target. Without it `'active'` keeps
// resolving to a dead tile and every routed focus/insert request is dropped.
// (Heal-to-visible in focus.ts covers the keep-alive-tab case where the pane
// stays mounted behind the front tab; this covers true unmounts.)
useEffect(() => () => releaseActiveComposer(target), [target])
useEffect(() => {
if (inputDisabled) {
return undefined
@@ -227,7 +244,13 @@ export function useComposerDraft({
return draftRef.current
}
const text = composerPlainText(editor)
// Same normalize-then-sanitize the rAF flush does. An emptied editor still
// holds the placeholder <br> that keeps the contenteditable from collapsing
// to a sliver, and that serializes as "\n" — so an editor the user just
// cleared would otherwise stash a one-newline draft and come back non-empty.
normalizeComposerEditorDom(editor)
const text = sanitizeComposerInput(composerPlainText(editor))
if (text !== draftRef.current) {
draftRef.current = text
@@ -255,7 +278,7 @@ export function useComposerDraft({
const editor = editorRef.current
if (editor && document.activeElement !== editor && composerPlainText(editor) !== text) {
renderComposerContents(editor, text)
renderComposerContents(editor, text, { trailingCommitted: true })
}
if (isBrowsingHistory(sessionIdRef.current) || queueEditRef.current) {
@@ -14,6 +14,7 @@ import { useResizeObserver } from '@/hooks/use-resize-observer'
import { COMPOSER_COMPACT_PILL_PX, COMPOSER_SINGLE_LINE_MAX_PX, COMPOSER_STACK_BREAKPOINT_PX } from '../composer-utils'
interface UseComposerMetricsArgs {
composerDockRef: RefObject<HTMLDivElement | null>
composerRef: RefObject<HTMLFormElement | null>
composerSurfaceRef: RefObject<HTMLDivElement | null>
editorRef: RefObject<HTMLDivElement | null>
@@ -28,7 +29,13 @@ interface UseComposerMetricsArgs {
* tree's computed style, and `tight` only flips when it crosses the breakpoint.
* Returns `stacked` (the only value the render needs).
*/
export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef, poppedOut }: UseComposerMetricsArgs): {
export function useComposerMetrics({
composerDockRef,
composerRef,
composerSurfaceRef,
editorRef,
poppedOut
}: UseComposerMetricsArgs): {
compactPill: boolean
stacked: boolean
} {
@@ -89,8 +96,11 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
const syncComposerMetrics = useCallback(() => {
const composer = composerRef.current
// The dock is the full docked footprint — strips, status stack, composer —
// so it, not the composer alone, is what the thread has to clear.
const dock = composerDockRef.current
if (!composer) {
if (!composer || !dock) {
return
}
@@ -108,7 +118,8 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
return
}
const { height, width } = composer.getBoundingClientRect()
const { height } = dock.getBoundingClientRect()
const { width } = composer.getBoundingClientRect()
const surfaceHeight = composerSurfaceRef.current?.getBoundingClientRect().height
if (width > 0) {
@@ -156,9 +167,9 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
setSurfaceVar(composer, COMPOSER_SURFACE_HEIGHT_VAR, `${bucket}px`)
}
}
}, [composerRef, composerSurfaceRef, editorRef])
}, [composerDockRef, composerRef, composerSurfaceRef, editorRef])
useResizeObserver(syncComposerMetrics, composerRef, composerSurfaceRef, editorRef)
useResizeObserver(syncComposerMetrics, composerDockRef, composerRef, composerSurfaceRef, editorRef)
// Toggling pop-out changes whether the composer reserves thread clearance.
// The ResizeObserver may not fire (the box can keep the same box size), so
@@ -170,10 +181,8 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
useEffect(() => {
// Resolve the owning surface while the composer is still attached; the
// unmount cleanup runs after React detached the node, where closest()
// can no longer find [data-chat-surface] and would clear the document
// root instead of this surface (same class of bug as the status stack's
// stale-clearance leak).
// unmount cleanup runs after React detached the node, where closest() can
// no longer find [data-chat-surface].
const root = chatSurfaceRoot(composerRef.current)
return () => {
@@ -223,3 +223,72 @@ describe('useComposerTrigger — free-text slash arguments', () => {
expect(editor.querySelector('[data-slash-kind]')?.getAttribute('data-ref-text')).toBe('/personality creative')
})
})
describe('useComposerTrigger — chip survival (the plaintext-demotion bug class)', () => {
it('keeps a leading command pill through a Backspace path-ascend', () => {
// The reported repro: `/work @folder…` then Backspace — both chips went
// plaintext because ascend re-rendered the whole editor from text.
const editor = mountEditor('/work @Desktop/')
const { hook } = mountTrigger(editor, [])
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
act(() => hook.result.current.refreshTrigger())
expect(hook.result.current.trigger).toMatchObject({ kind: '@', query: 'Desktop/' })
let ran = false
act(() => {
ran = hook.result.current.ascendTriggerPath()
})
expect(ran).toBe(true)
expect(composerPlainText(editor)).toBe('/work @')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
})
it('keeps a leading command pill when a folder pick commits its ref chip', () => {
const editor = mountEditor('/work @Desk')
const folder: Unstable_TriggerItem = {
id: 'folder:Desktop',
type: 'folder',
label: 'Desktop',
metadata: { rawText: '@folder:Desktop', insertId: 'Desktop' }
}
const { hook } = mountTrigger(editor, [folder])
act(() => hook.result.current.refreshTrigger())
act(() => hook.result.current.replaceTriggerWithChip(folder))
expect(composerPlainText(editor)).toBe('/work @folder:`Desktop` ')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
expect(editor.querySelector('[data-ref-kind="folder"]')).not.toBeNull()
})
it('commits in place when Chromium has split the token across text nodes', () => {
// Chromium fragments text nodes around contenteditable=false chips; the
// commit path must span the fragments instead of bailing to a full
// re-render.
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
editor.contentEditable = 'true'
document.body.append(editor)
editor.append(document.createTextNode('please run /c'), document.createTextNode('le'))
const caret = document.createRange()
caret.setStart(editor.lastChild!, 2)
caret.collapse(true)
const selection = window.getSelection()!
selection.removeAllRanges()
selection.addRange(caret)
const { hook } = mountTrigger(editor, [item('/clean')])
act(() => hook.result.current.refreshTrigger())
act(() => hook.result.current.replaceTriggerWithChip(item('/clean')))
expect(composerPlainText(editor)).toBe('please run /clean ')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
})
})
@@ -12,14 +12,59 @@ import {
slashCommandToken
} from '../composer-utils'
import {
appendComposerContents,
caretOffsetInEditor,
composerPlainText,
placeCaretEnd,
placeCaretAtOffset,
refChipElement,
renderComposerContents,
replaceBeforeCaret,
RICH_INPUT_SLOT,
slashChipElement
} from '../rich-editor'
import { detectTrigger, textBeforeCaret, type TriggerState } from '../text-utils'
/**
* Rebuild-from-text fallback for carets the range walk can't anchor (a
* non-collapsed selection, a caret not preceded by contiguous text). It
* re-renders the whole editor from serialized text, so it only runs when the
* in-place path reports failure — never as the default.
*
* The split is around the CARET, not the end of the draft. Slicing
* `length - tokenLength` off the end assumed the trigger token was the last
* thing in the editor: a completion picked mid-message chopped the trailing
* prose off and stranded a partial `folder:` in front of the chip, because the
* window it removed wasn't the token the user was typing.
*/
export function rebuildAroundCaret(editor: HTMLDivElement, tokenLength: number, insert: DocumentFragment | string) {
const current = composerPlainText(editor)
const caret = caretOffsetInEditor(editor)
const prefix = current.slice(0, Math.max(0, caret - tokenLength))
const suffix = current.slice(caret)
if (typeof insert === 'string') {
renderComposerContents(editor, `${prefix}${insert}${suffix}`)
placeCaretAtOffset(editor, prefix.length + insert.length)
return
}
// Measure before appending — moving a fragment empties it. Appending the
// element rather than re-serializing keeps mid-message slash pills alive:
// they have no text hydration, unlike `@` refs and the leading command.
const scratch = document.createElement('div')
scratch.dataset.slot = RICH_INPUT_SLOT
scratch.append(insert.cloneNode(true))
const inserted = composerPlainText(scratch)
renderComposerContents(editor, prefix)
editor.append(insert)
appendComposerContents(editor, suffix)
placeCaretAtOffset(editor, prefix.length + inserted.length)
}
interface CompletionSource {
adapter: Unstable_TriggerAdapter | null
loading: boolean
@@ -29,6 +74,10 @@ interface UseComposerTriggerOptions {
at: CompletionSource
draftRef: MutableRefObject<string>
editorRef: RefObject<HTMLDivElement | null>
/** `:joy` emoji completions — inserts the emoji character, never a chip. */
emoji?: CompletionSource
/** Bank the pre-commit state so a popover pick is a single undo step. */
recordUndoPoint?: () => void
requestMainFocus: () => void
setComposerText: (text: string) => void
slash: CompletionSource
@@ -47,6 +96,8 @@ export function useComposerTrigger({
at,
draftRef,
editorRef,
emoji,
recordUndoPoint,
requestMainFocus,
setComposerText,
slash
@@ -87,7 +138,7 @@ export function useComposerTrigger({
// is present do we pay the cost of the full walk + DOM range work.
const rawText = editor.textContent ?? ''
if (!rawText.includes('@') && !rawText.includes('/')) {
if (!rawText.includes('@') && !rawText.includes('/') && !rawText.includes(':')) {
if (trigger) {
setTrigger(null)
resetTriggerActive()
@@ -124,7 +175,13 @@ export function useComposerTrigger({
}, [editorRef, resetTriggerActive, trigger])
const triggerAdapter: Unstable_TriggerAdapter | null =
trigger?.kind === '@' ? at.adapter : trigger?.kind === '/' ? slash.adapter : null
trigger?.kind === '@'
? at.adapter
: trigger?.kind === '/'
? slash.adapter
: trigger?.kind === ':'
? (emoji?.adapter ?? null)
: null
useEffect(() => {
if (!trigger || !triggerAdapter?.search) {
@@ -142,7 +199,14 @@ export function useComposerTrigger({
setTriggerItems(trigger.inline ? items.filter(isSkillItem) : items)
}, [trigger, triggerAdapter])
const triggerLoading = trigger?.kind === '@' ? at.loading : trigger?.kind === '/' ? slash.loading : false
const triggerLoading =
trigger?.kind === '@'
? at.loading
: trigger?.kind === '/'
? slash.loading
: trigger?.kind === ':'
? (emoji?.loading ?? false)
: false
// Suppress the "No matches" empty state once a slash command is past its name:
// a no-arg command has nothing to offer, and a fully-typed arg commits on
@@ -214,17 +278,22 @@ export function useComposerTrigger({
return
}
// Bank the pre-commit state first — every path below mutates the editor,
// and a pick must be exactly one undo step.
recordUndoPoint?.()
const rebuildAround = (insert: DocumentFragment | string) => rebuildAroundCaret(editor, trigger.tokenLength, insert)
// Action items (e.g. "Browse all sessions…") run a side effect instead of
// inserting a chip: strip the typed trigger token, then fire the action.
const completionAction = (item.metadata as { action?: unknown } | undefined)?.action
const runAction = typeof completionAction === 'string' ? COMPLETION_ACTIONS[completionAction] : undefined
if (runAction) {
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
if (!replaceBeforeCaret(editor, trigger.tokenLength, document.createDocumentFragment())) {
rebuildAround('')
}
renderComposerContents(editor, prefix)
placeCaretEnd(editor)
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
closeTrigger()
@@ -247,19 +316,29 @@ export function useComposerTrigger({
? String((item.metadata as { insertId?: unknown } | undefined)?.insertId ?? '')
: ''
if (descendInto) {
const path = descendInto.endsWith('/') ? descendInto : `${descendInto}/`
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
renderComposerContents(editor, `${prefix}@${path}`)
placeCaretEnd(editor)
const finish = (keepOpen: boolean) => {
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
requestMainFocus()
window.setTimeout(refreshTrigger, 0)
keepOpen ? window.setTimeout(refreshTrigger, 0) : closeTrigger()
}
return
if (descendInto) {
const path = descendInto.endsWith('/') ? descendInto : `${descendInto}/`
// Carry the browse scope down with the path. Dropping it turned an
// explicit `@folder:` browse into a bare `@apps/desktop/` token halfway
// through, so the next completion silently widened back to files and the
// committed chip had to re-guess the kind from a trailing slash.
const scope = trigger.scope ? `${trigger.scope}:` : ''
const fragment = document.createDocumentFragment()
fragment.append(document.createTextNode(`@${scope}${path}`))
if (!replaceBeforeCaret(editor, trigger.tokenLength, fragment)) {
rebuildAround(`@${scope}${path}`)
}
return finish(true)
}
// Picking a bare arg-taking command (e.g. `/personality`) shouldn't commit
@@ -280,90 +359,78 @@ export function useComposerTrigger({
const slashKind = !expandsToArgs && trigger.kind === '/' ? slashChipKindForItem(item) : null
const keepTriggerOpen = starter || (expandsToArgs && argumentMode !== 'text')
const finish = () => {
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
requestMainFocus()
keepTriggerOpen ? window.setTimeout(refreshTrigger, 0) : closeTrigger()
}
const sel = window.getSelection()
const range = sel?.rangeCount ? sel.getRangeAt(0) : null
const node = range?.startContainer
const offset = range?.startOffset ?? 0
if (!sel || !range || node?.nodeType !== Node.TEXT_NODE || offset < trigger.tokenLength) {
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
if (slashKind) {
// Two-step arg picks (e.g. `/handoff` pill already inserted, now picking
// the platform) land here because the caret sits past a contenteditable
// chip. Rebuild the prefix and re-emit a single pill for the full command.
renderComposerContents(editor, prefix)
editor.append(slashChipElement(serialized, slashKind), document.createTextNode(' '))
placeCaretEnd(editor)
return finish()
}
renderComposerContents(editor, `${prefix}${text}`)
placeCaretEnd(editor)
return finish()
}
const replaceRange = document.createRange()
replaceRange.setStart(node, offset - trigger.tokenLength)
replaceRange.setEnd(node, offset)
replaceRange.deleteContents()
const chip = slashKind
? slashChipElement(serialized, slashKind)
: directive
? refChipElement(directive[1], directive[2])
? // Carry the picked row's own label into the chip rather than letting
// it re-derive one from the value. Upstream's DirectiveNode does the
// same (`__label = item.label`), and it's what makes the list and the
// chip agree: you get the string you just read, not a second guess at
// it. Falls back to the shared deriver for callers with no label.
refChipElement(directive[1], directive[2], (item.metadata as { display?: string })?.display || item.label)
: null
if (chip) {
const space = document.createTextNode(' ')
const fragment = document.createDocumentFragment()
fragment.append(chip, space)
replaceRange.insertNode(fragment)
// The trailing space is a convenience for "keep typing after the chip", so
// it's wrong when the caret already has whitespace in front of it — a pick
// made mid-sentence would leave a double space in the prose.
const followedBySpace = /^\s/.test(composerPlainText(editor).slice(caretOffsetInEditor(editor)))
const fragment = document.createDocumentFragment()
const caret = document.createRange()
caret.setStart(space, 1)
caret.collapse(true)
sel.removeAllRanges()
sel.addRange(caret)
chip
? fragment.append(chip, ...(followedBySpace ? [] : [document.createTextNode(' ')]))
: fragment.append(document.createTextNode(followedBySpace ? text.trimEnd() : text))
return finish()
if (!replaceBeforeCaret(editor, trigger.tokenLength, fragment)) {
// The failed in-place attempt never consumed the fragment, so the chip +
// trailing space are re-inserted around the caret here. Moving the
// element (rather than re-serializing) keeps mid-message slash pills
// alive — they have no text hydration, unlike `@` refs and the leading
// command.
rebuildAround(chip ? fragment : text)
}
document.execCommand('insertText', false, text)
finish()
finish(keepTriggerOpen)
}
/** Backspace inside an `@` path drops the last segment (`a/b/` → `a/`)
* instead of one character. Descending is one Tab per level, so climbing
* back out should cost one key too rather than a held delete. Returns
* instead of one character, and once the path is empty it drops the browse
* scope (`@folder:` → `@`) rather than nibbling `:`, `r`, `e`, `d`… back
* through the directive syntax the user never typed. Descending is one Tab
* per level, so climbing back out costs one key per level too. Returns
* false when the caret isn't in a path, so keydown falls through. */
const ascendTriggerPath = () => {
const editor = editorRef.current
if (!editor || trigger?.kind !== '@' || !trigger.query.includes('/')) {
if (!editor || trigger?.kind !== '@') {
return false
}
const scope = trigger.scope ? `${trigger.scope}:` : ''
if (!trigger.value.includes('/') && !scope) {
return false
}
// Trailing slash means we're listing a folder's children: drop that
// folder. Otherwise a partial segment is typed — drop just that.
const trimmed = trigger.query.replace(/\/$/, '')
// folder. Otherwise a partial segment is typed — drop just that. With the
// value already empty, the only thing left to drop is the scope itself.
const trimmed = trigger.value.replace(/\/$/, '')
const parent = trimmed.slice(0, trimmed.lastIndexOf('/') + 1)
const next = trigger.value ? `${scope}${parent}` : ''
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
recordUndoPoint?.()
const fragment = document.createDocumentFragment()
fragment.append(document.createTextNode(`@${next}`))
// In place first: the destructive re-render fallback rebuilds the editor
// from text, which is exactly what used to demote a leading command pill
// to plaintext on every Backspace inside a path.
if (!replaceBeforeCaret(editor, trigger.tokenLength, fragment)) {
rebuildAroundCaret(editor, trigger.tokenLength, `@${next}`)
}
renderComposerContents(editor, `${prefix}@${parent}`)
placeCaretEnd(editor)
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
window.setTimeout(refreshTrigger, 0)
@@ -7,8 +7,8 @@ import { triggerHaptic } from '@/lib/haptics'
import { $voiceConversationStartRequest, takeVoiceConversationStart } from '@/store/composer'
import { resetBrowseState } from '@/store/composer-input-history'
import { $gateway } from '@/store/gateway'
import { notifyError } from '@/store/notifications'
import { $autoSpeakReplies, setAutoSpeakReplies } from '@/store/voice-prefs'
import { notify, notifyError } from '@/store/notifications'
import { $autoSpeakReplies, $voiceStopPhrase, setAutoSpeakReplies } from '@/store/voice-prefs'
import { resumeWakeAfterVoice } from '@/store/wake-word'
import type { ComposerTarget } from '../focus'
@@ -27,6 +27,9 @@ interface UseComposerVoiceArgs {
focusInput: () => void
insertText: (text: string) => void
maxRecordingSeconds: number
/** Interrupt the in-flight agent turn (Stop-button seam) — fired when the
* user speaks over the model while it is still generating. */
onInterrupt?: () => Promise<void> | void
onSubmit: ChatBarProps['onSubmit']
onTranscribeAudio: ChatBarProps['onTranscribeAudio']
sessionId: string | null | undefined
@@ -48,6 +51,7 @@ export function useComposerVoice({
focusInput,
insertText,
maxRecordingSeconds,
onInterrupt,
onSubmit,
onTranscribeAudio,
sessionId,
@@ -129,6 +133,10 @@ export function useComposerVoice({
consumePendingResponse,
enabled: voiceConversationActive,
onFatalError: () => setVoiceConversationActive(false),
// Speaking over the model mid-generation interrupts the in-flight turn —
// the same seam as the Stop button — so the interjection becomes the next
// turn instead of waiting behind a reply the user already rejected.
onInterrupt,
// A spoken stop command ("stop", "never mind", "goodbye", …) ends the
// hands-free conversation. Flipping the flag is the authoritative off
// switch — the enabled=false prop + effect below drive conversation.end()
@@ -208,6 +216,26 @@ export function useComposerVoice({
}
}, [pauseWakeForVoice, resumeWakeIfPaused, voiceConversationActive])
// 'Say "stop" to end the voice chat.' notice when the conversation starts.
// Phrase comes from voice.stop_phrases (first entry) so a custom phrase
// renders correctly; a null phrase (stop_phrases: []) shows no notice.
useEffect(() => {
if (!voiceConversationActive) {
return
}
const phrase = $voiceStopPhrase.get()
if (phrase) {
notify({
id: 'voice-stop-hint',
kind: 'info',
icon: 'mic',
message: t.notifications.voice.sayStopToEnd(phrase)
})
}
}, [t, voiceConversationActive])
useEffect(() => resumeWakeIfPaused, [resumeWakeIfPaused])
// Explicit start/end for the on-screen conversation controls (the hotkey uses
@@ -0,0 +1,122 @@
import { useCallback } from 'react'
import { type CompletionEntry, type CompletionPayload, useLiveCompletionAdapter } from './use-live-completion-adapter'
/**
* `:shortcode:` completions for the composers, Slack-style (`:joy` → 😂).
*
* Draws from the same bundled emojibase-data the reaction picker uses (served
* at ./emojibase by the `hermes:emojibase-assets` vite plugin — offline, no
* CDN). The index lazy-loads on the first `:` trigger, then every query is
* answered from memory, so `isCached` skips the debounce and loading state
* after that first load.
*
* A pick inserts the emoji CHARACTER as plain text — not a chip. Directive
* chips exist to carry machine-readable references the backend resolves
* (@file:, /skill); a picked emoji is just text, so it rides the formatter's
* `rawText` path and lands inline.
*/
interface EmojiEntry {
emoji: string
/** Primary shortcode, e.g. "joy". */
code: string
/** Every shortcode, tag, and label that should match a search. */
haystack: string[]
}
let indexPromise: Promise<EmojiEntry[]> | null = null
let indexLoaded = false
async function loadIndex(): Promise<EmojiEntry[]> {
const [dataRes, codesRes] = await Promise.all([
fetch('./emojibase/en/data.json'),
fetch('./emojibase/en/shortcodes/emojibase.json')
])
const data: { emoji: string; hexcode: string; label: string; tags?: string[] }[] = await dataRes.json()
const codes: Record<string, string | string[]> = await codesRes.json()
const entries: EmojiEntry[] = []
for (const item of data) {
const raw = codes[item.hexcode]
if (!raw) {
continue
}
const shortcodes = Array.isArray(raw) ? raw : [raw]
entries.push({
emoji: item.emoji,
code: shortcodes[0],
haystack: [...shortcodes, ...(item.tags ?? []), item.label.toLowerCase()]
})
}
indexLoaded = true
return entries
}
/** Prefix matches on shortcodes rank first, then tag/label substring hits. */
async function searchEmoji(query: string, limit = 8): Promise<EmojiEntry[]> {
const index = await (indexPromise ??= loadIndex())
const q = query.toLowerCase()
const prefix: EmojiEntry[] = []
const loose: EmojiEntry[] = []
for (const entry of index) {
if (entry.code.startsWith(q) || entry.haystack.some(h => h.startsWith(q))) {
prefix.push(entry)
} else if (entry.haystack.some(h => h.includes(q))) {
loose.push(entry)
}
if (prefix.length >= limit) {
break
}
}
return [...prefix, ...loose].slice(0, limit)
}
export function useEmojiCompletions() {
const fetcher = useCallback(async (query: string): Promise<CompletionPayload> => {
const entries = await searchEmoji(query)
return {
query,
items: entries.map(entry => ({
text: entry.emoji,
display: `${entry.emoji} :${entry.code}:`,
meta: ''
}))
}
}, [])
const toItem = useCallback(
(entry: CompletionEntry, index: number) => ({
id: `${entry.text}|${index}`,
type: 'emoji',
label: typeof entry.display === 'string' ? entry.display : entry.text,
metadata: {
display: typeof entry.display === 'string' ? entry.display : entry.text,
// The formatter's serialize() returns rawText verbatim → the emoji
// character lands as plain inline text, no chip.
rawText: entry.text,
meta: '',
group: '',
action: ''
}
}),
[]
)
return useLiveCompletionAdapter({
enabled: true,
fetcher,
isCached: () => indexLoaded,
toItem
})
}
@@ -49,15 +49,7 @@ function gestureTargetOk(target: EventTarget | null) {
return false
}
// `composer-no-drag`: chrome that lives inside the composer root but isn't
// part of the draggable frame — the floating pill strips. The pills are
// `button`s and already excluded, but the strip's own box (the gaps between
// pills) isn't, so without this a press landing between two badges still
// drags. The strips are `w-fit`, so this costs the grab band only the width
// of the badges themselves.
return !target.closest(
'button, a, input, textarea, select, [role="menuitem"], [data-radix-popper-content-wrapper], [data-slot="composer-no-drag"]'
)
return !target.closest('button, a, input, textarea, select, [role="menuitem"], [data-radix-popper-content-wrapper]')
}
/** Floating composer's 5px outer frame — grab here to drag without long-press. */
@@ -0,0 +1,266 @@
import { act, cleanup, renderHook, waitFor } from '@testing-library/react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type { BargeMonitorCallbacks } from '@/lib/voice-barge-in'
import type { MicRecording } from './use-mic-recorder'
import { useVoiceConversation } from './use-voice-conversation'
// The full-duplex contract: the barge monitor is live across the WHOLE agent
// turn — generation (thinking) and playback (speaking) — so speaking over the
// model interrupts it mid-generation instead of the mic being deaf until TTS
// starts (the Windows report: interruption "never works" because the deaf
// window covered generation, and playback bleed made the old monitor's
// trigger unreachable).
const monitorCalls: BargeMonitorCallbacks[] = []
const stopMonitor = vi.fn()
vi.mock('@/lib/voice-barge-in', () => ({
monitorSpeechDuringPlayback: (callbacks: BargeMonitorCallbacks) => {
monitorCalls.push(callbacks)
return stopMonitor
}
}))
const markVoicePlaybackInterrupted = vi.fn()
const stopVoicePlayback = vi.fn()
vi.mock('@/lib/voice-playback', () => ({
markVoicePlaybackInterrupted: () => markVoicePlaybackInterrupted(),
playSpeechText: vi.fn(async () => true),
startSpeechStream: vi.fn(async () => null),
stopVoicePlayback: () => stopVoicePlayback()
}))
vi.mock('@/lib/thinking-sound', () => ({
startThinkingSound: vi.fn(),
stopThinkingSound: vi.fn()
}))
const micHandle = {
cancel: vi.fn(),
start: vi.fn(async () => undefined),
stop: vi.fn<() => Promise<MicRecording | null>>(async () => null)
}
vi.mock('./use-mic-recorder', () => ({
useMicRecorder: () => ({ handle: micHandle, level: 0, recording: false })
}))
vi.mock('@/i18n', () => ({
useI18n: () => ({
t: {
notifications: {
voice: {
configureSpeechToText: 'configure STT',
couldNotStartSession: 'could not start',
microphoneFailed: 'mic failed',
playbackFailed: 'playback failed',
transcriptionFailed: 'transcription failed',
unavailable: 'unavailable'
}
}
}
})
}))
vi.mock('@/store/notifications', () => ({
notify: vi.fn(),
notifyError: vi.fn()
}))
interface HookProps {
busy: boolean
}
function renderConversation(overrides: { onInterrupt?: () => void; transcript?: string } = {}) {
const onInterrupt = overrides.onInterrupt ?? vi.fn()
// Mirrors the real app: submitting a turn makes the agent busy.
const onBusyChange: { current: (busy: boolean) => void } = { current: () => undefined }
const onSubmit = vi.fn(async () => {
onBusyChange.current(true)
})
const onStopWord = vi.fn()
// First transcription is the turn that starts the conversation; subsequent
// ones are barge captures (the overridable transcript).
let transcriptions = 0
const onTranscribeAudio = vi.fn(async () =>
transcriptions++ === 0 ? 'kick off the task' : (overrides.transcript ?? 'and another thing')
)
const hook = renderHook(
({ busy }: HookProps) =>
useVoiceConversation({
busy,
consumePendingResponse: vi.fn(),
enabled: true,
onInterrupt,
onStopWord,
onSubmit,
onTranscribeAudio,
pendingResponse: () => null
}),
{ initialProps: { busy: false } }
)
onBusyChange.current = busy => hook.rerender({ busy })
return { hook, onInterrupt, onStopWord, onSubmit, onTranscribeAudio }
}
/** Drive the hook into the generation phase (turn submitted, model working). */
async function enterThinking(hook: ReturnType<typeof renderConversation>['hook']) {
await act(async () => {
await hook.result.current.start()
})
await waitFor(() => expect(hook.result.current.status).toBe('listening'))
micHandle.stop.mockResolvedValueOnce({
audio: new Blob(['q'], { type: 'audio/webm' }),
durationMs: 900,
heardSpeech: true
})
await act(async () => {
hook.result.current.stopTurn()
})
await waitFor(() => expect(hook.result.current.status).toBe('thinking'))
}
describe('useVoiceConversation full-duplex barge-in', () => {
beforeEach(() => {
monitorCalls.length = 0
vi.clearAllMocks()
micHandle.start.mockResolvedValue(undefined)
micHandle.stop.mockResolvedValue(null)
})
afterEach(cleanup)
it('arms the barge monitor during generation (before any reply audio exists)', async () => {
const { hook } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(hook.result.current.status).toBe('thinking'))
// busy=true + thinking → the full-duplex monitor must be live.
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
})
it('interrupts the in-flight turn when speech trips mid-generation', async () => {
const { hook, onInterrupt } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
act(() => {
monitorCalls.at(-1)?.onSpeech()
})
expect(onInterrupt).toHaveBeenCalledTimes(1)
expect(markVoicePlaybackInterrupted).toHaveBeenCalled()
expect(stopVoicePlayback).toHaveBeenCalled()
})
it('submits the captured interruption once the interrupt settles (busy clears)', async () => {
const { hook, onSubmit } = renderConversation({ transcript: 'no, do it differently' })
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
const monitor = monitorCalls.at(-1)
act(() => {
monitor?.onSpeech()
})
// Interrupt lands → the turn ends → busy flips false.
hook.rerender({ busy: false })
await act(async () => {
monitor?.onUtterance?.(new Blob(['x'], { type: 'audio/webm' }))
})
await waitFor(() => expect(onSubmit).toHaveBeenCalledWith('no, do it differently'))
})
it('does not interrupt when speech trips during playback (turn already done)', async () => {
const { hook, onInterrupt } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
// Turn finished; playback phase.
hook.rerender({ busy: false })
act(() => {
monitorCalls.at(-1)?.onSpeech()
})
expect(onInterrupt).not.toHaveBeenCalled()
expect(stopVoicePlayback).toHaveBeenCalled()
})
it('a spoken stop command in the barge capture ends the conversation instead of submitting', async () => {
const { hook, onStopWord, onSubmit } = renderConversation({ transcript: 'stop' })
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
const monitor = monitorCalls.at(-1)
act(() => {
monitor?.onSpeech()
})
hook.rerender({ busy: false })
await act(async () => {
monitor?.onUtterance?.(new Blob(['s'], { type: 'audio/webm' }))
})
await waitFor(() => expect(onStopWord).toHaveBeenCalledTimes(1))
// Only the kickoff turn was submitted — the "stop" capture never was.
expect(onSubmit).toHaveBeenCalledTimes(1)
expect(onSubmit).not.toHaveBeenCalledWith('stop')
})
it('re-arms a single monitor per turn (idempotent ensure)', async () => {
const { hook } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
const armed = monitorCalls.length
// Effect re-runs (busy toggles, status changes) must not open more mics.
hook.rerender({ busy: true })
hook.rerender({ busy: true })
expect(monitorCalls.length).toBe(armed)
})
})
@@ -1,6 +1,7 @@
import { useCallback, useEffect, useRef, useState } from 'react'
import { useI18n } from '@/i18n'
import { startThinkingSound, stopThinkingSound } from '@/lib/thinking-sound'
import { monitorSpeechDuringPlayback } from '@/lib/voice-barge-in'
import {
markVoicePlaybackInterrupted,
@@ -27,6 +28,9 @@ interface VoiceConversationOptions {
busy: boolean
enabled: boolean
onFatalError?: () => void
/** Interrupt the in-flight agent turn (the same seam as the Stop button).
* Fired when the user speaks while the model is still generating. */
onInterrupt?: () => Promise<void> | void
onStopWord?: () => void
onSubmit: (text: string) => Promise<void> | void
onTranscribeAudio?: (audio: Blob) => Promise<string>
@@ -37,10 +41,15 @@ interface VoiceConversationOptions {
beforeMicOpen?: () => Promise<void> | void
}
/** How long a barge-triggered interrupt may take to settle before we submit
* the captured utterance anyway. */
const INTERRUPT_SETTLE_TIMEOUT_MS = 5_000
export function useVoiceConversation({
busy,
enabled,
onFatalError,
onInterrupt,
onStopWord,
onSubmit,
onTranscribeAudio,
@@ -62,6 +71,7 @@ export function useVoiceConversation({
const speechSessionRef = useRef<null | SpeechStreamSession>(null)
const stopBargeMonitorRef = useRef<(() => void) | null>(null)
const bargeCapturePendingRef = useRef(false)
const bargedRef = useRef(false)
const speechStartSequenceRef = useRef(0)
const enabledRef = useRef(enabled)
const mutedRef = useRef(muted)
@@ -69,6 +79,12 @@ export function useVoiceConversation({
const statusRef = useRef<ConversationStatus>('idle')
const wasEnabledRef = useRef(enabled)
const onStopWordRef = useRef(onStopWord)
const onInterruptRef = useRef(onInterrupt)
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
onInterruptRef.current = onInterrupt
}, [onInterrupt])
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
@@ -113,6 +129,7 @@ export function useVoiceConversation({
stopBargeMonitorRef.current?.()
stopBargeMonitorRef.current = null
bargeCapturePendingRef.current = false
bargedRef.current = false
speechSessionRef.current = null
responseIdRef.current = null
spokenSourceLengthRef.current = 0
@@ -314,6 +331,25 @@ export function useVoiceConversation({
return
}
// A spoken stop command while barging means "stop everything" — the
// turn/playback was already cut at trip time; now end the conversation
// instead of submitting "stop" as a new prompt.
if (isVoiceStopCommand(transcript)) {
dropSpeechSession()
setStatus('idle')
onStopWordRef.current?.()
return
}
// A generation-phase barge interrupted the in-flight turn; the submit
// path refuses while `busy`, so wait for the interrupt to settle.
const deadline = Date.now() + INTERRUPT_SETTLE_TIMEOUT_MS
while (busyRef.current && Date.now() < deadline) {
await new Promise(resolve => window.setTimeout(resolve, 100))
}
awaitingSpokenResponseRef.current = true
dropSpeechSession()
consumePendingResponse()
@@ -327,24 +363,46 @@ export function useVoiceConversation({
[consumePendingResponse, onSubmit, onTranscribeAudio, voiceCopy.transcriptionFailed]
)
/** Barge-in monitor wiring shared by the live and fallback speech paths. */
const openBargeMonitor = useCallback(
(onBarge: () => void) =>
monitorSpeechDuringPlayback({
onSpeech: () => {
bargeCapturePendingRef.current = true
onBarge()
markVoicePlaybackInterrupted()
stopVoicePlayback()
},
onUtterance: audio => {
bargeCapturePendingRef.current = false
stopBargeMonitorRef.current = null
void submitCapturedUtterance(audio)
/**
* Full-duplex barge-in monitor for the WHOLE agent turn: armed at submit,
* live through generation (thinking) AND playback (speaking).
*
* - generation phase (`busy`): speech interrupts the in-flight turn via
* `onInterrupt` — the same seam as the Stop button — and cuts any TTS that
* managed to start, so the stale reply never speaks.
* - playback phase: speech cuts playback and the captured interruption is
* transcribed and submitted as the next turn.
*
* Idempotent — one monitor owns the mic per turn; re-arming while one is
* live is a no-op (the live/fallback speech paths and the turn-drive effect
* all call this).
*/
const ensureBargeMonitor = useCallback(() => {
if (stopBargeMonitorRef.current) {
return
}
stopBargeMonitorRef.current = monitorSpeechDuringPlayback({
isPlaying: () => $voicePlayback.get().status === 'speaking',
onSpeech: () => {
bargeCapturePendingRef.current = true
bargedRef.current = true
markVoicePlaybackInterrupted()
stopVoicePlayback()
if (busyRef.current) {
// Mid-generation: stop the in-flight turn so the captured utterance
// becomes the next one instead of queueing behind a stale reply.
void onInterruptRef.current?.()
}
}),
[submitCapturedUtterance]
)
},
onUtterance: audio => {
bargeCapturePendingRef.current = false
stopBargeMonitorRef.current = null
void submitCapturedUtterance(audio)
}
})
}, [submitCapturedUtterance])
/** Push any new reply text into the live session; finish when complete. */
const feedSpeechSession = useCallback(
@@ -396,12 +454,9 @@ export function useVoiceConversation({
return
}
let barged = false
stopBargeMonitorRef.current?.()
stopBargeMonitorRef.current = openBargeMonitor(() => {
barged = true
})
// The full-duplex monitor is normally already live (armed at submit);
// this is a safety net for read-aloud-style entries into the loop.
ensureBargeMonitor()
speechStartSequenceRef.current = $voicePlayback.get().sequence
@@ -410,14 +465,14 @@ export function useVoiceConversation({
.finally(() => {
if (responseIdRef.current === responseId) {
awaitingSpokenResponseRef.current = false
settleAfterSpeech(barged)
settleAfterSpeech(bargedRef.current)
}
})
}
poll()
},
[openBargeMonitor, pendingResponse, settleAfterSpeech, voiceCopy.playbackFailed]
[ensureBargeMonitor, pendingResponse, settleAfterSpeech, voiceCopy.playbackFailed]
)
/**
@@ -432,15 +487,11 @@ export function useVoiceConversation({
speechStartSequenceRef.current = $voicePlayback.get().sequence
setStatus('speaking')
let barged = false
// VAD barge-in: the user talking over the reply cuts playback, drops
// the not-yet-spoken remainder, AND keeps capturing — the interruption
// is transcribed from its first syllable instead of losing the opening
// words to a mic re-open.
stopBargeMonitorRef.current = openBargeMonitor(() => {
barged = true
})
// words to a mic re-open. Usually already live (armed at submit).
ensureBargeMonitor()
void (async () => {
const session = await startSpeechStream({ source: 'voice-conversation' })
@@ -483,10 +534,10 @@ export function useVoiceConversation({
}
awaitingSpokenResponseRef.current = false
settleAfterSpeech(barged)
settleAfterSpeech(bargedRef.current)
})()
},
[awaitFallbackSpeech, feedSpeechSession, openBargeMonitor, settleAfterSpeech]
[awaitFallbackSpeech, ensureBargeMonitor, feedSpeechSession, settleAfterSpeech]
)
const start = useCallback(async () => {
@@ -574,6 +625,22 @@ export function useVoiceConversation({
return () => window.removeEventListener('keydown', onKeyDown, { capture: true })
}, [enabled, stopTurn])
// Ambient "thinking" sound: while the agent works (status 'thinking') no
// audio flows, which reads as dead air mid-conversation. Calm bubble blips
// fill the gap; they stop the INSTANT speech starts, the mic re-arms, or the
// conversation ends. Gated by voice.thinking_sound + the shared sound mute.
useEffect(() => {
if (enabled && !muted && status === 'thinking') {
startThinkingSound()
return stopThinkingSound
}
stopThinkingSound()
return undefined
}, [enabled, muted, status])
// Drive the loop: when a voice-submitted reply appears, open a live speech
// session (which feeds itself from then on). Otherwise start listening when
// idle between turns.
@@ -584,6 +651,13 @@ export function useVoiceConversation({
}
if (awaitingSpokenResponseRef.current && status !== 'speaking') {
// Generation phase: the turn is in flight but no reply audio exists
// yet. Keep the mic live so speech can interrupt the model mid-
// generation (full-duplex) instead of going deaf until playback.
if (status === 'thinking' && (busy || bargeCapturePendingRef.current)) {
ensureBargeMonitor()
}
const response = pendingResponse()
if (response) {
@@ -592,8 +666,9 @@ export function useVoiceConversation({
return
}
if (!busy && status === 'thinking') {
// Turn finished without any speakable reply (tool-only, error).
if (!busy && status === 'thinking' && !bargeCapturePendingRef.current) {
// Turn finished without any speakable reply (tool-only, error). A
// live barge capture owns the loop instead — it submits or resumes.
awaitingSpokenResponseRef.current = false
dropSpeechSession()
pendingStartRef.current = true
@@ -610,7 +685,7 @@ export function useVoiceConversation({
if (pendingStartRef.current) {
void startListening()
}
}, [busy, enabled, muted, openLiveSpeech, pendingResponse, startListening, status])
}, [busy, enabled, muted, ensureBargeMonitor, openLiveSpeech, pendingResponse, startListening, status])
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
+229 -153
View File
@@ -11,6 +11,7 @@ import { sanitizeComposerInput } from '@/lib/composer-input-sanitize'
import { DATA_IMAGE_URL_RE } from '@/lib/embedded-images'
import { triggerHaptic } from '@/lib/haptics'
import { cn } from '@/lib/utils'
import { interceptsTypedVoiceStop } from '@/lib/voice-stop-word'
import { sessionCompacting } from '@/store/compaction'
import { browseBackward, browseForward, deriveUserHistory, isBrowsingHistory } from '@/store/composer-input-history'
import { POPOUT_WIDTH_REM } from '@/store/composer-popout'
@@ -48,12 +49,15 @@ import { useComposerTrigger } from './hooks/use-composer-trigger'
import { useComposerUndo } from './hooks/use-composer-undo'
import { useComposerUrlDialog } from './hooks/use-composer-url-dialog'
import { useComposerVoice } from './hooks/use-composer-voice'
import { useEmojiCompletions } from './hooks/use-emoji-completions'
import { useComposerMicroActions } from './hooks/use-micro-actions'
import { useSlashCompletions } from './hooks/use-slash-completions'
import { useSessionStatusPresence } from './hooks/use-status-presence'
import { ActionBadges } from './micro-actions'
import { chipTypedPathOnSpace, pathifyRefs } from './path-refs'
import { QueuePanel } from './queue-panel'
import {
COMPOSER_PLACEHOLDER_CLASS,
composerPlainText,
deleteChipBeforeCaret,
deleteSelectionInEditor,
@@ -64,7 +68,7 @@ import {
import { useComposerScope } from './scope'
import { ComposerStatusStack } from './status-stack'
import { CodingStatusRow } from './status-stack/coding-row'
import { extractClipboardImageBlobs } from './text-utils'
import { extractClipboardImageBlobs, openDirectiveScope } from './text-utils'
import { ComposerTriggerPopover } from './trigger-popover'
import type { ChatBarProps } from './types'
import { isRedoShortcut, isUndoShortcut } from './undo-history'
@@ -95,11 +99,32 @@ export function ChatBar({
onSubmit: onSubmitProp,
onTranscribeAudio
}: ChatBarProps) {
// Typed stop phrase during an active voice conversation ends it — same
// semantics as SAYING "stop" (voice-stop-word.ts) or clicking the pill's
// end control. Populated after useComposerVoice below (the submit wrapper
// is created first); render-time assignment keeps the ref current.
const voiceStopRef = useRef<{ active: boolean; end: () => void }>({ active: false, end: () => {} })
// Every send (typed, queued, voice) passes through the contributed
// middleware chain first — rewrite / pass-through / cancel. Empty chain =
// exact pass-through, so surfaces without contributions are byte-identical.
const onSubmit = useCallback<ChatBarProps['onSubmit']>(
async (value, options) => {
// Bare stop phrase typed while the voice conversation is live: end the
// conversation (mic off, pill dismissed) instead of sending "stop" to
// the agent. Spoken transcripts are already stop-checked inside
// use-voice-conversation, so this only catches typed/queued sends.
// Outside a voice conversation, typed "stop" is a normal message.
const voiceStop = voiceStopRef.current
if (interceptsTypedVoiceStop(voiceStop.active, value, options?.attachments?.length ?? 0)) {
voiceStop.end()
// Consumed (not rejected): report accepted so the submit engine
// clears the draft instead of restoring "stop" into the composer.
return true
}
const draft = await runComposerMiddleware({ text: value, attachments: options?.attachments })
if (!draft) {
@@ -139,6 +164,9 @@ export function ChatBar({
useComposerMicroActions(statusSessionId, busy)
const composerRef = useRef<HTMLFormElement | null>(null)
// The dock wraps the strips + status stack + composer; the thread's bottom
// clearance measures this, while the pop-out drag still tracks the composer.
const composerDockRef = useRef<HTMLDivElement | null>(null)
const composerSurfaceRef = useRef<HTMLDivElement | null>(null)
// Pop-out engine: docked↔floating state, dock/float/toggle, drag gestures, and
@@ -162,6 +190,7 @@ export function ChatBar({
const { availableThemes, themeName } = useTheme()
const at = useAtCompletions({ gateway: gateway ?? null, sessionId: sessionId ?? null, cwd: cwd ?? null })
const slash = useSlashCompletions({ activeSkin: themeName, gateway: gateway ?? null, skinThemes: availableThemes })
const emoji = useEmojiCompletions()
const { t } = useI18n()
const gatewayState = useStore($gatewayState)
@@ -255,7 +284,14 @@ export function ChatBar({
return onCancel()
}, [activeQueueSessionKeyRef, onCancel])
const { compactPill, stacked } = useComposerMetrics({ composerRef, composerSurfaceRef, editorRef, poppedOut })
const { compactPill, stacked } = useComposerMetrics({
composerDockRef,
composerRef,
composerSurfaceRef,
editorRef,
poppedOut
})
const hasComposerPayload = hasText || attachments.length > 0
const canSubmit = busy || hasComposerPayload
@@ -324,7 +360,7 @@ export function ChatBar({
triggerItems,
triggerKeyConsumedRef,
triggerLoading
} = useComposerTrigger({ at, draftRef, editorRef, requestMainFocus, setComposerText, slash })
} = useComposerTrigger({ at, draftRef, editorRef, emoji, recordUndoPoint, requestMainFocus, setComposerText, slash })
// Pull the live contentEditable text into draftRef + the AUI composer state
// (which drives `hasComposerPayload` → the send button). Shared by the input
@@ -456,8 +492,13 @@ export function ChatBar({
// Links in the paste land as `@url:` chips rather than a wall of URL text —
// the same reference the "Add URL" dialog inserts, parsed in place so a link
// mid-sentence keeps its position. Bare `@path` tokens promote the same way.
// A paste into an open `@url:`/`@file:` scope CONSUMES that scope instead of
// stacking on it — the scope is the browse mode the user is pasting into,
// not text they typed and want to keep (`@url:@url:\`https://…\``).
const scope = openDirectiveScope(event.currentTarget)
recordUndoPoint()
insertComposerContentsAtCaret(event.currentTarget, pathifyRefs(linkifyUrls(pastedText)))
insertComposerContentsAtCaret(event.currentTarget, pathifyRefs(linkifyUrls(pastedText)), scope)
scheduleFlushEditorToDraft(event.currentTarget)
}
@@ -547,6 +588,17 @@ export function ChatBar({
return
}
// The popover is open but its items are still in flight (debounce + RPC).
// Tab must not fall through to the browser — it would move focus out of
// the composer mid-completion, which reads as the popover "eating" the
// keypress. Swallow it; the refresh lands with the items.
if (trigger && triggerLoading && triggerItems.length === 0 && event.key === 'Tab') {
event.preventDefault()
triggerKeyConsumedRef.current = true
return
}
if (trigger && triggerItems.length > 0) {
if (event.key === 'ArrowDown') {
event.preventDefault()
@@ -834,12 +886,19 @@ export function ChatBar({
focusInput,
insertText,
maxRecordingSeconds,
// Voice barge-in mid-generation halts the run like the Stop button.
onInterrupt: haltRun,
onSubmit,
onTranscribeAudio,
sessionId,
target: scope.target
})
// Keep the typed-stop interceptor (see onSubmit above) in sync with the
// live conversation state. Render-time ref assignment, same pattern as
// dispatchSubmitRef — no effect needed for a plain mirror.
voiceStopRef.current = { active: voiceConversationActive, end: endConversation }
const contextMenu = (
<ContextMenu
onInsertText={insertText}
@@ -888,7 +947,7 @@ export function ChatBar({
autoCorrect="off"
className={cn(
'min-h-[1.625rem] min-h-(--composer-input-min-height) max-h-(--composer-input-max-height) cursor-text overflow-y-auto whitespace-pre-wrap break-words [overflow-wrap:anywhere] bg-transparent pb-1 pr-1 pt-1 leading-normal text-foreground outline-none disabled:cursor-not-allowed',
'empty:before:content-[attr(data-placeholder)] empty:before:text-muted-foreground/60',
COMPOSER_PLACEHOLDER_CLASS,
'**:data-ref-text:cursor-default',
stacked && 'pl-3',
stacked ? 'w-full' : 'min-w-(--composer-input-inline-min-width) flex-1'
@@ -978,36 +1037,27 @@ export function ChatBar({
/>
)}
<ComposerPrimitive.Unstable_TriggerPopoverRoot>
<ComposerPrimitive.Root
{/* Dock column: owns the composer's POSITION and stacks, bottom-up,
[micro actions] · [status stack] · [composer] · [underside].
Anchored at the bottom, so in-flow children grow upward and still
overlay the thread — no absolute lane needed.
The strips are siblings of the composer, not children: the pop-out
drag region is `absolute inset-0` INSIDE the composer, so anything
rendered in there is inside the grab area by construction. Keeping
them out here is what makes that impossible rather than excluded. */}
<div
className={cn(
'group/composer z-30 overflow-visible rounded-2xl',
poppedOut
? // Floating: the composer (with its own border) floats with an even
// 5px transparent grab margin around it — drag that to move it.
'fixed w-[var(--composer-popout-width)] max-w-[calc(100vw-1.5rem)] bg-transparent p-[5px]'
: 'absolute bottom-0 left-1/2 w-[min(var(--composer-width),calc(100%-2rem))] max-w-full -translate-x-1/2 pt-2 pb-[var(--composer-shell-pad-block-end)]',
dragging && 'cursor-grabbing select-none touch-none'
'z-30 flex flex-col',
poppedOut ? 'fixed max-w-[calc(100vw-1.5rem)]' : 'absolute bottom-0 left-1/2 max-w-full -translate-x-1/2'
)}
data-drag-active={dragActive ? '' : undefined}
data-popped-out={poppedOut ? '' : undefined}
data-slot="composer-root"
data-status-stack={statusStackVisible ? '' : undefined}
data-slot="composer-dock"
data-thread-scrolled-up={scrolledUp ? '' : undefined}
onDragEnter={handleDragEnter}
onDragLeave={handleDragLeave}
onDragOver={handleDragOver}
onDrop={handleDrop}
onPointerDown={popoutAllowed ? onComposerGesturePointerDown : undefined}
onSubmit={e => {
e.preventDefault()
if (composingRef.current) {
return
}
submitDraft()
}}
ref={composerRef}
// Measured for the thread's bottom clearance: the dock is the box
// that contains the strips, the status stack, AND the composer, so
// one measurement covers everything the thread must clear.
ref={composerDockRef}
style={
poppedOut
? {
@@ -1019,22 +1069,16 @@ export function ChatBar({
: undefined
}
>
{isHelpHint && <HelpHint />}
{trigger && !argStageEmpty && (
<ComposerTriggerPopover
activeIndex={triggerActive}
items={triggerItems}
kind={trigger.kind}
loading={triggerLoading}
onHover={setTriggerActive}
onPick={replaceTriggerWithChip}
/>
)}
{/* Aligned to the composer SURFACE, which sits inside the composer's
5px transparent grab margin — so both strips carry the same inset
and share one left edge with it. */}
<div className={cn(composerFloatingStrip, 'px-[5px] pb-1.5 empty:hidden')}>
<ActionBadges sessionId={statusSessionId} />
</div>
{/* Session-scoped status stack (todos, subagents, background tasks,
queue). Out of flow so it never inflates the composer's measured
height; it overlays the chat instead of pushing it, and publishes
its own --status-stack-measured-height so the thread's clearance
accounts for it. Collapses to nothing when every status is empty. */}
queue). An in-flow dock child: the dock is bottom-anchored, so it
grows upward over the thread and the dock's own measurement covers
it. Collapses to nothing when every status is empty. */}
<ComposerStatusStack
queue={
activeQueueSessionKey && queuedPrompts.length > 0 ? (
@@ -1064,134 +1108,166 @@ export function ChatBar({
}
sessionId={statusSessionId}
/>
{!poppedOut && (
<div
className="pointer-events-none absolute inset-0 rounded-[inherit]"
style={{ background: COMPOSER_FADE_BACKGROUND }}
/>
)}
{/* Drag region: covers the transparent grab margin around the surface.
<ComposerPrimitive.Root
className={cn(
'group/composer relative w-full overflow-visible rounded-2xl',
poppedOut && 'bg-transparent',
dragging && 'cursor-grabbing select-none touch-none'
)}
data-drag-active={dragActive ? '' : undefined}
data-popped-out={poppedOut ? '' : undefined}
data-slot="composer-root"
data-status-stack={statusStackVisible ? '' : undefined}
data-thread-scrolled-up={scrolledUp ? '' : undefined}
onDragEnter={handleDragEnter}
onDragLeave={handleDragLeave}
onDragOver={handleDragOver}
onDrop={handleDrop}
onPointerDown={popoutAllowed ? onComposerGesturePointerDown : undefined}
onSubmit={e => {
e.preventDefault()
if (composingRef.current) {
return
}
submitDraft()
}}
ref={composerRef}
>
{isHelpHint && <HelpHint />}
{trigger && !argStageEmpty && (
<ComposerTriggerPopover
activeIndex={triggerActive}
items={triggerItems}
kind={trigger.kind}
loading={triggerLoading}
onHover={setTriggerActive}
onPick={replaceTriggerWithChip}
scope={trigger.scope}
/>
)}
{!poppedOut && (
<div
className="pointer-events-none absolute inset-0 rounded-[inherit]"
style={{ background: COMPOSER_FADE_BACKGROUND }}
/>
)}
{/* Drag region: covers the transparent grab margin around the surface.
The surface sits on top (z-4) so only the exposed ring receives this
element's hover/cursor — grab cursor + a diagonal hatch (/////)
appear when you hover the draggable margin, never over the input.
The hatch pattern + opacity ladder live in styles.css. */}
{popoutAllowed && (
<div
aria-hidden
className={cn('pointer-events-auto absolute inset-0', dragging ? 'cursor-grabbing' : 'cursor-grab')}
data-dragging={dragging ? '' : undefined}
data-slot="composer-drag-region"
onDoubleClick={event => {
// The pill strips paint above this region; a double-click that
// lands on one must not float the composer. onPointerDown goes
// through gestureTargetOk, but this handler doesn't.
if (!(event.target as Element).closest('[data-slot="composer-no-drag"]')) {
handleComposerToggle()
}
}}
/>
)}
<div className="relative w-full rounded-[inherit]">
<div
className={cn(
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
COMPOSER_DROP_FADE_CLASS,
dragActive && COMPOSER_DROP_ACTIVE_CLASS
)}
data-slot="composer-surface"
ref={composerSurfaceRef}
>
{popoutAllowed && (
<div
aria-hidden
className={cn(
'pointer-events-none absolute inset-0 -z-10 rounded-[inherit]',
composerFill,
composerSurfaceGlass
)}
/>
<CodingStatusRow
onBranchOff={handleBranchOff}
onConvertBranch={handleConvertBranch}
onListBranches={handleListBranches}
onOpen={toggleReview}
onOpenWorktree={openInWorktree}
onSwitchBranch={handleSwitchBranch}
repoPath={cwd}
className={cn('pointer-events-auto absolute inset-0', dragging ? 'cursor-grabbing' : 'cursor-grab')}
data-dragging={dragging ? '' : undefined}
data-slot="composer-drag-region"
onDoubleClick={handleComposerToggle}
/>
)}
<div className="relative w-full rounded-[inherit]">
<div
className={cn(
'relative z-1 flex min-h-0 w-full flex-col gap-(--composer-row-gap) overflow-hidden rounded-[inherit] px-(--composer-surface-pad-x) py-(--composer-surface-pad-y) transition-opacity duration-200 ease-out',
scrolledUp
? 'opacity-30 group-hover/composer:opacity-100 group-focus-within/composer-surface:opacity-100'
: 'opacity-100'
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
COMPOSER_DROP_FADE_CLASS,
dragActive && COMPOSER_DROP_ACTIVE_CLASS
)}
data-slot="composer-fade"
data-slot="composer-surface"
ref={composerSurfaceRef}
>
{/* Contribution seams: banners above, a row below, inline
additions beside the "+" menu and before the controls.
All four render nothing until something contributes. */}
<ContribSlot area={COMPOSER_AREAS.top} />
<VoiceActivity state={voiceActivityState} />
<VoicePlaybackActivity />
{queueEdit && editingQueuedPrompt && (
<div className="flex items-center justify-between gap-2 rounded-lg border border-[color-mix(in_srgb,var(--dt-composer-ring)_32%,transparent)] bg-accent/18 px-2 py-1">
<div className="min-w-0 text-[0.7rem] text-muted-foreground/88">
{t.composer.editingQueuedInComposer}
</div>
<div className="flex shrink-0 items-center gap-1">
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('cancel')}
type="button"
variant="ghost"
>
{t.common.cancel}
</Button>
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('save')}
type="button"
>
{t.common.save}
</Button>
</div>
</div>
)}
{attachments.length > 0 && <AttachmentList attachments={attachments} onRemove={onRemoveAttachment} />}
<div
aria-hidden
className={cn(
'pointer-events-none absolute inset-0 -z-10 rounded-[inherit]',
composerFill,
composerSurfaceGlass
)}
/>
<CodingStatusRow
onBranchOff={handleBranchOff}
onConvertBranch={handleConvertBranch}
onListBranches={handleListBranches}
// A tile's rail reviews ITS worktree: pin the pane's scope to
// this surface's cwd. Main keeps the classic follow-the-
// active-session scope (null).
onOpen={() => toggleReview(scope.target === 'main' ? null : (cwd ?? null))}
onOpenWorktree={openInWorktree}
onSwitchBranch={handleSwitchBranch}
repoPath={cwd}
/>
<div
className={cn(
'grid w-full',
stacked
? 'grid-cols-[auto_1fr] gap-(--composer-row-gap) [grid-template-areas:"input_input"_"menu_controls"]'
: 'grid-cols-[auto_1fr_auto] items-center gap-(--composer-control-gap) [grid-template-areas:"menu_input_controls"]'
'relative z-1 flex min-h-0 w-full flex-col gap-(--composer-row-gap) overflow-hidden rounded-[inherit] px-(--composer-surface-pad-x) py-(--composer-surface-pad-y) transition-opacity duration-200 ease-out',
scrolledUp
? 'opacity-30 group-hover/composer:opacity-100 group-focus-within/composer-surface:opacity-100'
: 'opacity-100'
)}
data-slot="composer-fade"
>
<div className="flex translate-y-[3px] items-start gap-(--composer-control-gap) self-start [grid-area:menu]">
{contextMenu}
<ContribSlot area={COMPOSER_AREAS.leading} />
</div>
<div className="min-w-0 [grid-area:input]">{input}</div>
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<ContribSlot area={COMPOSER_AREAS.actions} />
{controls}
{/* Contribution seams: banners above, a row below, inline
additions beside the "+" menu and before the controls.
All four render nothing until something contributes. */}
<ContribSlot area={COMPOSER_AREAS.top} />
<VoiceActivity state={voiceActivityState} />
<VoicePlaybackActivity />
{queueEdit && editingQueuedPrompt && (
<div className="flex items-center justify-between gap-2 rounded-lg border border-[color-mix(in_srgb,var(--dt-composer-ring)_32%,transparent)] bg-accent/18 px-2 py-1">
<div className="min-w-0 text-[0.7rem] text-muted-foreground/88">
{t.composer.editingQueuedInComposer}
</div>
<div className="flex shrink-0 items-center gap-1">
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('cancel')}
type="button"
variant="ghost"
>
{t.common.cancel}
</Button>
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('save')}
type="button"
>
{t.common.save}
</Button>
</div>
</div>
)}
{attachments.length > 0 && <AttachmentList attachments={attachments} onRemove={onRemoveAttachment} />}
<div
className={cn(
'grid w-full',
stacked
? 'grid-cols-[auto_1fr] gap-(--composer-row-gap) [grid-template-areas:"input_input"_"menu_controls"]'
: 'grid-cols-[auto_1fr_auto] items-center gap-(--composer-control-gap) [grid-template-areas:"menu_input_controls"]'
)}
>
<div className="flex translate-y-[3px] items-start gap-(--composer-control-gap) self-start [grid-area:menu]">
{contextMenu}
<ContribSlot area={COMPOSER_AREAS.leading} />
</div>
<div className="min-w-0 [grid-area:input]">{input}</div>
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<ContribSlot area={COMPOSER_AREAS.actions} />
{controls}
</div>
</div>
<ContribSlot area={COMPOSER_AREAS.bottom} />
</div>
<ContribSlot area={COMPOSER_AREAS.bottom} />
</div>
</div>
</div>
{/* Underside: a floating strip BELOW the whole composer surface.
Chrome-free by design — contributions bring their own pill/skin,
like the micro-action strip above. In flow (the root is
bottom-anchored, so this grows the composer upward and stays on
screen) but OUTSIDE the surface, so it escapes the surface's
clipping, border, and scroll fade. Shares the micro-action
strip's grid so the two bracket the composer on one vertical
line. Renders nothing until something contributes. */}
<div className={cn(composerFloatingStrip, 'pt-1.5 empty:hidden')} data-slot="composer-no-drag">
</ComposerPrimitive.Root>
{/* Underside: chrome-free strip BELOW the composer. Outside the root
for the same reason as the micro actions — it must not fall inside
the pop-out drag region. Same px as the strip above, so the two
bracket the composer on one vertical line. */}
<div className={cn(composerFloatingStrip, 'px-[5px] pt-1.5 empty:hidden')}>
<ContribSlot area={COMPOSER_AREAS.underside} />
</div>
</ComposerPrimitive.Root>
</div>
</ComposerPrimitive.Unstable_TriggerPopoverRoot>
<UrlDialog
@@ -0,0 +1,72 @@
import { describe, expect, it } from 'vitest'
import { refAttrs, refAttrsHtml } from '@/components/assistant-ui/directive-text'
import { REFERENCE_STYLES, referenceKind, referenceStyle } from '@/components/assistant-ui/reference-kinds'
/**
* There is ONE inline-reference system: `class="ref"` + `data-ref="<kind>"`.
* A pasted link, an `@file:` chip, a `/skill`, a `@session:` the agent wrote —
* all the same markup, styled by the `.ref` rules in styles.css.
*/
describe('the inline reference contract', () => {
it('marks any element as a reference of a given kind', () => {
expect(refAttrs('file')).toEqual({ className: 'ref', 'data-ref': 'file' })
expect(refAttrsHtml('skill')).toBe('class="ref" data-ref="skill"')
})
it('an unkinded reference is a plain link, not a broken one', () => {
// A bare external link has no kind — it keeps the default link colour
// rather than being tagged with a wrong one.
expect(refAttrs()).toEqual({ className: 'ref' })
expect(refAttrsHtml()).toBe('class="ref"')
})
it('normalises an unknown kind instead of emitting it raw', () => {
// A kind CSS has no rule for would silently render unstyled; coercing to
// `other` keeps it inside the system.
expect(refAttrs('wat')['data-ref']).toBe('other')
expect(referenceKind('wat')).toBe('other')
})
it('ships no colour from TypeScript — the theme owns every accent', () => {
// The whole point of keying on `data-ref`: a skin restyles all references
// at once, and no hex or color-mix() is hardcoded in a component.
for (const [kind, style] of Object.entries(REFERENCE_STYLES)) {
expect(style, `${kind} must not carry a colour`).not.toHaveProperty('color')
}
expect(JSON.stringify(refAttrs('url'))).not.toMatch(/color|#[0-9a-f]{3}/i)
})
it('gives every kind a glyph and a label', () => {
for (const [kind, style] of Object.entries(REFERENCE_STYLES)) {
expect(style.codicon, `${kind} codicon`).toBeTruthy()
expect(style.label, `${kind} label`).toBeTruthy()
// Emoji rows render the emoji itself instead of a glyph.
if (kind !== 'emoji') {
expect(style.paths.length, `${kind} paths`).toBeGreaterThan(0)
}
}
})
it('keeps commands and skills visually distinct', () => {
// Different data-ref values, so the stylesheet can accent them apart.
expect(refAttrs('skill')['data-ref']).not.toBe(refAttrs('command')['data-ref'])
expect(referenceStyle('skill').codicon).not.toBe(referenceStyle('command').codicon)
})
})
describe('references are text, not badges', () => {
it('carries no layout, padding, or background of its own', () => {
// Everything visual lives in the stylesheet. If a component starts adding
// its own chrome here, that's the drift this system exists to prevent.
const { className } = refAttrs('file')
expect(className).toBe('ref')
for (const chrome of ['bg-', 'rounded', 'px-', 'py-', 'border', 'inline-flex', 'text-[']) {
expect(className).not.toContain(chrome)
}
})
})
@@ -1,8 +1,9 @@
import { memo, useState } from 'react'
import { Codicon } from '@/components/ui/codicon'
import { useSessionSlice } from '@/lib/use-session-slice'
import { cn } from '@/lib/utils'
import type { ComposerAction } from '@/store/composer-actions'
import { $composerActionsBySession, type ComposerAction } from '@/store/composer-actions'
import { notifyError } from '@/store/notifications'
/**
@@ -31,19 +32,14 @@ const PILL = cn(
* (`composerFloatingStrip`), this owns only the pills, so the strip above the
* surface and the `composer.underside` strip below it can't drift apart.
*/
export const ActionBadges = memo(function ActionBadges({
actions,
sessionId
}: {
actions: ComposerAction[]
sessionId: string
}) {
export const ActionBadges = memo(function ActionBadges({ sessionId }: { sessionId: null | string }) {
const actions = useSessionSlice($composerActionsBySession, sessionId)
// A pill can kick off async work (a gateway call, a submit). Track which one
// is in flight so it can spin and lock instead of double-firing.
const [runningId, setRunningId] = useState<null | string>(null)
const run = async (action: ComposerAction) => {
if (runningId) {
if (runningId || !sessionId) {
return
}

Some files were not shown because too many files have changed in this diff Show More