Merge updated tool metrics into skill metrics

Signed-off-by: Alex Fournier <afournier@nvidia.com>

# Conflicts:
#	hermes_cli/observability/schemas/hermes.shared_metrics.v1.schema.json
#	scripts/smoke_nemo_relay_shared_metrics.py
#	tests/agent/test_skill_commands.py
#	tests/hermes_cli/test_relay_shared_metrics.py
#	tests/hermes_cli/test_relay_shared_metrics_runtime.py
#	tests/tools/test_skill_manager_tool.py
#	tests/tools/test_skill_usage.py
#	tests/tools/test_skills_tool.py
This commit is contained in:
Alex Fournier
2026-07-31 07:30:30 -07:00
2642 changed files with 61301 additions and 392241 deletions
+28
View File
@@ -53,7 +53,19 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures
# (connection reset, auth token timeout, rate limiting). The action
# generates a unique builder name per invocation so the retry doesn't
# collide with the failed first attempt. A genuine persistent failure
# still fails the job — only the first attempt has continue-on-error.
# Refs: docker/setup-buildx-action#510
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
# Build once, load into the local daemon for testing. Cached
@@ -146,7 +158,15 @@ jobs:
- name: Checkout trusted source
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
@@ -208,7 +228,15 @@ jobs:
pattern: digest-*
merge-multiple: true
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
+16 -2
View File
@@ -97,9 +97,18 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# The trailing extras beyond all/dev are the lazy-install features
# (tools/lazy_deps.py) that tests exercise for real: provider.anthropic,
# stt/tts.mistral, image.fal, terminal.modal, terminal.daytona,
# memory.hindsight, search.parallel. The hermetic test env forbids
# mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so the SDKs those tests need must be in the
# venv up front — resolved from uv.lock like everything else, which
# also honors the exact supply-chain pins these extras carry.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
@@ -216,9 +225,14 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# Same extras as the test job's sync above: the hermetic test env
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
# in the venv up front.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
+4
View File
@@ -190,3 +190,7 @@ infographics/
infograficos/
infografico/
native/fts5_cjk/*.so
# Runtime marker written by hermes update when a lazy dependency refresh is
# interrupted; consumed by launch-time recovery. Never commit it (was tracked
# by accident via 3a69e34702, removed in the #72002 salvage).
.lazy-refresh-incomplete
-2
View File
@@ -1,2 +0,0 @@
started=1785343166.9711895
pid=2093896
+146
View File
@@ -0,0 +1,146 @@
# Message reactions (desktop tapbacks)
Two-way emoji reactions on individual messages in the desktop transcript: the
user reacts to any message, the agent reacts to a user message, and both sides
read the other's reactions as conversational signal.
## What already exists
Hermes already models reactions on the **platform** side — the desktop is the
only surface without them.
| Surface | Reaction support | Where |
|---|---|---|
| Agent → platform message | `send_message(action="react"/"unreact")` | `tools/send_message_tool.py:266` `_handle_react()` |
| Photon / iMessage | tapbacks in + out, routed only for messages we sent | `plugins/platforms/photon/adapter.py:1240-1283` |
| Telegram | `setMessageReaction`, config-gated | `plugins/platforms/telegram/adapter.py:9669+` |
| Slack / Matrix / Feishu / Discord | inbound reaction events → hooks | `gateway/run.py:4688` `_handle_reaction_event()` → `HookRegistry.emit("reaction:added")` |
| Adapter contract | `add_reaction()` / `remove_reaction()` coroutines, `set_reaction_handler()` | `gateway/platforms/base.py:3330` |
| Core "affection" detector | regex on user text → `vibe`, drives CLI pet / TUI heart / desktop hearts | `agent/reactions.py`, `agent/turn_context.py:592-604` |
Two things follow from that table:
1. **The agent-facing verb already exists.** `send_message(action="react")` is
the established shape. A desktop reaction should extend that tool, not add a
new core tool — every new tool ships on every API call (AGENTS.md footprint
ladder).
2. **The inbound convention already exists.** Photon turns a tapback into a
normal message event with `reply_to_message_id` + `reply_to_is_own_message`,
and the gateway prefixes `[Replying to your previous message: "…"]`
(`gateway/run.py:13125-13132`). Desktop reactions should read the same way to
the model.
Nothing exists on the desktop side: `grep -ri reaction` across `apps/desktop`
finds only the pet-overlay hearts.
## Prior art
**iOS Tapback** ([Apple](https://support.apple.com/guide/iphone/react-with-tapbacks-iph018d3c336/ios)):
double-tap or touch-and-hold a message → floating pill above the bubble with
heart / thumbs-up / thumbs-down / haha / ‼️ / ❓, swipe left for suggested emoji
and stickers, or tap the emoji button for the full keyboard. **One tapback per
message per person** — tapping the same one again removes it, tapping a
different one replaces it. Multiple people's tapbacks stack on the badge.
**Platform data models** converge on the same shape:
| Platform | Model | Add / remove |
|---|---|---|
| Slack | `{name, count, users[]}` | [`reactions.add`](https://docs.slack.dev/reference/methods/reactions.add) / `reactions.remove`, emits `reaction_added` |
| Discord | `{emoji, count, me}` on the message object | `PUT`/`DELETE .../reactions/{emoji}/@me` |
| Telegram | `reaction: [{type:"emoji", emoji:"👍"}]` — replaces the whole set | `setMessageReaction`, `is_big` for the big animation |
Telegram's "set the whole array" is the closest match to iOS semantics and the
simplest thing to persist.
**assistant-ui has no reaction primitive.** `@assistant-ui/react` 0.14.24 (MIT,
vendored at `apps/desktop/node_modules`): zero hits for "reaction" in `core/src`,
`react/src`, `dist/`, or the 2.2 MB `llms-full.txt` docs dump. What exists is a
hard-coded binary `FeedbackAdapter` (`"positive" | "negative"`,
`core/src/adapters/feedback.ts`) that throws when unconfigured and only writes
back onto assistant messages. Not usable for emoji, not usable on user messages.
**But `metadata.custom` is the supported extension channel** and this repo
already uses it: `ThreadUserMessage`/`ThreadAssistantMessage`/`ThreadSystemMessage`
all carry `metadata.custom: Record<string, unknown>` (`core/src/types/message.ts:319-366`),
and `chat-runtime.ts:397` already ships `custom: { attachmentRefs }` through it.
**Emoji picker survey** (npm week of 2026-07-22, sizes measured from the
published ESM entry):
| Library | License | Weekly DL | gzip | Headless | Latest |
|---|---|---|---|---|---|
| **frimousse** | MIT | 573k | **8.5 kB** | ✅ fully unstyled, composable parts | 0.3.0 · 2025-07-15 |
| emoji-picker-react | MIT | 1.31M | 87 kB | ❌ own CSS-in-JS (flairup) | 4.19.1 · 2026-04-27 |
| emoji-mart | MIT | 2.22M | ~120 kB w/ data | ❌ Preact + shadow styling | 5.6.0 · **2024-04-25**, 217 open issues |
| emoji-picker-element | Apache-2.0 | 183k | — | ❌ Web Component / Shadow DOM | 1.29.1 · 2026-03-01 |
No picker is currently a dependency (only `emoji-regex`, transitive). Already
paid for and reusable: `radix-ui` (Popover), `motion`, `@tanstack/react-virtual`,
Tailwind v4.
## Recommendation
**Hand-roll the tapback pill; add frimousse only behind the "+".** Six fixed
emoji in a pill is ~40 lines of JSX against existing tokens — pulling 87 kB of
`emoji-picker-react` to render six buttons, plus a CSS engine that fights
`DESIGN.md`, is backwards. frimousse is headless, dependency-free, 10× smaller,
and exposes `emojibaseUrl` so the data can be bundled as a Vite asset instead of
hitting jsDelivr (Electron must work offline).
### Data model
One reaction per author per message, Telegram-style whole-set replacement:
```ts
type MessageReaction = { emoji: string; author: 'user' | 'agent'; at: number }
```
Persisted in the existing `messages.display_metadata` JSON column
(`hermes_state_common.py:215`) — no new table. It already survives insert,
compaction, and every read projection, and
`set_latest_matching_message_display_kind()` (`hermes_state.py:5292`) is the
precedent for stamping metadata onto an already-persisted row.
### Model context
Reactions must reach the model **without breaking prompt caching**. The
`api_messages` build loop strips `display_metadata` from every outgoing copy
(`agent/conversation_loop.py:1443-1446`) precisely so display state never
becomes a provider field. Two candidate paths:
| Path | Cache impact | Notes |
|---|---|---|
| Rewrite the reacted-to message's content to carry the annotation | **Breaks the cached prefix** — mutates past context | Rejected. AGENTS.md: prompt caching is sacred. |
| Deliver the reaction as the *next* turn's leading annotation, mirroring photon | Prefix untouched; only the new turn carries it | Matches `[Replying to your previous message: "…"]` (`gateway/run.py:13125`), which the agent already understands |
The second is the same trick the platform adapters already use, so the model
sees a familiar shape and no existing conversation is rewritten.
### Attach points
| Concern | File | Lines |
|---|---|---|
| Assistant hover bar | `apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx` | 134–175 |
| User hover cluster | `apps/desktop/src/components/assistant-ui/thread/user-message.tsx` | 296–336 |
| Callback threading (ref caveat 79–99) | `apps/desktop/src/components/assistant-ui/thread/index.tsx` | 109–133 |
| `metadata.custom` → runtime | `apps/desktop/src/lib/chat-runtime.ts` | 384–432 |
| RPC client ↔ server pattern | `sidebar/session-actions-menu.tsx:62-89` ↔ `tui_gateway/server.py:8322` | — |
| Persistence | `hermes_state_common.py:192-216`, `hermes_state.py:5292-5324` | — |
| Prompt injection / strip | `agent/conversation_loop.py` | 1430–1529 |
### Known gaps to solve first
- **No durable message id crosses the gateway RPC path.** `_history_to_messages()`
(`tui_gateway/server.py:6545`) builds `{"role", "text"}` and drops the id. The
REST path carries `messages.id` incidentally via `SELECT *` but TS
`SessionMessage` (`types/hermes.ts:513-533`) doesn't declare it. Renderer ids
are ephemeral and change shape between rehydrated (`<ts>-<i>-<role>`), live
(`assistant-<ms>`), and optimistic (`user-<ms>-<rand>`) messages. A reaction
needs a stable key — this is the first thing to fix.
- **WeakMap identity cache** in `apps/desktop/src/app/chat/runtime-repository.ts:26-66`
keys normalized `ThreadMessage` by `ChatMessage` identity. A reaction change
must produce a **new** `ChatMessage` object or the UI renders stale.
- **Rewind rewrites rows** (`replace_messages`), so anything keyed by row id
needs cascade handling — an argument for keeping reactions in
`display_metadata` on the row itself rather than a side table.
+3 -2
View File
@@ -1284,14 +1284,15 @@ def profile_env(tmp_path, monkeypatch):
### Python
**ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces
hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8,
`-n auto` xdist workers, in-tree subprocess-isolation plugin). Direct `pytest`
per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist,
worker count auto-scaled from CPU count). Direct `pytest`
on a 16+ core developer machine with API keys set diverges from CI in ways
that have caused multiple "works locally, fails in CI" incidents (and the reverse).
```bash
scripts/run_tests.sh # full suite, CI-parity
scripts/run_tests.sh tests/gateway/ # one directory
scripts/run_tests.sh tests/agent/test_foo.py::test_x # one test
scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular)
scripts/run_tests.sh -v --tb=long # pass-through pytest flags
```
+3 -2
View File
@@ -201,7 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes
### Run tests
```bash
# Preferred — matches CI (hermetic env, 4 xdist workers); see AGENTS.md
# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation
# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md
scripts/run_tests.sh
# Alternative (activate the venv first). The wrapper is still recommended
@@ -848,7 +849,7 @@ that touches the OS, assume *any* platform can hit your code path.
Tests that use POSIX-only syscalls need a skip marker. Common ones:
- Symlinks → `@pytest.mark.skipif(sys.platform == "win32", ...)`
- `0o600` file modes → `@pytest.mark.skipif(sys.platform.startswith("win"), ...)`
- `signal.SIGALRM` → Unix-only (see `tests/conftest.py::_enforce_test_timeout`)
- `signal.SIGALRM` → Unix-only (per-test timeouts no longer use it directly; see the win32 timeout-method shim in `tests/conftest.py::pytest_configure`)
- `os.setsid` / `os.fork` → Unix-only
- Live Winsock / Windows-specific regression tests →
`@pytest.mark.skipif(sys.platform != "win32", reason="Windows-specific regression")`
+1 -1
View File
@@ -26,7 +26,7 @@ Use any model you want — [Nous Portal](https://portal.nousresearch.com), OpenR
<tr><td><b>A closed learning loop</b></td><td>Agent-curated memory with periodic nudges. Autonomous skill creation after complex tasks. Skills self-improve during use. FTS5 session search with LLM summarization for cross-session recall. <a href="https://github.com/plastic-labs/honcho">Honcho</a> dialectic user modeling. Compatible with the <a href="https://agentskills.io">agentskills.io</a> open standard.</td></tr>
<tr><td><b>Scheduled automations</b></td><td>Built-in cron scheduler with delivery to any platform. Daily reports, nightly backups, weekly audits — all in natural language, running unattended.</td></tr>
<tr><td><b>Delegates and parallelizes</b></td><td>Spawn isolated subagents for parallel workstreams. Write Python scripts that call tools via RPC, collapsing multi-step pipelines into zero-context-cost turns.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Six terminal backends — local, Docker, SSH, Singularity, Modal, and Daytona. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Seven terminal backends — local, Docker, SSH, Singularity, Modal, Daytona, and Vercel Sandbox. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Research-ready</b></td><td>Batch trajectory generation, trajectory compression for training the next generation of tool-calling models.</td></tr>
</table>
+2 -2
View File
@@ -114,6 +114,7 @@ def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, st
load_config,
)
from hermes_cli.models import fetch_api_models
from hermes_cli.providers import custom_provider_slug
except ImportError:
return []
@@ -145,8 +146,7 @@ def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, st
base_url = str(entry.get("base_url", "") or "").strip()
if not name or not base_url:
continue
slug_source = provider_key or name
slug = "custom:" + slug_source.strip().lower().replace(" ", "-")
slug = custom_provider_slug(name, provider_key)
api_key = str(entry.get("api_key", "") or "").strip()
if not api_key:
+43 -11
View File
@@ -455,8 +455,7 @@ def init_agent(
command: str = None,
args: list[str] | None = None,
model: str = "",
max_iterations: int = 500, # Default tool-calling iterations (shared with subagents)
tool_delay: float = 1.0,
max_iterations: int = 90, # Default tool-calling iterations (shared with subagents)
enabled_toolsets: List[str] = None,
disabled_toolsets: List[str] = None,
save_trajectories: bool = False,
@@ -529,8 +528,7 @@ def init_agent(
requested_provider (str): Original provider identity before runtime canonicalization
api_mode (str): API mode override: "chat_completions" or "codex_responses"
model (str): Model name to use (default: "anthropic/claude-opus-4.6")
max_iterations (int): Maximum number of tool calling iterations (default: 500)
tool_delay (float): Delay between tool calls in seconds (default: 1.0)
max_iterations (int): Maximum number of tool calling iterations (default: 90)
enabled_toolsets (List[str]): Only enable tools from these toolsets (optional)
disabled_toolsets (List[str]): Disable tools from these toolsets (optional)
save_trajectories (bool): Whether to save conversation trajectories to JSONL files (default: False)
@@ -576,7 +574,6 @@ def init_agent(
# Shared iteration budget — parent creates, children inherit.
# Consumed by every LLM turn across parent + all subagents.
agent.iteration_budget = iteration_budget or IterationBudget(max_iterations)
agent.tool_delay = tool_delay
agent.save_trajectories = save_trajectories
agent.verbose_logging = verbose_logging
agent.quiet_mode = quiet_mode
@@ -843,7 +840,7 @@ def init_agent(
# sessions with >5-minute pauses between turns (#14971).
agent._cache_ttl = "5m"
try:
from hermes_cli.config import load_config as _load_pc_cfg
from hermes_cli.config import load_config_readonly as _load_pc_cfg
_pc_cfg = _load_pc_cfg().get("prompt_caching", {}) or {}
_ttl = _pc_cfg.get("cache_ttl", "5m")
@@ -1093,7 +1090,7 @@ def init_agent(
# Guardrail config — read from config.yaml at init time.
agent._bedrock_guardrail_config = None
try:
from hermes_cli.config import load_config as _load_br_cfg
from hermes_cli.config import load_config_readonly as _load_br_cfg
_gr = _load_br_cfg().get("bedrock", {}).get("guardrail", {})
if _gr.get("guardrail_identifier") and _gr.get("guardrail_version"):
agent._bedrock_guardrail_config = {
@@ -1164,8 +1161,8 @@ def init_agent(
client_kwargs["default_headers"] = hermes_xai_default_headers()
elif "default_headers" not in client_kwargs:
# Fall back to profile.default_headers for providers that
# declare custom headers (e.g. Kimi User-Agent on non-kimi.com
# endpoints).
# declare custom headers (e.g. Vercel AI Gateway attribution,
# Kimi User-Agent on non-kimi.com endpoints).
try:
from providers import get_provider_profile as _gpf
_ph = _gpf(agent.provider)
@@ -1480,7 +1477,7 @@ def init_agent(
# reads the JSON files directly. See run_agent._save_session_log.
agent._session_json_enabled = False
try:
from hermes_cli.config import load_config as _load_sess_cfg
from hermes_cli.config import load_config_readonly as _load_sess_cfg
_sess_cfg = (_load_sess_cfg().get("sessions") or {})
agent._session_json_enabled = bool(_sess_cfg.get("write_json_snapshots", False))
except Exception:
@@ -1551,7 +1548,7 @@ def init_agent(
# Load config once for memory, skills, and compression sections
try:
from hermes_cli.config import load_config as _load_agent_config
from hermes_cli.config import load_config_readonly as _load_agent_config
_agent_cfg = _load_agent_config()
except Exception:
_agent_cfg = {}
@@ -1966,6 +1963,31 @@ def init_agent(
compression_in_place = is_truthy_value(
_compression_cfg.get("in_place"), default=True
)
# Opt-in (default False): a micro-compaction pass rewrites already-sent
# history every turn, which breaks the provider prompt-cache prefix on a
# per-turn cadence rather than at an episodic boundary. That is the cost
# `proactive_prune_min_reclaim_tokens` exists to amortize, so the feature
# stays off until an operator opts in and accepts the tradeoff.
compression_micro_compact = is_truthy_value(
_compression_cfg.get("micro_compact"), default=False
)
# How often a pass runs, in completed turns. Each pass rewrites
# already-sent history and costs one prompt-cache break, so this is the
# dial for how often that cost is paid: 1 = every turn (most aggressive
# reclaim), 5 = one break per five turns. Clamped to >= 1.
compression_micro_compact_every_n_turns = max(
1,
_parse_prune_int(_compression_cfg.get("micro_compact_every_n_turns", 1), 1),
)
# Rolling-summary defrag threshold, in tokens. Lived on the compressor as
# a hardcoded attribute with no path from config until now.
compression_micro_compact_defrag_tokens = max(
1,
_parse_prune_int(
_compression_cfg.get("micro_compact_defrag_threshold_tokens", 2000),
2000,
),
)
codex_app_server_auto_compaction = str(
_compression_cfg.get("codex_app_server_auto", "native") or "native"
).lower()
@@ -2421,6 +2443,16 @@ def init_agent(
pass
agent.compression_enabled = compression_enabled
agent.compression_in_place = compression_in_place
# Apply micro-compaction settings to the compressor (feature is opt-in)
_cc = getattr(agent, "context_compressor", None)
if _cc is not None and hasattr(_cc, "_micro_compact_enabled"):
_cc._micro_compact_enabled = compression_micro_compact
if _cc is not None and hasattr(_cc, "_micro_compact_every_n_turns"):
_cc._micro_compact_every_n_turns = compression_micro_compact_every_n_turns
if _cc is not None and hasattr(_cc, "_micro_compact_defrag_threshold_tokens"):
_cc._micro_compact_defrag_threshold_tokens = (
compression_micro_compact_defrag_tokens
)
agent.codex_app_server_auto_compaction = codex_app_server_auto_compaction
agent.max_compression_attempts = compression_max_attempts
agent.compression_idle_compact_after_seconds = (
+52 -102
View File
@@ -249,12 +249,42 @@ def sanitize_tool_call_arguments(
*,
logger=None,
session_id: str = None,
cursor: Optional[dict] = None,
) -> int:
"""Repair corrupted assistant tool-call argument JSON in-place."""
"""Repair corrupted assistant tool-call argument JSON in-place.
``cursor`` (optional) is a caller-owned dict used to skip re-validating
messages already validated on a previous call. It stores, under
``"prefix"``, the exact message *objects* (strong references) validated
last time, in order. On the next call, the longest contiguous prefix of
``messages`` whose objects are ``is``-identical to the stored prefix is
skipped; scanning starts at the first divergence (conservative: any
reordering, truncation, compression rewrite, or mid-list insertion breaks
identity at that index and everything from there is re-scanned).
Safety argument for skipping: a message in the matched prefix was fully
scanned before — every tool_call argument was either already valid JSON
or was rewritten to ``"{}"`` (valid). The only code paths that mutate
``function["arguments"]`` on live history dicts between calls are the
surrogate / non-ASCII sanitizers, which substitute characters *inside*
JSON string values and cannot invalidate JSON syntax. Compression,
repair, undo, and steer paths replace or reorder message dicts, which
breaks the identity match and forces a re-scan. Holding strong
references (the objects themselves, not ``id()``s) makes address reuse
aliasing (#50372-style) impossible.
"""
log = logger or logging.getLogger(__name__)
if not isinstance(messages, list):
return 0
start_index = 0
if cursor is not None:
prev_prefix = cursor.get("prefix")
if isinstance(prev_prefix, list):
limit = min(len(prev_prefix), len(messages))
while start_index < limit and messages[start_index] is prev_prefix[start_index]:
start_index += 1
repaired = 0
marker = _ra().AIAgent._TOOL_CALL_ARGUMENTS_CORRUPTION_MARKER
@@ -275,7 +305,7 @@ def sanitize_tool_call_arguments(
existing_text = str(existing)
tool_msg["content"] = f"{marker}\n{existing_text}"
message_index = 0
message_index = start_index
while message_index < len(messages):
msg = messages[message_index]
if not isinstance(msg, dict) or msg.get("role") != "assistant":
@@ -356,6 +386,12 @@ def sanitize_tool_call_arguments(
message_index += 1
if cursor is not None:
# Strong references to the exact objects validated this call, in
# order. Any future divergence (compression, undo, repair, steer)
# breaks identity at the divergent index and re-scans from there.
cursor["prefix"] = messages[:]
return repaired
@@ -3295,89 +3331,17 @@ def intent_ack_continuation_enabled(agent) -> bool:
def copy_reasoning_content_for_api(agent, source_msg: dict, api_msg: dict) -> None:
"""Copy provider-facing reasoning fields onto an API replay message."""
if source_msg.get("role") != "assistant":
return
"""Copy provider-facing reasoning fields onto an API replay message.
needs_thinking_pad = agent._needs_thinking_reasoning_pad()
Forwarder — the strip-vs-repad POLICY is owned by
``agent.message_sanitization.apply_reasoning_content_policy`` (audit F4);
this only supplies the agent's cached provider-direction flag.
"""
from agent.message_sanitization import apply_reasoning_content_policy
# 1. Explicit reasoning_content already set.
#
# When the active provider enforces the thinking-mode echo-back
# (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their
# own space-placeholder written at creation time and any valid reasoning
# from the same provider. Sessions persisted BEFORE #17341 have
# empty-string placeholders pinned at creation time; DeepSeek V4 Pro
# rejects those with HTTP 400, so upgrade "" → " " on replay.
#
# When the active provider does NOT enforce echo-back, strip the field
# entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq,
# SambaNova, …) reject ANY reasoning_content key in input messages with
# HTTP 400/422 ("Extra inputs are not permitted"), even an empty string
# or a single-space pad. This is the cross-provider fallback case: a
# reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a
# fallback to a strict provider replays that pad and 422s. Stripping
# here covers the rebuild path; reapply_reasoning_echo_for_provider()
# covers the already-built api_messages path. Refs #45655.
existing = source_msg.get("reasoning_content")
if isinstance(existing, str):
if not needs_thinking_pad:
api_msg.pop("reasoning_content", None)
elif existing == "":
api_msg["reasoning_content"] = " "
else:
api_msg["reasoning_content"] = existing
return
# 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi,
# if the source turn has tool_calls AND a 'reasoning' field but no
# 'reasoning_content' key, the 'reasoning' text was written by a
# prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message
# pins reasoning_content at creation time for tool-call turns, so the
# shape (reasoning set, reasoning_content absent, tool_calls present)
# is unreachable from same-provider DeepSeek history after this fix.
# Inject a single space to satisfy the API without leaking another
# provider's chain of thought to DeepSeek/Kimi. Space (not "")
# because DeepSeek V4 Pro rejects empty-string reasoning_content
# in thinking mode (refs #17341).
normalized_reasoning = source_msg.get("reasoning")
if (
needs_thinking_pad
and source_msg.get("tool_calls")
and isinstance(normalized_reasoning, str)
and normalized_reasoning
):
api_msg["reasoning_content"] = " "
return
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content'
# for providers that use the internal 'reasoning' key.
# This must happen before the unconditional empty-string fallback so
# genuine reasoning content is not overwritten (#15812 regression in
# PR #15478). Only promote for providers that enforce echo-back —
# strict providers reject the field (refs #45655).
if isinstance(normalized_reasoning, str) and normalized_reasoning:
if needs_thinking_pad:
api_msg["reasoning_content"] = normalized_reasoning
else:
api_msg.pop("reasoning_content", None)
return
# 4. DeepSeek / Kimi thinking mode: all assistant messages need
# reasoning_content. Inject a single space to satisfy the provider's
# requirement when no explicit reasoning content is present. Covers
# both tool-call turns (already-poisoned history with no reasoning
# at all) and plain text turns. Space (not "") because DeepSeek V4
# Pro tightened validation and rejects empty string with HTTP 400
# ("The reasoning content in the thinking mode must be passed back
# to the API"). Refs #17341.
if needs_thinking_pad:
api_msg["reasoning_content"] = " "
return
# 5. reasoning_content was present but not a string (e.g. None after
# context compaction). Don't pass null to the API.
api_msg.pop("reasoning_content", None)
apply_reasoning_content_policy(
source_msg, api_msg, agent._needs_thinking_reasoning_pad()
)
def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int:
@@ -3409,25 +3373,11 @@ def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int:
Returns the number of assistant turns whose reasoning_content was added or
removed.
"""
needs_pad = agent._needs_thinking_reasoning_pad()
changed = 0
for api_msg in api_messages:
if api_msg.get("role") != "assistant":
continue
if needs_pad:
if api_msg.get("reasoning_content"):
continue
copy_reasoning_content_for_api(agent, api_msg, api_msg)
if api_msg.get("reasoning_content"):
changed += 1
else:
# Strict provider — strip any stale reasoning_content pad left
# over from a reasoning primary so the fallback request doesn't
# 400/422 on it.
if "reasoning_content" in api_msg:
api_msg.pop("reasoning_content", None)
changed += 1
return changed
from agent.message_sanitization import reapply_reasoning_echo
return reapply_reasoning_echo(
api_messages, agent._needs_thinking_reasoning_pad()
)
def _iter_httpx_pool_objects(http_client: Any):
+50 -18
View File
@@ -543,6 +543,7 @@ _API_KEY_PROVIDER_AUX_MODELS_FALLBACK: Dict[str, str] = {
"kimi-coding-cn": "kimi-k2-turbo-preview",
"gmi": "google/gemini-3.1-flash-lite-preview",
"anthropic": "claude-haiku-4-5-20251001",
"ai-gateway": "google/gemini-3-flash",
"opencode-zen": "gemini-3-flash",
"opencode-go": "glm-5",
"kilocode": "google/gemini-3.6-flash",
@@ -676,15 +677,15 @@ def build_or_headers(or_config: dict | None = None) -> dict:
Overrides ``openrouter.response_cache_ttl`` in config.yaml.
*or_config* is the ``openrouter`` section from config.yaml. When *None*,
falls back to reading config from disk via ``load_config()``.
falls back to reading config from disk via ``load_config_readonly()``.
"""
headers = dict(_OR_HEADERS_BASE)
# Resolve config from disk if not provided.
if or_config is None:
try:
from hermes_cli.config import load_config
or_config = load_config().get("openrouter", {})
from hermes_cli.config import load_config_readonly
or_config = load_config_readonly().get("openrouter", {})
except Exception:
or_config = {}
@@ -729,6 +730,15 @@ def build_nvidia_nim_headers(base_url: str | None) -> dict:
return {}
# Vercel AI Gateway app attribution headers. HTTP-Referer maps to
# referrerUrl and X-Title maps to appName in the gateway's analytics.
from hermes_cli import __version__ as _HERMES_VERSION
_AI_GATEWAY_HEADERS = {
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
}
# Nous Portal extra_body for product attribution.
# Callers should pass this as extra_body in chat.completions.create()
@@ -2317,8 +2327,8 @@ def _read_main_model() -> str:
if isinstance(override, str) and override.strip():
return override.strip()
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
model_cfg = cfg.get("model", {})
if isinstance(model_cfg, str) and model_cfg.strip():
return model_cfg.strip()
@@ -2344,8 +2354,8 @@ def _read_main_provider() -> str:
if isinstance(override, str) and override.strip():
return override.strip().lower()
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
model_cfg = cfg.get("model", {})
if isinstance(model_cfg, dict):
provider = model_cfg.get("provider", "")
@@ -2512,6 +2522,7 @@ def _relay_auxiliary_call(callback):
"attempt_count": 0,
"provider": "",
"model": "",
"response_model": None,
"api_mode": "chat_completions",
})
try:
@@ -2537,6 +2548,7 @@ def _relay_auxiliary_call_async(callback):
"attempt_count": 0,
"provider": "",
"model": "",
"response_model": None,
"api_mode": "chat_completions",
})
try:
@@ -2560,6 +2572,7 @@ def _set_relay_auxiliary_route(
return
context["provider"] = str(provider or "auxiliary")
context["model"] = str(model or "unknown")
context["response_model"] = None
context["api_mode"] = str(api_mode or "chat_completions")
@@ -3040,12 +3053,12 @@ def _try_azure_foundry(
try:
from hermes_cli.runtime_provider import _resolve_azure_foundry_runtime
from hermes_cli.auth import AuthError
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
except ImportError:
return None, None
try:
cfg = load_config()
cfg = load_config_readonly()
model_cfg = cfg.get("model") if isinstance(cfg, dict) else {}
if not isinstance(model_cfg, dict):
model_cfg = {}
@@ -3159,8 +3172,8 @@ def _try_anthropic(explicit_api_key: str = None) -> Tuple[Optional[Any], Optiona
# see issue #52608.
base_url = _pool_runtime_base_url(entry, _ANTHROPIC_DEFAULT_BASE_URL) if pool_present else _ANTHROPIC_DEFAULT_BASE_URL
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
model_cfg = cfg.get("model")
if isinstance(model_cfg, dict):
cfg_provider = str(model_cfg.get("provider") or "").strip().lower()
@@ -4764,10 +4777,10 @@ def _try_main_fallback_chain(
participate in the same order as the main agent.
"""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
from hermes_cli.fallback_config import get_fallback_chain
chain = get_fallback_chain(load_config())
chain = get_fallback_chain(load_config_readonly())
except Exception as exc:
logger.debug("Auxiliary %s: could not load main fallback chain: %s", task or "call", exc)
return None, None, ""
@@ -5725,7 +5738,8 @@ def resolve_provider_client(
else:
# Fall back to profile.default_headers for providers that declare
# client-level attribution headers on their profile (e.g. GMI
# User-Agent for traffic identification).
# User-Agent for traffic identification, Vercel AI Gateway
# Referer/Title for analytics).
try:
from providers import get_provider_profile as _gpf_main
_ph_main = _gpf_main(provider)
@@ -5986,11 +6000,11 @@ def _main_model_supports_vision(provider: str, model: Optional[str]) -> bool:
"""
try:
from agent.image_routing import _lookup_supports_vision
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
except ImportError:
return True
try:
supports = _lookup_supports_vision(provider, model, load_config())
supports = _lookup_supports_vision(provider, model, load_config_readonly())
except Exception: # pragma: no cover - defensive
return True
if supports is None:
@@ -6959,8 +6973,8 @@ def _get_auxiliary_task_config(task: str) -> Dict[str, Any]:
if not task:
return {}
try:
from hermes_cli.config import load_config
config = load_config()
from hermes_cli.config import load_config_readonly
config = load_config_readonly()
except ImportError:
return {}
aux = config.get("auxiliary", {}) if isinstance(config, dict) else {}
@@ -7463,6 +7477,7 @@ def _validate_llm_response(
except (AttributeError, TypeError, IndexError) as exc:
recovered = _recover_aux_response_message(response)
if recovered is not None:
_record_relay_auxiliary_response_model(response)
_complete_relay_auxiliary_call()
return recovered
response_type = type(response).__name__
@@ -7473,6 +7488,7 @@ def _validate_llm_response(
f"Expected object with .choices[0].message — check provider "
f"adapter or custom endpoint compatibility."
) from exc
_record_relay_auxiliary_response_model(response)
_complete_relay_auxiliary_call()
return response
@@ -7487,9 +7503,25 @@ def _complete_relay_auxiliary_call(*, outcome: str = "success") -> None:
relay_llm.complete_logical_call(
str(context.get("request_id") or ""),
outcome=outcome,
model_name=str(context.get("model") or "unknown"),
provider_name=str(context.get("provider") or "auxiliary"),
response_model_name=context.get("response_model"),
)
def _record_relay_auxiliary_response_model(response: Any) -> None:
"""Retain the provider-reported model for terminal route attribution."""
context = _RELAY_AUX_CALL_CONTEXT.get()
if context is None:
return
if isinstance(response, dict):
model = response.get("model")
else:
model = getattr(response, "model", None)
if isinstance(model, str) and model.strip():
context["response_model"] = model
def _fail_relay_auxiliary_call() -> None:
"""Close a terminally failed call without replacing its original error."""
try:
+2 -2
View File
@@ -70,8 +70,8 @@ def _resolve_review_runtime(agent: Any) -> Dict[str, Any]:
"routed": False,
}
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception:
return parent
aux = cfg.get("auxiliary", {}) if isinstance(cfg.get("auxiliary"), dict) else {}
+1 -1
View File
@@ -34,7 +34,7 @@ from __future__ import annotations
import logging
import math
import os
from dataclasses import dataclass, field
from dataclasses import dataclass
from typing import Any, Optional
logger = logging.getLogger(__name__)
+1 -1
View File
@@ -17,7 +17,7 @@ from __future__ import annotations
import logging
import os
import uuid
from dataclasses import dataclass, field
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import Any, Optional
+2 -2
View File
@@ -4028,9 +4028,9 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
# env var ``HERMES_LOCAL_STREAM_STALE_TIMEOUT`` overrides for escape-hatch.
_local_default = 900.0
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
_cfg = load_config()
_cfg = load_config_readonly() # read-only consumer — no deepcopy
_agent_cfg = _cfg.get("agent") if isinstance(_cfg, dict) else None
if isinstance(_agent_cfg, dict):
_v = _agent_cfg.get("local_stream_stale_timeout")
+5 -4
View File
@@ -18,6 +18,7 @@ import uuid
from types import SimpleNamespace
from typing import Any, Dict, List, Optional
from agent.message_sanitization import deterministic_call_id
from agent.prompt_builder import DEFAULT_AGENT_IDENTITY
logger = logging.getLogger(__name__)
@@ -182,13 +183,13 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str:
def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Thin wrapper over the single policy owner
``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept
as a module-level name because run_agent and tests import it from here.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return f"call_{digest}"
return deterministic_call_id(fn_name, arguments, index)
def _clamp_responses_call_id(call_id: str) -> str:
-1
View File
@@ -18,7 +18,6 @@ from __future__ import annotations
import json
import logging
import os
import time
from types import SimpleNamespace
from typing import Any, Callable, Dict, List
+2 -2
View File
@@ -337,9 +337,9 @@ def _coding_mode(config: Optional[dict[str, Any]]) -> str:
"""Return the normalized ``agent.coding_context`` mode (auto/focus/on/off)."""
if config is None:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
config = load_config()
config = load_config_readonly()
except Exception:
config = {}
raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto")
+879 -6
View File
@@ -137,6 +137,12 @@ LEGACY_SUMMARY_PREFIX = "[CONTEXT SUMMARY]:"
# "is_compressed_summary" would reach the wire and trip exactly that.
COMPRESSED_SUMMARY_METADATA_KEY = "_compressed_summary"
COMPRESSED_SUMMARY_HAS_USER_TURN_KEY = "_compressed_summary_has_user_turn"
# Distinguishes rolling micro-compaction markers from batch-compaction
# markers (both carry COMPRESSED_SUMMARY_METADATA_KEY so resume/handoff
# treat them alike). Supersede/defrag/rehydration must only ever touch
# micro markers: a batch marker's content is NOT contained in the micro
# rolling summary, so dropping or rewriting one destroys history.
MICRO_COMPACT_MARKER_KEY = "_micro_compact_marker"
_DB_PERSISTED_MARKER = "_db_persisted"
_NO_USER_TASK_SENTINEL = "None. This session contains no user-authored turns."
@@ -170,6 +176,35 @@ def _fresh_compaction_message_copy(msg: Dict[str, Any]) -> Dict[str, Any]:
return fresh
def _template_visible_role(message: Any) -> Optional[str]:
"""Role as counted by strict chat-template alternation checks.
Mistral-family templates (Devstral, Mistral Small 3.x, Magistral)
enforce user/assistant alternation at render time but EXEMPT the tool
flow from the check: ``tool`` results and assistant messages carrying
``tool_calls`` are skipped. A summary role chosen against the *literal*
neighbouring roles can therefore still violate alternation as the
template sees it. The canonical failure: the protected head ends
``[user, assistant(tool_calls), tool]``, so the literal last role is
``tool`` and the summary is pinned to ``role="user"`` -- but the last
role the template counts is ``user``, the template sees user -> user,
and llama.cpp / Mistral-hosted backends reject the ENTIRE request with
a Jinja alternation error (HTTP 500). Because the summary persists in
the stored conversation, every retry replays the same poisoned history
and the session is unrecoverable.
Returns ``None`` for messages the alternation check skips.
"""
if not isinstance(message, dict):
return None
role = message.get("role")
if role == "tool":
return None
if role == "assistant" and message.get("tool_calls"):
return None
return role
def _strip_persistence_markers(messages: List[Dict[str, Any]]) -> None:
"""Enforce the compaction invariant: no assembled message carries a
session-store persistence marker.
@@ -335,6 +370,11 @@ _SUMMARY_RATIO = 0.20
# itself a context-pressure source and slows every compaction.
_SUMMARY_TOKENS_CEILING = 10_000
# Micro-compaction failure guard: after this many consecutive failures on the
# same cursor position, skip the stuck exchange and advance the cursor so the
# system doesn't busy-loop on an unsummarizable exchange every turn.
_MICRO_COMPACT_MAX_CONSECUTIVE_FAILURES = 3
# Aggregate cap on the serialized turn block fed to the summarizer prompt
# (chars). Per-message truncation (_CONTENT_MAX / _TOOL_ARGS_MAX) alone is
# not enough: a compression window with hundreds of already-truncated turns
@@ -1261,6 +1301,15 @@ class ContextCompressor(ContextEngine):
self._active_compression_telemetry = None
self._compression_telemetry_seed = None
# Micro-compaction state reset
self._micro_compact_cursor = 0
self._micro_compact_rolling_summary = ""
self._micro_compact_consecutive_failures = 0
self._micro_compact_last_failure_cursor = -1
self._micro_compact_passes = 0
self._micro_compact_tokens_saved_total = 0
self._micro_compact_turns_since_pass = 0
def _begin_compression_telemetry(
self,
*,
@@ -2107,6 +2156,25 @@ class ContextCompressor(ContextEngine):
# deterministic "summary unavailable" handoff and drop the middle window.
self.abort_on_summary_failure = abort_on_summary_failure
# ── Micro-compaction (per-turn rolling compaction) ─────────
# Default: OFF. Each pass rewrites already-sent history, so it breaks
# the prompt-cache prefix every turn instead of at an episodic
# boundary. Operators opt in via `compression.micro_compact: true`.
self._micro_compact_enabled: bool = False
self._micro_compact_cursor: int = 0
self._micro_compact_rolling_summary: str = ""
self._micro_compact_consecutive_failures: int = 0
self._micro_compact_last_failure_cursor: int = -1
self._micro_compact_defrag_threshold_tokens: int = 2000
self._micro_compact_passes: int = 0
self._micro_compact_tokens_saved_total: int = 0
# Cadence: run a pass every Nth completed turn. Each pass rewrites
# already-sent history and so breaks the prompt-cache prefix, which
# makes this the dial that sets how often that break is paid. 1 =
# every turn (most aggressive reclaim, one break per turn).
self._micro_compact_every_n_turns: int = 1
self._micro_compact_turns_since_pass: int = 0
# Defer context-length resolution to first access (#32221):
# get_model_context_length() can issue a synchronous /models HTTP
# probe, which must not block AIAgent construction. The small-context
@@ -4958,6 +5026,750 @@ This compaction should PRIORITISE preserving all information related to the focu
# Main compression entry point
# ------------------------------------------------------------------
def _resolve_compact_cursor(
self,
messages: List[Dict[str, Any]],
head_end: int,
tail_start: int,
) -> int:
"""Derive the micro-compaction cursor from in-memory state or transcript scan.
Returns the index of the first message that has NOT yet been absorbed
into the rolling summary. If the in-memory cursor ``_micro_compact_cursor``
is valid (non-zero and within the compressible window), use it directly.
Otherwise scan from *head_end* through *tail_start* for the last context
summary marker and set the cursor past it.
"""
if self._micro_compact_cursor > head_end and self._micro_compact_cursor < tail_start:
return self._micro_compact_cursor
# Scan transcript for the last summary marker
last_summary_idx = -1
for idx in range(head_end, tail_start):
if self._is_context_summary_message(messages[idx]):
last_summary_idx = idx
if last_summary_idx >= head_end:
cursor = last_summary_idx + 1
# Resumed session: in-memory state is gone but the marker survives.
# Carry its text forward so the next pass merges into the existing
# history instead of replacing it with a single-exchange summary.
if not self._micro_compact_rolling_summary.strip():
recovered = self._rolling_summary_from_marker(
messages[last_summary_idx].get("content")
)
if recovered:
self._micro_compact_rolling_summary = recovered
# Rehydration is containment proof: this marker's text now
# lives inside the rolling summary, so it becomes
# supersede/defrag-eligible. This also covers a BATCH
# marker adopted as the rolling base after a batch
# compaction reset — safe precisely because we just
# absorbed its content. Markers whose content we did NOT
# absorb never get the key and are never dropped.
messages[last_summary_idx][MICRO_COMPACT_MARKER_KEY] = True
logger.info(
"Micro-compaction: recovered rolling summary from "
"transcript (%d chars)", len(recovered),
)
else:
cursor = head_end
self._micro_compact_cursor = cursor
return cursor
def _find_one_exchange(
self,
messages: List[Dict[str, Any]],
start: int,
tail_start: int,
) -> Optional[tuple[int, int]]:
"""Find the next complete exchange starting at *start*.
An exchange is one full agent turn: the first assistant message after
*start* plus everything through the end of that turn — tool results
and any follow-up assistant iterations — up to (exclusive) the next
``user`` message. Returns ``(exchange_start, exchange_end)`` indices
into *messages*, or ``None`` if no complete, safely-spliceable turn is
available before *tail_start*.
The full-turn shape is an alternation-safety requirement, not a
convenience: the splice replaces the span with a single
``assistant``-role summary marker, so the span must be bounded by
user messages on the right (``messages[exchange_end]`` is ``user``).
Absorbing only the first assistant+tools group of a multi-iteration
turn would leave the marker adjacent to the turn's next assistant
message — two consecutive assistant turns, which strict providers
reject and ``repair_message_sequence`` would then mangle.
User messages are deliberately NOT part of an exchange. The walk skips
past them to reach the assistant message, and ``exchange_start`` is that
assistant index, so user turns are never absorbed into the rolling
summary and their text stays verbatim for the life of the session.
This is the intended behaviour, not an oversight: what the assistant
emits is largely an account of what it did, which survives summarising,
while the user's own words are the instructions everything else is
derived from and are the one thing that cannot be reconstructed from
context. They are also cheap — a prompt is normally a tiny fraction
of the tokens a single tool result costs.
"""
idx = start
n = len(messages)
if idx >= n or idx >= tail_start:
return None
# Walk past user messages and existing summary markers until we hit a
# real assistant message with actual output (content or tool_calls).
# Summary markers are assistant-role themselves, so they must be
# skipped explicitly or a rehydrated cursor could try to absorb the
# marker that carries the compacted history.
while idx < tail_start and idx < n:
msg = messages[idx]
if msg.get("role") == "assistant" and not self._is_context_summary_message(msg):
break
idx += 1
if idx >= tail_start or idx >= n:
return None
exchange_start = idx
# Consume the full turn: assistant / tool messages until the next
# user message (or an existing summary marker) ends the turn.
idx += 1
while idx < tail_start and idx < n:
msg = messages[idx]
role = msg.get("role")
if role not in ("assistant", "tool"):
break
if self._is_context_summary_message(msg):
break
idx += 1
if idx <= exchange_start:
return None
# Splice-boundary guard: the message right after the exchange must
# close the turn. If the walk stopped because it ran into
# *tail_start* mid-turn (boundary is assistant or tool — including
# an assistant-role summary marker), splicing here would leave the
# assistant-role marker adjacent to the turn's remaining
# assistant/tool messages — invalid alternation. Skip this pass; the
# tail recedes as the conversation grows and the turn becomes
# absorbable later. Any other boundary role (user, or a stray
# system/injected message) is a safe splice point — the marker is
# assistant-role, so no same-role adjacency is possible — and
# accepting them keeps one odd message from wedging the cursor
# forever.
if idx >= n:
return None
boundary = messages[idx]
if not isinstance(boundary, dict) or boundary.get("role") in ("assistant", "tool"):
return None
return (exchange_start, idx)
def _serialize_one_exchange(
self,
messages: List[Dict[str, Any]],
start: int,
end: int,
) -> str:
"""Serialize a single exchange for the micro-summarizer.
Delegates to the batch path's ``_serialize_for_summary`` (same
truncation, redaction, think-block stripping, and media labeling),
scoped to one exchange — one serializer, one place to fix.
"""
return self._serialize_for_summary(messages[start:end])
def _build_micro_summary_prompt(
self,
existing_summary: str,
exchange_text: str,
) -> List[Dict[str, str]]:
"""Build the prompt messages for a single-exchange micro-summary."""
if existing_summary.strip():
summary_block = existing_summary
else:
summary_block = "(No previous summary yet.)"
user_prompt = (
"You are a summarization agent creating a compact record of an "
"ongoing conversation. You are given a running summary and the "
"next exchange from the conversation. Merge the exchange's key "
"decisions, requirements, file paths, and open questions into the "
"summary. Preserve the summary's structure. Drop resolved details "
"that are no longer relevant. Add new decisions, file paths, and "
"open questions.\n\n"
"NEVER include API keys, tokens, passwords, secrets, credentials, "
"or connection strings in the summary \u2014 replace any that appear "
f"with [REDACTED].\n\n"
f"## Current Running Summary\n{summary_block}\n\n"
f"## Next Exchange to Merge\n{exchange_text}\n\n"
"Return ONLY the updated summary text, no preamble or explanation. "
"Do not include this instruction block in your output."
)
return [
{"role": "system", "content": "You are a conversation summarization assistant."},
{"role": "user", "content": user_prompt},
]
def _micro_summarize_one(
self,
exchange_text: str,
) -> Optional[str]:
"""Micro-summarize one exchange into the rolling summary via aux LLM.
Calls the same auxiliary compression model as the batch path, with
a focused prompt that merges one exchange into the running summary.
Returns the updated summary text, or ``None`` on failure.
"""
from agent.auxiliary_client import call_llm, aux_interrupt_protection
messages = self._build_micro_summary_prompt(
self._micro_compact_rolling_summary,
exchange_text,
)
call_kwargs = {
"task": "compression",
"messages": messages,
"max_tokens": min(1500, self.max_summary_tokens or 1500),
"temperature": 0.1,
}
if self.summary_model:
call_kwargs["model"] = self.summary_model
if self.model:
call_kwargs.setdefault("main_runtime", {
"model": self.model,
"provider": self.provider or "",
"base_url": self.base_url or "",
"api_key": self.api_key or "",
"api_mode": getattr(self, "api_mode", "") or "",
})
try:
with aux_interrupt_protection():
response = call_llm(**call_kwargs)
except Exception as exc:
logger.info("micro-summarization call failed: %s", exc)
return None
message = response.choices[0].message
if isinstance(message, dict):
content = message.get("content")
else:
content = getattr(message, "content", message)
if not isinstance(content, str):
content = str(content) if content else ""
content = content.strip()
if not content:
logger.info("micro-summarization returned empty content")
return None
from agent.agent_runtime_helpers import strip_think_blocks
stripped = strip_think_blocks(None, content).strip()
return stripped if stripped else None
def _needs_defrag(self) -> bool:
"""Return True when the rolling summary is large enough to defrag."""
content_tokens = estimate_tokens_rough(self._micro_compact_rolling_summary)
return content_tokens >= self._micro_compact_defrag_threshold_tokens
def _defrag_rolling_summary(
self,
messages: List[Dict[str, Any]],
) -> bool:
"""Re-summarize the rolling summary TEXT and rewrite the marker in place.
Merging exchange after exchange makes the rolling summary baggy —
repetitive, and larger than the material justifies. Defrag compacts
the summary *itself*: one aux call over the accumulated summary text,
then the existing marker's content is rewritten in place.
Deliberately transcript-shape-neutral: no messages are spliced, no
user turns are touched, and the cursor does not move. The original
implementation serialized the whole remaining middle (user turns
included) and spliced it into the marker, which silently absorbed
user messages — violating the feature's core "your messages are never
compacted" invariant. Un-absorbed exchanges stay where they are and
get absorbed by later per-exchange passes.
Returns True when a pass actually rewrote the summary.
"""
old_summary = self._micro_compact_rolling_summary
if not old_summary.strip():
return False
# Feed the old summary through the merge prompt with an empty base:
# "merge these decisions into (no previous summary)" is exactly a
# rewrite-compactly instruction for the accumulated text.
self._micro_compact_rolling_summary = ""
fresh_summary = self._micro_summarize_one(old_summary)
if not fresh_summary:
self._micro_compact_rolling_summary = old_summary
return False
self._micro_compact_rolling_summary = fresh_summary
# Rewrite the newest MICRO marker's content in place so the transcript
# and the in-memory summary stay in step (resume rehydrates from it).
# Scoped to micro-tagged markers: rewriting a batch-compaction marker
# would overwrite history the rolling summary does not contain.
for idx in range(len(messages) - 1, -1, -1):
entry = messages[idx]
if (
isinstance(entry, dict)
and entry.get(COMPRESSED_SUMMARY_METADATA_KEY)
and entry.get(MICRO_COMPACT_MARKER_KEY)
):
entry["content"] = self._render_micro_marker_content(fresh_summary)
# Content changed after a possible flush — clear the persisted
# stamp so the DB sync/flush rewrites the row.
entry.pop(_DB_PERSISTED_MARKER, None)
break
logger.info(
"Micro-compaction defrag: rolling summary re-summarized "
"(%d -> %d chars)", len(old_summary), len(fresh_summary),
)
return True
def _micro_compact(
self,
messages: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
"""Run one round of micro-compaction on the conversation.
Absorbs the oldest uncompacted exchange into the rolling summary,
advancing the in-memory cursor. Runs in post-turn idle time.
This is the public entry point called from ``finalize_turn()``.
Returns the (possibly modified) message list.
NOTE: the in-memory splice alone is not persisted — the subsequent
``_persist_session`` flush is append-only, so old DB rows stay
``active=1`` and a session resume double-loads both the summary and
the original exchanges. This method therefore also calls
``archive_and_compact`` on the session DB to soft-archive old rows
and insert the compacted set atomically.
"""
if not self._micro_compact_enabled:
return messages
# Cadence gate. A pass rewrites already-sent history, so it costs one
# prompt-cache break; `every_n_turns` is how an operator trades reclaim
# frequency against that cost. Counted per invocation rather than per
# committed pass so a turn that finds nothing to absorb still advances
# the cadence and cannot wedge it.
every_n = max(1, int(self._micro_compact_every_n_turns or 1))
if every_n > 1:
self._micro_compact_turns_since_pass += 1
if self._micro_compact_turns_since_pass < every_n:
return messages
self._micro_compact_turns_since_pass = 0
n_messages = len(messages)
if n_messages < 4:
return messages
head_size = self._protect_head_size(messages)
compress_start = self._align_boundary_forward(messages, head_size)
compress_end = self._find_tail_cut_by_tokens(messages, compress_start)
if compress_start >= compress_end:
return messages
cursor = self._resolve_compact_cursor(messages, compress_start, compress_end)
if cursor >= compress_end:
return messages
# Find the next exchange
exchange = self._find_one_exchange(messages, cursor, compress_end)
if exchange is None:
return messages
exchange_start, exchange_end = exchange
# Baseline for telemetry. Taken only once an exchange is in hand, so
# turns that no-op early don't pay for the scan.
_started_at = time.monotonic()
_tokens_before = estimate_messages_tokens_rough(messages)
_messages_before = n_messages
def _elapsed_ms() -> int:
return int((time.monotonic() - _started_at) * 1000)
# Check for defrag trigger: the rolling summary itself has grown
# baggy. Defrag rewrites the summary text and the existing marker in
# place — no splice, no cursor movement, no user turns touched — so
# the transcript shape is unchanged and this pass does not also
# absorb an exchange (one aux call per turn either way).
if self._needs_defrag():
defragged = self._defrag_rolling_summary(messages)
if defragged:
self._sync_micro_compact_to_db(messages)
self._micro_compact_consecutive_failures = 0
self._micro_compact_last_failure_cursor = -1
self._emit_micro_compaction_telemetry(
outcome="defrag" if defragged else "defrag_failed",
messages_before=_messages_before,
messages_after=len(messages),
tokens_before=_tokens_before,
tokens_after=estimate_messages_tokens_rough(messages),
duration_ms=_elapsed_ms(),
)
return messages
# Whether this pass's summary will be cumulative — i.e. whether it
# subsumes any earlier marker. Captured before summarizing.
_cumulative = bool(self._micro_compact_rolling_summary.strip())
# Micro-summarize one exchange
exchange_text = self._serialize_one_exchange(messages, exchange_start, exchange_end)
_exchange_tokens = estimate_tokens_rough(exchange_text)
updated_summary = self._micro_summarize_one(exchange_text)
if updated_summary is None:
# Track consecutive failures on the same cursor position so we
# don't busy-loop on an unsummarizable exchange every turn.
if exchange_start == self._micro_compact_last_failure_cursor:
self._micro_compact_consecutive_failures += 1
else:
self._micro_compact_consecutive_failures = 1
self._micro_compact_last_failure_cursor = exchange_start
if self._micro_compact_consecutive_failures >= _MICRO_COMPACT_MAX_CONSECUTIVE_FAILURES:
logger.info(
"Micro-compaction: skipping exchange at cursor %d "
"after %d consecutive failures",
exchange_start, self._micro_compact_consecutive_failures,
)
# Advance the cursor past the stuck exchange so we don't
# retry it every turn. The skipped messages remain in the
# transcript and will be absorbed by the next batch
# compression or defrag.
self._micro_compact_cursor = exchange_end
self._micro_compact_consecutive_failures = 0
self._micro_compact_last_failure_cursor = -1
_outcome = "exchange_skipped"
else:
_outcome = "summarize_failed"
self._emit_micro_compaction_telemetry(
outcome=_outcome,
messages_before=_messages_before,
messages_after=len(messages),
tokens_before=_tokens_before,
tokens_after=_tokens_before,
exchange_tokens=_exchange_tokens,
duration_ms=_elapsed_ms(),
)
return messages
self._micro_compact_rolling_summary = updated_summary
self._micro_compact_cursor = exchange_end
self._micro_compact_consecutive_failures = 0
self._micro_compact_last_failure_cursor = -1
result = self._splice_micro_compact_result(
messages, exchange_start, exchange_end, supersede=_cumulative,
)
self._micro_compact_cursor = self._cursor_after_splice(result, exchange_start + 1)
self._sync_micro_compact_to_db(result)
self._emit_micro_compaction_telemetry(
outcome="absorbed",
messages_before=_messages_before,
messages_after=len(result),
tokens_before=_tokens_before,
tokens_after=estimate_messages_tokens_rough(result),
exchange_tokens=_exchange_tokens,
duration_ms=_elapsed_ms(),
)
return result
@staticmethod
def _rolling_summary_from_marker(content: Any) -> str:
"""Recover the rolling-summary text from a summary marker's content.
The rolling summary lives in memory, but a resumed session starts with
an empty one while the marker holding every previous exchange is still
in the transcript. Without rehydrating from it, the first post-resume
pass would build a marker from nothing and supersede the one carrying
the whole history.
"""
if not isinstance(content, str) or not content.strip():
return ""
body = content
# rfind, not find: SUMMARY_PREFIX itself references the heading text,
# so the first occurrence is inside the preamble, not the real heading.
idx = body.rfind(HISTORICAL_TASK_HEADING)
if idx != -1:
body = body[idx + len(HISTORICAL_TASK_HEADING):]
end = body.find(_SUMMARY_END_MARKER)
if end != -1:
body = body[:end]
return body.strip()
def _cursor_after_splice(
self,
result: List[Dict[str, Any]],
fallback: int,
) -> int:
"""Cursor position just past the summary marker in *result*.
The cursor must be derived from the spliced list, never carried over
from pre-splice indices. A splice collapses the absorbed span (an
assistant plus its tool results -- often several messages) into a
single marker, and may also drop a superseded marker further back, so
every index after it shifts. Reusing the old ``exchange_end`` left the
cursor pointing into the middle of a *later* exchange's tool group;
the next pass then walked forward to the following assistant and
skipped that exchange entirely, so roughly half the work silently
never happened on tool-bearing conversations.
"""
for idx in range(len(result) - 1, -1, -1):
entry = result[idx]
if isinstance(entry, dict) and entry.get(COMPRESSED_SUMMARY_METADATA_KEY):
return idx + 1
return fallback
def _emit_micro_compaction_telemetry(
self,
*,
outcome: str,
messages_before: int,
messages_after: int,
tokens_before: int | None,
tokens_after: int | None,
exchange_tokens: int | None = None,
duration_ms: int | None = None,
) -> None:
"""Emit one content-free JSON log line describing a micro-compaction pass.
Mirrors ``_emit_compression_attempt_telemetry`` for the batch path.
Message counts move by one or two even when the saving is large, so the
token fields are the ones that actually answer "is this helping?".
``tokens_delta`` is negative when the pass shrank the transcript, and
the ``*_total`` fields accumulate across the session so a whole run can
be summarised from the last line alone.
"""
try:
delta = None
if tokens_before is not None and tokens_after is not None:
delta = tokens_after - tokens_before
self._micro_compact_tokens_saved_total -= delta
self._micro_compact_passes += 1
# Cached reads only. The ``threshold_tokens`` / ``context_length``
# properties resolve lazily and can fire a synchronous /models
# probe on first access (#32221) — telemetry must never be the
# thing that blocks a turn. Unresolved simply reports null.
threshold = self._threshold_tokens
context_limit = self._resolved_context_length
occupancy = None
if threshold and tokens_after is not None and threshold > 0:
occupancy = round(tokens_after / threshold * 100, 1)
payload = {
"event": "micro_compaction",
"session_id": getattr(self, "_session_id", "") or "",
"outcome": outcome,
"messages_before": messages_before,
"messages_after": messages_after,
"tokens_before": _safe_int(tokens_before),
"tokens_after": _safe_int(tokens_after),
"tokens_delta": _safe_int(delta),
"exchange_tokens": _safe_int(exchange_tokens),
"rolling_summary_tokens": estimate_tokens_rough(
self._micro_compact_rolling_summary
),
"cursor": _safe_int(self._micro_compact_cursor),
"passes_total": self._micro_compact_passes,
"tokens_saved_total": self._micro_compact_tokens_saved_total,
"duration_ms": _safe_int(duration_ms),
# Headroom, not efficiency: how full the window is being kept.
# This is the number that says whether the session can keep
# going without a hard batch compaction.
"threshold_tokens": _safe_int(threshold),
"context_limit": _safe_int(context_limit),
"occupancy_pct": occupancy,
"main_model": self.model or "",
"aux_model": self.summary_model or "",
}
logger.info(
"micro compaction telemetry: %s",
json.dumps(payload, sort_keys=True, separators=(",", ":")),
)
except Exception as exc:
logger.debug("failed to emit micro-compaction telemetry: %s", exc)
def _sync_micro_compact_to_db(
self,
compacted_messages: List[Dict[str, Any]],
) -> None:
"""Persist the micro-compacted message set to the session DB.
Soft-archives every currently-active message row (``active = 0``)
and inserts *compacted_messages* as fresh active rows — atomically,
via ``archive_and_compact``. Then stamps ``_DB_PERSISTED_MARKER`` on
every dict so the upcoming append-only flush (``_persist_session`` →
``_flush_messages_to_session_db_unlocked``) skips them: they are
already correctly stored.
Without this, the in-memory-only splice leaves old exchange rows at
``active=1``, and a session resume double-loads both the summary and
the original messages — blowing past the model's context limit.
"""
session_db = getattr(self, "_session_db", None)
session_id = getattr(self, "_session_id", "")
if not session_db or not session_id:
return
try:
session_db.archive_and_compact(session_id, compacted_messages)
for msg in compacted_messages:
if isinstance(msg, dict):
msg[_DB_PERSISTED_MARKER] = True
except Exception:
logger.info(
"Micro-compaction DB sync failed — resume will double-load "
"compacted messages until the next batch compression"
)
def _splice_micro_compact_result(
self,
messages: List[Dict[str, Any]],
splice_start: int,
splice_end: int,
supersede: bool = True,
) -> List[Dict[str, Any]]:
"""Replace *messages[splice_start:splice_end]* with a summary marker.
The summary marker carries the rolling summary text and the
``_compressed_summary`` metadata flag so downstream consumers
(resume, handoff, /compress) handle it identically to batch
compaction summaries.
Alternation safety: the marker is ``assistant``-role. An exchange is
a full agent turn bounded by user messages on both sides (see
``_find_one_exchange``), so the spliced result is
``user → marker(assistant) → user`` — valid alternation that the
pre-request ``repair_message_sequence`` pass leaves untouched. A
``user``-role marker in that position produced ``user → user → user``,
and repair then merged the marker into the neighbouring real user
message: metadata gone, cursor unrecoverable, and the summary text
duplicated into the transcript on every subsequent pass.
Superseding an earlier marker removes the assistant turn that stood
between two real user messages, leaving them adjacent. Those two are
merged (plain-text only, ``\\n\\n``-joined — the same repair pass 2
would apply) so the transcript is alternation-valid as returned
rather than relying on downstream repair to fix it up.
"""
summary_text = self._micro_compact_rolling_summary
if not summary_text.strip():
return messages
summary_msg = {
"role": "assistant",
"content": self._render_micro_marker_content(summary_text),
COMPRESSED_SUMMARY_METADATA_KEY: True,
# Micro-created marker: eligible for supersede/defrag rewrites.
# Batch markers never carry this key and are never touched —
# their content is not contained in the rolling summary.
MICRO_COMPACT_MARKER_KEY: True,
# Honest provenance (#64650): this marker absorbs only
# assistant/tool content — user turns are never micro-compacted,
# so they remain in the transcript and _transcript_has_real_user_turn
# keeps reporting them directly.
COMPRESSED_SUMMARY_HAS_USER_TURN_KEY: False,
}
result = messages[:splice_start] + [summary_msg] + messages[splice_end:]
# The rolling summary is cumulative: this marker already contains
# everything every earlier micro-compaction marker held. Leaving those
# in place stacks near-duplicate copies of the same text — each with
# its own prefix/heading/end-marker scaffolding — so the transcript
# grows with every turn instead of shrinking, which defeats the point.
# Keep only the newest marker.
# Two containment gates before dropping an earlier marker:
# 1. supersede (the rolling summary was non-empty going into this
# pass) — a pass that started from nothing (a resume that could
# not rehydrate) covers one exchange, and dropping the previous
# marker would throw away the entire compacted history.
# 2. MICRO_COMPACT_MARKER_KEY on the candidate — only markers whose
# text is provably inside the rolling summary (created by our own
# splice, or rehydrated into the summary by
# _resolve_compact_cursor) carry it. A batch-compaction marker
# that landed after our last pass holds MORE history than the
# stale rolling summary; dropping it would destroy that history.
if supersede:
marker_idxs = [
i for i, m in enumerate(result)
if isinstance(m, dict)
and m.get(COMPRESSED_SUMMARY_METADATA_KEY)
and m.get(MICRO_COMPACT_MARKER_KEY)
]
if len(marker_idxs) > 1:
superseded = set(marker_idxs[:-1])
result = [m for i, m in enumerate(result) if i not in superseded]
result = self._merge_adjacent_user_turns(result)
# NOTE: deliberately NO _strip_persistence_markers here. The batch
# path strips because compress() copies head/tail into a rotated
# child session (#57491); micro-compaction archives in place under
# the SAME session id, and the surviving dicts' _db_persisted stamps
# are accurate. Stripping them meant an archive_and_compact failure
# left every previously-persisted message unstamped, and the next
# append-only flush re-inserted them as duplicate active rows on top
# of the still-active originals. _sync_micro_compact_to_db re-stamps
# everything after a SUCCESSFUL archive; on failure the old stamps
# keep the flush idempotent (only the new marker row is appended).
return result
@staticmethod
def _render_micro_marker_content(summary_text: str) -> str:
"""Assemble the marker content wrapper around *summary_text*."""
return (
f"{SUMMARY_PREFIX}\n\n"
f"{HISTORICAL_TASK_HEADING}\n"
f"{summary_text.strip()}"
f"\n\n{_SUMMARY_END_MARKER}"
)
@staticmethod
def _merge_adjacent_user_turns(
result: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
"""Merge consecutive plain-text real user turns left by a supersede.
Dropping a superseded marker removes the assistant turn that separated
two real user messages. Merging them here (``\\n\\n``-joined, exactly
what ``repair_message_sequence`` pass 2 does) keeps every byte the
user typed while restoring alternation deliberately, so the marker
and cursor state are never collateral damage of the downstream repair.
Multimodal (list) content is left alone, mirroring the repair pass.
"""
from agent.turn_context import drop_stale_api_content
merged: List[Dict[str, Any]] = []
for msg in result:
prev = merged[-1] if merged else None
if (
isinstance(msg, dict)
and isinstance(prev, dict)
and msg.get("role") == "user"
and prev.get("role") == "user"
and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY)
and not prev.get(COMPRESSED_SUMMARY_METADATA_KEY)
and isinstance(prev.get("content"), str)
and isinstance(msg.get("content"), str)
):
prev_content = prev["content"]
new_content = msg["content"]
prev["content"] = (
(prev_content + "\n\n" + new_content)
if prev_content and new_content
else (prev_content or new_content)
)
# Merged content invalidates the api_content sidecar (exact
# bytes previously sent for the pre-merge message).
drop_stale_api_content(prev)
continue
merged.append(msg)
return merged
def compress(
self,
messages: List[Dict[str, Any]],
@@ -5420,8 +6232,43 @@ This compaction should PRIORITISE preserving all information related to the focu
# last_head_role reads the assembled (post-strip) head; first_tail_role
# reads the assembled (post-strip) tail_messages — a stripped stale
# handoff must not influence alternation-safe role selection.
last_head_role = compressed[-1].get("role", "user") if compressed else "user"
first_tail_role = tail_messages[0].get("role", "user") if tail_messages else None
# Both are TEMPLATE-VISIBLE roles (``_template_visible_role``), not the
# literal list neighbours: strict Mistral-style templates skip tool
# results and assistant tool-call messages when enforcing
# user/assistant alternation, so the summary must alternate against
# the nearest message the template actually counts. Selecting against
# the literal neighbour (previously ``compressed[-1]``) emitted the
# summary as role="user" behind a ``[user, assistant(tool_calls),
# tool]`` head — which every Mistral-strict backend rejects with a
# Jinja alternation 500, permanently poisoning the session.
last_head_role: Optional[str] = "user"
if compressed:
last_head_role = next(
(
role
for role in (
_template_visible_role(m) for m in reversed(compressed)
)
if role is not None
),
# Head holds only template-exempt messages: the summary will
# be the first message the template counts, and the sequence
# must open with "user" (handled below alongside the forced
# cases).
None,
)
first_tail_role = None
if tail_messages:
first_tail_role = next(
(
role
for role in (
_template_visible_role(m) for m in tail_messages
)
if role is not None
),
None,
)
# When the only protected head message is the system prompt, the
# summary becomes the first *visible* message in the API request
# (most adapters — Anthropic, Bedrock — send the system prompt as
@@ -5455,9 +6302,15 @@ This compaction should PRIORITISE preserving all information related to the focu
)
if not _user_survives:
_force_user_leading = True
# Pick a role that avoids consecutive same-role with both neighbors.
# Priority: avoid colliding with head (already committed), then tail.
if last_head_role in {"assistant", "tool"} or _force_user_leading:
# Pick a role that alternates with both template-visible neighbors.
# Priority: alternate against the head (already committed), then tail.
# ``None`` (all-exempt head) means the summary opens the visible
# sequence, which strict templates require to start with "user".
if (
last_head_role is None
or last_head_role in {"assistant", "tool"}
or _force_user_leading
):
summary_role = "user"
else:
summary_role = "assistant"
@@ -5465,7 +6318,14 @@ This compaction should PRIORITISE preserving all information related to the focu
# collide with the head, flip it.
if first_tail_role is not None and summary_role == first_tail_role:
flipped = "assistant" if summary_role == "user" else "user"
if flipped != last_head_role and not _force_user_leading:
# ``last_head_role is None`` (all-exempt head) pins the summary to
# "user" above; flipping to "assistant" would make the visible
# sequence open with "assistant", which strict templates reject.
if (
flipped != last_head_role
and last_head_role is not None
and not _force_user_leading
):
summary_role = flipped
else:
# Both roles would create consecutive same-role messages
@@ -5593,6 +6453,19 @@ This compaction should PRIORITISE preserving all information related to the focu
_strip_persistence_markers(compressed)
self._last_compression_made_progress = True
# Batch compaction invalidates micro-compaction state: the batch
# marker now holds MORE history than the in-memory rolling summary
# (it summarized everything in the window, including exchanges micro
# never absorbed). Keeping the stale summary would let the next micro
# pass supersede-drop or defrag-rewrite content it does not contain.
# Reset instead; the next micro pass rehydrates from the batch marker
# via _resolve_compact_cursor, which re-tags it as micro-eligible
# only after absorbing its content into the rolling summary.
self._micro_compact_rolling_summary = ""
self._micro_compact_cursor = 0
self._micro_compact_consecutive_failures = 0
self._micro_compact_last_failure_cursor = -1
return compressed
+19 -3
View File
@@ -22,9 +22,7 @@ import os
import random
import re
import ssl
import threading
import time
import uuid
from typing import Any, Dict, List, Optional
from agent.codex_responses_adapter import _summarize_user_message_for_log
@@ -40,7 +38,6 @@ from agent.conversation_compression import (
from agent.context_engine import automatic_compaction_status_message
from agent.display import KawaiiSpinner
from agent.error_classifier import FailoverReason, classify_api_error
from agent.iteration_budget import IterationBudget
from agent.turn_context import (
_compression_warrants_another_preflight_pass,
build_turn_context,
@@ -1386,10 +1383,23 @@ def run_conversation(
# However, providers like Moonshot AI require a separate 'reasoning_content' field
# on assistant messages with tool_calls. We handle both cases here.
request_logger = getattr(agent, "logger", None) or logging.getLogger(__name__)
# Per-agent validation cursor: skips re-json.loads-ing tool_call
# arguments on history messages already validated in a previous
# iteration. Identity-keyed (strong refs) — compression/undo/repair
# rewriting the list breaks the prefix match and forces a re-scan
# from the divergence point. See sanitize_tool_call_arguments.
_sanitize_cursor = getattr(agent, "_sanitize_args_cursor", None)
if _sanitize_cursor is None:
_sanitize_cursor = {}
try:
agent._sanitize_args_cursor = _sanitize_cursor
except Exception:
pass
repaired_tool_calls = agent._sanitize_tool_call_arguments(
messages,
logger=request_logger,
session_id=agent.session_id,
cursor=_sanitize_cursor,
)
if repaired_tool_calls > 0:
request_logger.info(
@@ -1435,6 +1445,12 @@ def run_conversation(
api_msg.pop("display_kind", None)
api_msg.pop("display_metadata", None)
# Durable row identity stamped by _rows_to_conversation so the
# desktop can address a specific persisted message (reactions).
# Bookkeeping, never a provider field — only the chat-completions
# transport strips underscore keys, so drop it centrally here.
api_msg.pop("_row_id", None)
# Inject ephemeral context into the current turn's user message.
# Sources: memory manager prefetch + plugin pre_llm_call hooks
# with target="user_message" (the default). Both are
+4 -5
View File
@@ -138,8 +138,8 @@ def is_paused() -> bool:
def _load_config() -> Dict[str, Any]:
"""Read curator.* config from ~/.hermes/config.yaml. Tolerates missing file."""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator: %s", e)
return {}
@@ -902,7 +902,6 @@ def _reconcile_classification(
Every removed skill is placed in exactly one bucket.
"""
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
heur_pruned = {e["name"] for e in heuristic.get("pruned", [])}
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
@@ -1876,9 +1875,9 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
_acp_args = None
_model_name = ""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
from hermes_cli.runtime_provider import resolve_runtime_provider
_cfg = load_config()
_cfg = load_config_readonly()
_binding = _resolve_review_runtime(_cfg)
_provider, _model_name = _binding.provider, _binding.model
_rp = resolve_runtime_provider(
+2 -2
View File
@@ -147,8 +147,8 @@ def _utc_id(now: Optional[datetime] = None) -> str:
def _load_config() -> Dict[str, Any]:
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator backup: %s", e)
return {}
+2 -2
View File
@@ -197,8 +197,8 @@ def _config_language_cached() -> str | None:
(e.g. after the setup wizard).
"""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
lang = (cfg.get("display") or {}).get("language")
if lang:
return _normalize_lang(lang)
+2 -2
View File
@@ -91,9 +91,9 @@ def get_active_provider() -> Optional[ImageGenProvider]:
"""
configured: Optional[str] = None
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
section = cfg.get("image_gen") if isinstance(cfg, dict) else None
if isinstance(section, dict):
raw = section.get("provider")
-1
View File
@@ -403,7 +403,6 @@ def _category_counts(payload: dict[str, Any]) -> list[tuple[str, int]]:
def category_color_map(payload: dict[str, Any]) -> dict[str, str]:
"""Deterministic, evenly-spread hue per skill category (theme-independent)."""
clusters = _category_counts(payload)
n = max(1, len(clusters))
# Golden-angle hue spacing so adjacent categories never collide in color.
return {cat: rgb_to_hex(_hsl_to_rgb((i * 137.508) % 360, 0.55, 0.62)) for i, (cat, _c) in enumerate(clusters)}
+1 -1
View File
@@ -55,7 +55,7 @@ def register_subparser(subparsers: argparse._SubParsersAction) -> None:
help="Even attempt servers marked manual-install (best effort)",
)
sub_restart = sub.add_parser(
sub.add_parser(
"restart",
help="Tear down running LSP clients (next edit re-spawns)",
)
-1
View File
@@ -30,7 +30,6 @@ import logging
import os
import shutil
import subprocess
import sys
import threading
from pathlib import Path
from typing import Any, Dict, Optional
+2 -2
View File
@@ -196,8 +196,8 @@ class LSPService:
itself returns ``is_active()`` False when LSP is disabled.
"""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e: # noqa: BLE001
logger.debug("LSP config load failed: %s", e)
return None
+375
View File
@@ -14,6 +14,7 @@ re-exports from ``run_agent`` remain in place so existing imports
from __future__ import annotations
import hashlib
import json
import logging
import re
@@ -474,4 +475,378 @@ __all__ = [
"_sanitize_tools_non_ascii",
"_strip_images_from_messages",
"_sanitize_structure_non_ascii",
# call_id policy owners (F4 consolidation)
"deterministic_call_id",
"coalesce_tool_call_id",
"uniquify_tool_call_ids",
# reasoning_content policy owners (F4 consolidation)
"reasoning_echo_family",
"matches_reasoning_echo_family",
"needs_reasoning_echo",
"apply_reasoning_content_policy",
"reapply_reasoning_echo",
]
# ---------------------------------------------------------------------------
# call_id policy — single owner (audit F4, incident chain I4)
# ---------------------------------------------------------------------------
#
# Three forked policy sites converged here:
# * agent/codex_responses_adapter.py `_deterministic_call_id` — hash
# synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2).
# * run_agent.AIAgent._get_tool_call_id_static — `call_id or id`
# coalescing for dicts and SDK objects.
# * run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with
# deterministic `_d<n>` suffixes (#58327 loss class).
#
# NOT consolidated (different scheme on purpose):
# agent/transports/codex_event_projector._deterministic_call_id maps codex
# app-server ITEM ids (`codex_<type>_<item_id>`), not chat tool-call
# content; merging the two would change ids and invalidate prompt caches.
#
# HARD INVARIANT: everything here must stay deterministic (never uuid4) and
# byte-identical for existing inputs — these ids feed prompt-cache prefixes.
def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return f"call_{digest}"
def coalesce_tool_call_id(tc: Any) -> str:
"""Extract the effective call ID from a tool_call entry (dict or object).
Single owner for the ``call_id or id`` coalescing rule: Codex Responses
tool calls carry ``call_id`` (authoritative pairing key), Chat
Completions ones carry ``id`` only. Returns ``""`` when neither is set.
"""
if isinstance(tc, dict):
return (tc.get("call_id", "") or tc.get("id", "") or "").strip()
return (getattr(tc, "call_id", "") or getattr(tc, "id", "") or "").strip()
def uniquify_tool_call_ids(tool_calls: list) -> list:
"""Ensure every tool call in a single assistant turn has a distinct id.
Some models/providers reuse one call id across different calls in a
single batch (observed with native Kimi Responses replays, Ollama-
compatible endpoints, and degraded models at long context; same bug
class as openclaw/openclaw#110518 / #110956). Duplicate ids are lossy
downstream: the pre-API sanitizer keeps only the first call/result
pair per id (#58327), so the later call's result silently vanishes
from every replayed payload, and strict providers (Anthropic
tool_use, DeepSeek) reject duplicate ids outright.
The first occurrence keeps its id; later collisions get a
deterministic ``<id>_d<n>`` suffix — never a random UUID, which would
break prompt-cache prefix stability across replays. Mutates the
entries in place (SDK models / SimpleNamespace / dicts) and returns
the same list. Blank/missing ids are left for the deterministic
fallback in ``build_assistant_message``.
"""
seen: set = set()
for tc in tool_calls or []:
# Same coalescing rule as ``coalesce_tool_call_id`` but tolerant of
# non-string ids (degraded models can emit ints/None here).
if isinstance(tc, dict):
raw = tc.get("call_id") or tc.get("id") or ""
else:
raw = getattr(tc, "call_id", None) or getattr(tc, "id", None) or ""
raw = raw.strip() if isinstance(raw, str) else ""
if not raw:
continue
# Composite Responses ids ("call_x|fc_y") collide on the call
# half — that's the pairing key providers enforce per turn.
cid = raw.split("|", 1)[0]
if not cid:
continue
if cid not in seen:
seen.add(cid)
continue
n = 2
new_id = f"{cid}_d{n}"
while new_id in seen:
n += 1
new_id = f"{cid}_d{n}"
seen.add(new_id)
def _renamed(value):
# Preserve a composite id's response-item half so the
# provider's real fc_/item id survives the rename.
if isinstance(value, str) and "|" in value:
return f"{new_id}|{value.split('|', 1)[1]}"
return new_id
try:
if isinstance(tc, dict):
if tc.get("id"):
tc["id"] = _renamed(tc["id"])
else:
tc["id"] = new_id
if tc.get("call_id"):
tc["call_id"] = new_id
else:
tc.id = _renamed(getattr(tc, "id", None))
if getattr(tc, "call_id", None):
tc.call_id = new_id
except Exception:
logger.warning(
"Could not uniquify duplicate tool call id %s", cid
)
continue
_fn = tc.get("function") if isinstance(tc, dict) else getattr(tc, "function", None)
_fn_name = (_fn.get("name") if isinstance(_fn, dict) else getattr(_fn, "name", None)) or "?"
logger.warning(
"Model reused tool call id %s within one turn; renamed the "
"duplicate to %s (tool=%s) to keep call/result pairing "
"lossless.", cid, new_id, _fn_name,
)
return tool_calls
# ---------------------------------------------------------------------------
# reasoning_content policy — single owner (audit F4)
# ---------------------------------------------------------------------------
#
# The strip-vs-repad decision was previously forked across the wire files in
# separate incident commits (2b3a4f0af8 strip for strict providers,
# b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi pad). The
# POLICY — which provider direction gets which treatment — lives here as one
# rule table + apply functions; adapters keep only SYNTAX mapping (e.g.
# anthropic_adapter turning reasoning_content into a thinking block).
#
# Direction table:
# require-side (echo-back enforced; replays 400 without the field):
# kimi — provider kimi-coding/kimi-coding-cn, or host api.kimi.com /
# moonshot.ai / moonshot.cn. Host-driven on purpose:
# aggregators re-exporting kimi models reject the echo.
# deepseek — provider "deepseek", model contains "deepseek", or host
# api.deepseek.com (#15250; V4 rejects empty-string pads,
# hence the " " single-space pad, #17341).
# mimo — provider "xiaomi", model contains "mimo", or host
# *.xiaomimimo.com.
# strict side (field rejected with 400/422 "Extra inputs are not
# permitted"): everyone else — Mistral, Cerebras, Groq, SambaNova, …
# (#45655). Strip the key entirely, even a single-space pad.
_REASONING_ECHO_RULES: tuple = (
# (family, exact providers (raw), exact providers (lowered),
# model substrings (lowered), base_url hosts)
("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (),
("api.kimi.com", "moonshot.ai", "moonshot.cn")),
("deepseek", frozenset(), frozenset({"deepseek"}), ("deepseek",),
("api.deepseek.com",)),
("mimo", frozenset(), frozenset({"xiaomi"}), ("mimo",),
("api.xiaomimimo.com", "xiaomimimo.com")),
)
def _family_rule(family: str) -> tuple:
for rule in _REASONING_ECHO_RULES:
if rule[0] == family:
return rule
raise KeyError(family)
def matches_reasoning_echo_family(
family: str, provider: Any, model: Any, base_url: Any
) -> bool:
"""True when (provider, model, base_url) matches one echo-back family.
Families can overlap (e.g. a deepseek-named model pointed at a kimi
host); this membership test is independent per family so per-family
predicates keep their original semantics.
"""
from utils import base_url_host_matches
_, raw_providers, lowered_providers, model_subs, hosts = _family_rule(family)
provider_lower = (provider or "").lower()
model_lower = (model or "").lower()
if provider in raw_providers or provider_lower in lowered_providers:
return True
if any(sub in model_lower for sub in model_subs):
return True
return any(base_url_host_matches(base_url, host) for host in hosts)
def reasoning_echo_family(provider: Any, model: Any, base_url: Any) -> "str | None":
"""Classify the provider direction for the reasoning_content echo policy.
Returns ``"kimi"``, ``"deepseek"``, or ``"mimo"`` (first match in table
order) when the target endpoint enforces reasoning_content echo-back on
assistant turns, else ``None`` (strict/indifferent side — the field must
be stripped).
"""
for rule in _REASONING_ECHO_RULES:
if matches_reasoning_echo_family(rule[0], provider, model, base_url):
return rule[0]
return None
def needs_reasoning_echo(provider: Any, model: Any, base_url: Any) -> bool:
"""True when the endpoint requires reasoning_content echo-back."""
return reasoning_echo_family(provider, model, base_url) is not None
def apply_reasoning_content_policy(
source_msg: dict, api_msg: dict, needs_thinking_pad: bool
) -> None:
"""Copy provider-facing reasoning fields onto an API replay message.
``needs_thinking_pad`` is the require-side flag (see
``needs_reasoning_echo`` / the agent's cached
``_needs_thinking_reasoning_pad``). Mutates ``api_msg`` in place.
"""
if source_msg.get("role") != "assistant":
return
# 1. Explicit reasoning_content already set.
#
# When the active provider enforces the thinking-mode echo-back
# (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their
# own space-placeholder written at creation time and any valid reasoning
# from the same provider. Sessions persisted BEFORE #17341 have
# empty-string placeholders pinned at creation time; DeepSeek V4 Pro
# rejects those with HTTP 400, so upgrade "" → " " on replay.
#
# When the active provider does NOT enforce echo-back, strip the field
# entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq,
# SambaNova, …) reject ANY reasoning_content key in input messages with
# HTTP 400/422 ("Extra inputs are not permitted"), even an empty string
# or a single-space pad. This is the cross-provider fallback case: a
# reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a
# fallback to a strict provider replays that pad and 422s. Stripping
# here covers the rebuild path; ``reapply_reasoning_echo`` covers the
# already-built api_messages path. Refs #45655.
existing = source_msg.get("reasoning_content")
if isinstance(existing, str):
if not needs_thinking_pad:
api_msg.pop("reasoning_content", None)
elif existing == "":
api_msg["reasoning_content"] = " "
else:
api_msg["reasoning_content"] = existing
return
# 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi,
# if the source turn has tool_calls AND a 'reasoning' field but no
# 'reasoning_content' key, the 'reasoning' text was written by a
# prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message
# pins reasoning_content at creation time for tool-call turns, so the
# shape (reasoning set, reasoning_content absent, tool_calls present)
# is unreachable from same-provider DeepSeek history after this fix.
# Inject a single space to satisfy the API without leaking another
# provider's chain of thought to DeepSeek/Kimi. Space (not "")
# because DeepSeek V4 Pro rejects empty-string reasoning_content
# in thinking mode (refs #17341).
normalized_reasoning = source_msg.get("reasoning")
if (
needs_thinking_pad
and source_msg.get("tool_calls")
and isinstance(normalized_reasoning, str)
and normalized_reasoning
):
api_msg["reasoning_content"] = " "
return
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content'
# for providers that use the internal 'reasoning' key.
# This must happen before the unconditional empty-string fallback so
# genuine reasoning content is not overwritten (#15812 regression in
# PR #15478). Only promote for providers that enforce echo-back —
# strict providers reject the field (refs #45655).
if isinstance(normalized_reasoning, str) and normalized_reasoning:
if needs_thinking_pad:
api_msg["reasoning_content"] = normalized_reasoning
else:
api_msg.pop("reasoning_content", None)
return
# 4. DeepSeek / Kimi thinking mode: all assistant messages need
# reasoning_content. Inject a single space to satisfy the provider's
# requirement when no explicit reasoning content is present. Covers
# both tool-call turns (already-poisoned history with no reasoning
# at all) and plain text turns. Space (not "") because DeepSeek V4
# Pro tightened validation and rejects empty string with HTTP 400
# ("The reasoning content in the thinking mode must be passed back
# to the API"). Refs #17341.
if needs_thinking_pad:
api_msg["reasoning_content"] = " "
return
# 5. reasoning_content was present but not a string (e.g. None after
# context compaction). Don't pass null to the API.
api_msg.pop("reasoning_content", None)
def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int:
"""Re-pad (or strip) assistant turns' reasoning_content for the active provider.
``api_messages`` is built once, before the retry loop, while the *primary*
provider is active. A mid-conversation fallback can then switch providers,
so the reasoning fields baked into ``api_messages`` are shaped for the
*prior* provider and must be reconciled against the *current* one:
* Switching TO a require-side provider (DeepSeek / Kimi / MiMo thinking
mode): assistant turns built when the prior provider did NOT need the
echo-back go out without ``reasoning_content`` and the new provider
rejects them with HTTP 400 ("The reasoning_content in the thinking mode
must be passed back"). Re-apply the pad.
* Switching TO a strict provider that rejects the field (Mistral,
Cerebras, Groq, SambaNova, …): assistant turns built under a reasoning
primary carry a ``reasoning_content`` pad (often a single space ``" "``),
and the strict provider rejects it with HTTP 400/422 ("Extra inputs are
not permitted"). Strip the field. This is the exact cross-provider
fallback bug from #45655 — a DeepSeek primary pads history with ``" "``,
the request falls back to Mistral, and Mistral 422s on the stale pad.
Calling this immediately before building the request kwargs reconciles the
fields against the *current* provider. It is idempotent and safe to call
every iteration; it covers every fallback path.
Returns the number of assistant turns whose reasoning_content was added or
removed.
"""
changed = 0
for api_msg in api_messages:
if api_msg.get("role") != "assistant":
continue
if needs_thinking_pad:
if api_msg.get("reasoning_content"):
continue
apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad)
if api_msg.get("reasoning_content"):
changed += 1
else:
# Strict provider — strip any stale reasoning_content pad left
# over from a reasoning primary so the fallback request doesn't
# 400/422 on it.
if "reasoning_content" in api_msg:
api_msg.pop("reasoning_content", None)
changed += 1
return changed
# ---------------------------------------------------------------------------
# Image / multimodal parts — evaluated, NOT consolidated (verdict: syntax)
# ---------------------------------------------------------------------------
#
# The per-adapter image handling is format-specific SYNTAX, not shared policy:
# * anthropic_adapter (~1817): data-URL → Anthropic `source: {type: base64}`
# block mapping — Anthropic wire shape only.
# * codex_responses_adapter (~113/165/812): chat `image_url` parts →
# Responses `input_image` items and image counting for log summaries —
# Responses wire shape only.
# * transports/chat_completions: pass-through (native format).
# The one genuinely shared image POLICY — removing images when a server
# rejects them while preserving tool_call_id pairing — already has a single
# owner here: ``_strip_images_from_messages`` above.
+85 -7
View File
@@ -72,7 +72,7 @@ def _resolve_requests_verify() -> bool | str:
_PROVIDER_PREFIXES: frozenset[str] = frozenset({
"openrouter", "nous", "openai-codex", "copilot", "copilot-acp",
"gemini", "ollama-cloud", "zai", "kimi-coding", "kimi-coding-cn", "stepfun", "minimax", "minimax-oauth", "minimax-cn", "anthropic", "deepseek", "deepinfra",
"opencode-zen", "opencode-go", "kilocode", "alibaba", "novita",
"opencode-zen", "opencode-go", "ai-gateway", "kilocode", "alibaba", "novita",
"qwen-oauth",
"xiaomi",
"arcee",
@@ -84,7 +84,7 @@ _PROVIDER_PREFIXES: frozenset[str] = frozenset({
"glm", "z-ai", "z.ai", "zhipu", "github", "github-copilot",
"github-models", "kimi", "moonshot", "kimi-cn", "moonshot-cn", "claude", "deep-seek", "deep-infra",
"ollama",
"stepfun", "opencode", "zen", "go", "kilo", "dashscope", "aliyun", "qwen",
"stepfun", "opencode", "zen", "go", "vercel", "kilo", "dashscope", "aliyun", "qwen",
"mimo", "xiaomi-mimo",
"tencent", "tokenhub", "tencent-cloud", "tencentmaas",
"arcee-ai", "arceeai",
@@ -2851,14 +2851,92 @@ def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
image — the Anthropic pricing model — instead of counting raw base64
character length. Without this, a single ~1MB screenshot would be
estimated at ~250K tokens and trigger premature context compression.
Per-message results are memoized (see ``_estimate_message_tokens_cached``)
keyed on a deep *identity fingerprint* of the message, so re-walking a
long history every iteration only pays for messages whose object graph
actually changed. The memo is exact: equal fingerprints imply identical
leaf objects and structure, hence an identical estimate.
"""
_IMAGE_TOKEN_COST = 1500
text_tokens = 0
image_tokens = 0
total = 0
for msg in messages:
text_tokens += _estimate_message_tokens_without_images(msg)
image_tokens += _count_image_tokens(msg, _IMAGE_TOKEN_COST)
return text_tokens + image_tokens
total += _estimate_message_tokens_cached(msg, _IMAGE_TOKEN_COST)
return total
# --- Per-message token-estimate memo -------------------------------------
#
# ``estimate_messages_tokens_rough`` is called on the full history every
# loop iteration (conversation_loop preflight), repeatedly during compaction
# telemetry, and inside an O(n^2) shrink loop in moa_loop. The per-message
# helpers are pure functions of the message's value, so a memo keyed on a
# fingerprint that uniquely determines the value is exactly equivalent.
#
# Fingerprint design (soundness argument):
# * strings are fingerprinted by ``id()`` AND pinned (a strong reference is
# stored in the cache entry). While the entry lives, that id cannot be
# reused by another object, so id-equality implies object-equality —
# strings are immutable, so value-equality too (no #50372-style aliasing).
# * ints/floats/bools/None are fingerprinted by value.
# * dicts/lists recurse structurally, preserving key order — ``str(shadow)``
# depends on insertion order, so order is part of the key.
# * any other type aborts the memo and falls through to a direct compute.
# Equal fingerprints therefore imply deep-equal messages built from identical
# immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical estimate.
#
# Because the api_messages build shallow-copies history dicts each iteration,
# the copies share the same content strings — so unchanged history messages
# hit the memo even though the outer dicts are fresh objects every turn.
_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {}
_MSG_TOKENS_CACHE_MAX = 4096
def _msg_fingerprint(value: Any, pins: list) -> Any:
if value is None or value is True or value is False:
return value
t = type(value)
if t is str:
pins.append(value)
return ("s", id(value))
if t is int or t is float:
return ("n", t.__name__, value)
if t is dict:
return ("d", tuple(
(_msg_fingerprint(k, pins), _msg_fingerprint(v, pins))
for k, v in value.items()
))
if t is list:
return ("l", tuple(_msg_fingerprint(v, pins) for v in value))
if t is tuple:
return ("t", tuple(_msg_fingerprint(v, pins) for v in value))
raise ValueError("unfingerprintable message value")
def _estimate_message_tokens_cached(msg: Any, image_cost: int) -> int:
try:
pins: list = []
key = _msg_fingerprint(msg, pins)
hash(key)
except Exception:
return (
_estimate_message_tokens_without_images(msg)
+ _count_image_tokens(msg, image_cost)
)
cached = _MSG_TOKENS_CACHE.get(key)
if cached is not None:
return cached[1]
tokens = (
_estimate_message_tokens_without_images(msg)
+ _count_image_tokens(msg, image_cost)
)
_MSG_TOKENS_CACHE[key] = (pins, tokens)
while len(_MSG_TOKENS_CACHE) > _MSG_TOKENS_CACHE_MAX:
try:
_MSG_TOKENS_CACHE.pop(next(iter(_MSG_TOKENS_CACHE)))
except (StopIteration, KeyError, RuntimeError):
break
return tokens
def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
+1
View File
@@ -168,6 +168,7 @@ PROVIDER_TO_MODELS_DEV: Dict[str, str] = {
"alibaba": "alibaba",
"qwen-oauth": "alibaba",
"copilot": "github-copilot",
"ai-gateway": "vercel",
"opencode-zen": "opencode",
"opencode-go": "opencode-go",
"kilocode": "kilo",
+2 -2
View File
@@ -210,8 +210,8 @@ def _resolve_trust_policy(plugin_id: str) -> _TrustPolicy:
return _TrustPolicy(plugin_id="")
try:
from hermes_cli.config import load_config
config = load_config() or {}
from hermes_cli.config import load_config_readonly
config = load_config_readonly() or {}
except Exception: # pragma: no cover — config IO failure
return _TrustPolicy(plugin_id=plugin_id)
+122 -25
View File
@@ -19,13 +19,18 @@ from typing import Optional
from agent.runtime_cwd import resolve_agent_cwd
from agent.skill_utils import (
EXCLUDED_SKILL_DIRS,
ORG_ACTIVE_MARKER,
ORG_MIRROR_DIR_NAME,
ORG_PROVENANCE_FILE,
SKILL_SUPPORT_DIRS,
extract_skill_conditions,
extract_skill_description,
get_all_skills_dirs,
get_disabled_skill_names,
iter_skill_index_files,
org_id_of_path,
parse_frontmatter,
read_active_org_id,
skill_matches_environment,
skill_matches_platform,
skill_matches_platform_list,
@@ -567,16 +572,18 @@ def computer_use_guidance(platform_name: Optional[str] = None) -> str:
"Background delivery is the DEFAULT and the co-work path, but it is "
"the first rung, not the only one. Read each action's structured "
"result and climb only when the driver tells you to:\n"
"- `effect: 'confirmed'` + `verified: true` — the driver read the "
"result back. Done.\n"
"- `effect: 'confirmed'` (or `verified: true`) — done, even if an "
"advisory escalation is also present. Never repeat successful input.\n"
"- `effect: 'unverifiable'` — the input was delivered but the driver "
"can't confirm it. Re-capture and check the screenshot/tree yourself "
"before deciding it worked.\n"
"- `effect: 'suspected_noop'`, `code: 'background_unavailable'`, or an "
"`escalation.recommended` field — the action did NOT land. Follow "
"`escalation.recommended`:\n"
"can't confirm it. Get fresh state and check it before any retry; an "
"escalation recommendation does not override this rule.\n"
"- `effect: 'suspected_noop'` or a structured refusal such as "
"`code: 'background_unavailable'` — escalation is allowed. Follow "
"the recommended rung when present:\n"
" - `'px'` → re-issue addressing the target by `coordinate=[x,y]` "
"read off the screenshot instead of `element`.\n"
" - `'page'` → use the exact-bound typed browser page rung below "
"before native foreground escalation. Do not start a legacy page workflow.\n"
" - `'foreground'` (or a pixel click still didn't land) → re-issue "
"the SAME action with `delivery_mode='foreground'`. This briefly "
"raises the window; it needs its own approval and is only appropriate "
@@ -586,6 +593,21 @@ def computer_use_guidance(platform_name: Optional[str] = None) -> str:
"as a prediction from the app being Electron/Chromium/GTK. Do not "
"silently retry the same rung expecting a different result, and do "
"not conclude 'cua-driver can't drive this app' — climb the ladder.\n\n"
"## Typed browser page rung\n"
"For `recommended='page'` or supported browser PAGE content, use the namespaced "
"`cua_browser_*` actions: bind with `cua_browser_state` using the exact "
"native `(pid, window_id)`, require `binding_quality='exact'` and "
"`mutation_allowed=true`, select its opaque `tab_id`, then take a "
"fresh semantic snapshot before using a current `ref`. After every "
"typed mutation, call `cua_browser_state` again before another action. "
"Input defaults to trusted; `input_route='dom_event'` is an explicit "
"downgrade, never an automatic retry. Use native capture/input for "
"browser chrome, OS permission prompts, native dialogs, and unsupported "
"targets. Browser setup is a separately approved action; attaching an "
"existing profile is enforced by cua-driver's immutable permission "
"mode: standard requires a certified protected host and fails closed "
"when Hermes has none; explicit Hermes YOLO uses a private unrestricted "
"daemon after the user's launch/session risk acceptance.\n\n"
"## Background mode rules\n"
"- Do NOT use `raise_window=true` on `focus_app` unless the user "
"explicitly asked you to bring a window to front. Input routing to "
@@ -959,7 +981,7 @@ WSL_ENVIRONMENT_HINT = (
# misleading — the agent should only see the machine it can actually touch.
_REMOTE_TERMINAL_BACKENDS = frozenset({
"docker", "singularity", "modal", "daytona", "ssh",
"managed_modal",
"vercel_sandbox", "managed_modal",
})
@@ -973,6 +995,7 @@ _BACKEND_FALLBACK_DESCRIPTIONS: dict[str, str] = {
"modal": "a Modal sandbox (Linux)",
"managed_modal": "a managed Modal sandbox (Linux)",
"daytona": "a Daytona workspace (Linux)",
"vercel_sandbox": "a Vercel sandbox (Linux)",
"ssh": "a remote host reached over SSH (likely Linux)",
}
@@ -1047,7 +1070,7 @@ def _probe_remote_backend(env_type: str) -> str | None:
}
container_config = None
if env_type in {"docker", "singularity", "modal", "daytona"}:
if env_type in {"docker", "singularity", "modal", "daytona", "vercel_sandbox"}:
container_config = {
"container_cpu": config.get("container_cpu", 1),
"container_memory": config.get("container_memory", 5120),
@@ -1137,7 +1160,7 @@ def build_environment_hints() -> str:
and a Windows-only note that `terminal` shells out to bash, not
PowerShell).
- For **remote / sandbox** terminal backends (docker, singularity,
modal, daytona, ssh): host info is **suppressed**
modal, daytona, ssh, vercel_sandbox): host info is **suppressed**
because the agent's tools can't touch the host — only the backend
matters. A live probe inside the backend reports its OS, user, $HOME,
and cwd. Falls back to a static summary if the probe fails.
@@ -1224,10 +1247,10 @@ def build_environment_hints() -> str:
extra = (os.getenv("HERMES_ENVIRONMENT_HINT") or "").strip()
if not extra:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
extra = str(
(load_config().get("agent", {}) or {}).get("environment_hint", "")
(load_config_readonly().get("agent", {}) or {}).get("environment_hint", "")
).strip()
except Exception as e:
logger.debug("Could not read agent.environment_hint from config: %s", e)
@@ -1278,9 +1301,9 @@ def _get_context_file_max_chars(context_length: Optional[int] = None) -> int:
3. ``CONTEXT_FILE_MAX_CHARS`` (20K) as the upstream-compatible fallback.
"""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
val = load_config().get("context_file_max_chars")
val = load_config_readonly().get("context_file_max_chars")
if isinstance(val, (int, float)) and val > 0:
return int(val)
except Exception as e:
@@ -1323,7 +1346,9 @@ def drain_truncation_warnings() -> list:
_SKILLS_PROMPT_CACHE_MAX = 8
_SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict()
_SKILLS_PROMPT_CACHE_LOCK = threading.Lock()
_SKILLS_SNAPSHOT_VERSION = 1
# v2: entries gained org provenance fields (org_id/org_author/rel_dir) for M2
# org-shared skills; older snapshots are discarded and rebuilt.
_SKILLS_SNAPSHOT_VERSION = 2
def _skills_prompt_snapshot_path() -> Path:
@@ -1342,13 +1367,32 @@ def clear_skills_system_prompt_cache(*, clear_snapshot: bool = False) -> None:
def _build_skills_manifest(skills_dir: Path) -> dict[str, list[int]]:
"""Build an mtime/size manifest of all SKILL.md and DESCRIPTION.md files."""
"""Build an mtime/size manifest of all SKILL.md and DESCRIPTION.md files.
Org mirrors (M2): only the ACTIVE org's mirror participates, and the
``.active_org`` marker itself is included — so switching/leaving an org
invalidates the snapshot even when no SKILL.md changed.
"""
manifest: dict[str, list[int]] = {}
skills_dir_str = str(skills_dir)
base = os.path.join(skills_dir_str, "")
prefix_len = len(base)
active_org = read_active_org_id(skills_dir)
org_root = os.path.join(skills_dir_str, ORG_MIRROR_DIR_NAME)
marker_path = os.path.join(org_root, ORG_ACTIVE_MARKER)
try:
st = os.stat(marker_path)
manifest[ORG_MIRROR_DIR_NAME + "/" + ORG_ACTIVE_MARKER] = [
int(st.st_mtime), int(st.st_size),
]
except OSError:
pass
for root, dirs, files in os.walk(skills_dir_str, followlinks=True):
has_skill_md = "SKILL.md" in files
if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None:
dirs.remove(ORG_MIRROR_DIR_NAME)
elif root == org_root:
dirs[:] = [d for d in dirs if d == active_org]
dirs[:] = [
d
for d in dirs
@@ -1413,6 +1457,15 @@ def _build_snapshot_entry(
"""Build a serialisable metadata dict for one skill."""
rel_path = skill_file.relative_to(skills_dir)
parts = rel_path.parts
# M2 org mirror: strip the `_org/<org_id>/` prefix so category/name derive
# from the path WITHIN the mirror (same shape the org tree was built
# from), and record provenance for labeling + fail-loud collisions.
org_id: str | None = None
if len(parts) >= 3 and parts[0] == ORG_MIRROR_DIR_NAME:
org_id = parts[1]
parts = parts[2:]
if len(parts) >= 2:
skill_name = parts[-2]
category = "/".join(parts[:-2]) if len(parts) > 2 else parts[0]
@@ -1424,7 +1477,7 @@ def _build_snapshot_entry(
if isinstance(platforms, str):
platforms = [platforms]
return {
entry = {
"skill_name": skill_name,
"category": category,
"frontmatter_name": str(frontmatter.get("name", skill_name)),
@@ -1432,6 +1485,22 @@ def _build_snapshot_entry(
"platforms": [str(p).strip() for p in platforms if str(p).strip()],
"conditions": extract_skill_conditions(frontmatter),
}
if org_id:
entry["org_id"] = org_id
# Author from the pull-time provenance sidecar (token-verified at
# push by the plane's author_mismatch guard). Best-effort.
try:
import json as _json
prov_path = (
skills_dir / ORG_MIRROR_DIR_NAME / org_id / ORG_PROVENANCE_FILE
)
prov = _json.loads(prov_path.read_text(encoding="utf-8"))
device = str(prov.get("author_device") or "")
entry["org_author"] = device or str(prov.get("author_user_id") or "")
except Exception:
entry["org_author"] = ""
return entry
# =========================================================================
@@ -1567,6 +1636,10 @@ def build_skills_system_prompt(
skills_by_category: dict[str, list[tuple[str, str]]] = {}
category_descriptions: dict[str, str] = {}
# Unified visible-entry list (both paths) so the org labeling +
# fail-loud collision pass below runs identically for snapshot and scan.
visible_entries: list[dict] = []
skill_entries: list[dict] = []
if snapshot is not None:
# Fast path: use pre-parsed metadata from disk
@@ -1574,7 +1647,6 @@ def build_skills_system_prompt(
if not isinstance(entry, dict):
continue
skill_name = entry.get("skill_name") or ""
category = entry.get("category") or "general"
frontmatter_name = entry.get("frontmatter_name") or skill_name
platforms = entry.get("platforms") or []
if not skill_matches_platform_list(platforms):
@@ -1587,16 +1659,13 @@ def build_skills_system_prompt(
available_toolsets,
):
continue
skills_by_category.setdefault(category, []).append(
(frontmatter_name, entry.get("description", ""))
)
visible_entries.append(entry)
category_descriptions = {
str(k): str(v)
for k, v in (snapshot.get("category_descriptions") or {}).items()
}
else:
# Cold path: full filesystem scan + write snapshot for next time
skill_entries: list[dict] = []
for skill_file in iter_skill_index_files(skills_dir, "SKILL.md"):
is_compatible, frontmatter, desc = _parse_skill_file(skill_file)
entry = _build_snapshot_entry(skill_file, skills_dir, frontmatter, desc)
@@ -1612,10 +1681,38 @@ def build_skills_system_prompt(
available_toolsets,
):
continue
skills_by_category.setdefault(entry["category"], []).append(
(entry["frontmatter_name"], entry["description"])
)
visible_entries.append(entry)
# ── M2 org labeling + FAIL-LOUD collisions ─────────────────────────
# An org skill lists with an explicit provenance tag. When a personal and
# an org skill share a name, NEITHER silently wins: both list qualified
# (personal keeps the bare name is the wrong default — silent divergence
# from the org set; org winning silently shadows the user's own work) —
# so both entries carry a [name collision] flag and skill_view refuses
# the ambiguous bare name (its existing multi-candidate guard).
name_owners: dict[str, set[str]] = {}
for entry in visible_entries:
fm = entry.get("frontmatter_name") or entry.get("skill_name") or ""
kind = "org" if entry.get("org_id") else "personal"
name_owners.setdefault(fm, set()).add(kind)
for entry in visible_entries:
fm = entry.get("frontmatter_name") or entry.get("skill_name") or ""
desc = entry.get("description", "")
org_id = entry.get("org_id")
collided = len(name_owners.get(fm, set())) > 1
if org_id:
author = entry.get("org_author") or ""
tag = f"[org-shared{': by ' + author if author else ''}]"
desc = f"{tag} {desc}".strip()
category = f"org:{org_id}"
else:
category = entry.get("category") or "general"
if collided:
desc = f"[name collision — also exists {'personally' if org_id else 'in your org'}; load via category path] {desc}".strip()
skills_by_category.setdefault(category, []).append((fm, desc))
if snapshot is None:
# (continuation of the cold path below: category descriptions + write)
# Read category-level DESCRIPTION.md files
for desc_file in iter_skill_index_files(skills_dir, "DESCRIPTION.md"):
try:
+24 -2
View File
@@ -140,6 +140,11 @@ _ENV_ASSIGN_RE = re.compile(
# The colon-form URL guard (skip when ``://`` present) lives at the call site.
_SECRET_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential|auth)"
_CFG_VALUE = r"(['\"]?)([^\s&]+?)\2(?=[\s&]|$)"
# Linear pre-gate for the _CFG_*_RE subs below: a text with no secret keyword
# can never match either pattern, so the (potentially backtrack-heavy) subs
# are skipped entirely for such text. See the call site in
# redact_sensitive_text().
_CFG_SECRET_WORD_RE = re.compile(_SECRET_CFG_NAMES, re.IGNORECASE)
# Programmatic env lookups (``os.getenv(...)``, ``os.environ[...]``,
# ``os.environ.get(...)``, ``process.env.X``, ``$ENV{X}``) reference variable
@@ -377,8 +382,17 @@ _STRICT_URL_PARAM_RE = re.compile(
# Match userinfo in both absolute (``scheme://user:pass@host``) and
# network-path (``//user:pass@host``) references. The authority boundary stops
# at path/query/fragment delimiters so an ``@`` elsewhere in a URL is ignored.
#
# Anchored on the mandatory ``//`` rather than an optional scheme prefix: the
# scheme sits outside the match either way (replacement callbacks re-emit
# group(1), so ``https:`` stays untouched in the surrounding text), and the
# old optional-scheme prefix ``(?:[A-Za-z][A-Za-z0-9+.-]*:)?`` backtracked
# catastrophically (O(n²)) on long unbroken alphanumeric runs — a 320KB
# synthetic compaction payload spent ~55s inside this pattern per sub() call.
# Output-equivalence to the old pattern was fuzz-verified (20k random strings
# plus targeted URL forms).
_STRICT_URL_USERINFO_RE = re.compile(
r"((?:[A-Za-z][A-Za-z0-9+.-]*:)?//)([^/\s?#@]+)@"
r"(//)([^/\s?#@]+)@"
)
# HTTP access logs often use a relative request target rather than a full URL:
@@ -706,7 +720,15 @@ def redact_sensitive_text(
# web-URL query params are intentionally passed through (see note
# near the bottom of this function); _DB_CONNSTR_RE still guards
# connection-string passwords.
if "://" not in text:
#
# Extra gate: every _CFG_*_RE match requires a secret keyword in
# the key, so a text without any secret keyword cannot match —
# skipping is exact. This matters because _CFG_DOTTED_RE
# backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs
# (e.g. base64/hex blobs in compaction payloads); the linear
# keyword scan prevents that pathological path on secret-free
# text.
if "://" not in text and _CFG_SECRET_WORD_RE.search(text):
text = _CFG_DOTTED_RE.sub(_redact_env, text)
text = _CFG_ANCHORED_RE.sub(_redact_env, text)
+78 -9
View File
@@ -336,6 +336,14 @@ class ManagedLlmStream(Iterator[Any]):
self._callback_error: BaseException | None = None
self._logical: tuple[relay_runtime.RelayTurnContext, Any, str] | None = None
self._defer_logical_completion = defer_logical_completion
if str((metadata or {}).get("call_role") or "").startswith("auxiliary:"):
self._logical_model_name: str | None = model_name
self._logical_provider_name: str | None = name
self._logical_response_model_name: str | None = None
else:
self._logical_model_name = None
self._logical_provider_name = None
self._logical_response_model_name = None
self._on_chunk = on_chunk
self._chunk_adapter = chunk_adapter or _namespace
self._accept_chunk = accept_chunk
@@ -445,8 +453,12 @@ class ManagedLlmStream(Iterator[Any]):
return None
try:
if self.final_response is not None:
return _jsonable(self.final_response)
return _jsonable(run_callback(finalizer))
response = self.final_response
else:
response = run_callback(finalizer)
if self._logical_model_name is not None:
self._logical_response_model_name = _response_model_name(response)
return _jsonable(response)
except BaseException as exc:
self._callback_error = exc
raise
@@ -488,6 +500,9 @@ class ManagedLlmStream(Iterator[Any]):
_complete_logical(
self._logical,
outcome="cancelled" if _is_cancellation(exc) else "failed",
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
loop.close()
@@ -520,7 +535,13 @@ class ManagedLlmStream(Iterator[Any]):
if self._raw_chunks:
self.output_modified = True
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome="success")
_complete_logical(
self._logical,
outcome="success",
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
self._close(logical_outcome="cancelled")
raise StopIteration from None
@@ -593,7 +614,13 @@ class ManagedLlmStream(Iterator[Any]):
)
loop.close()
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome="success")
_complete_logical(
self._logical,
outcome="success",
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
def _close(self, *, logical_outcome: str) -> None:
@@ -623,7 +650,13 @@ class ManagedLlmStream(Iterator[Any]):
exc_info=True,
)
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome=logical_outcome)
_complete_logical(
self._logical,
outcome=logical_outcome,
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
return
close = getattr(self._stream, "aclose", None)
@@ -638,7 +671,13 @@ class ManagedLlmStream(Iterator[Any]):
if self._close_error is None:
self._close_error = exc
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome=logical_outcome)
_complete_logical(
self._logical,
outcome=logical_outcome,
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
loop.close()
@@ -777,6 +816,9 @@ def _complete_logical(
logical: tuple[relay_runtime.RelayTurnContext, Any, str] | None,
*,
outcome: str,
model_name: str | None = None,
provider_name: str | None = None,
response_model_name: str | None = None,
) -> None:
if logical is None:
return
@@ -791,11 +833,16 @@ def _complete_logical(
if lease.session is None:
return
try:
output = {"outcome": outcome}
if model_name is not None and provider_name is not None:
output.update({"model": model_name, "provider": provider_name})
if response_model_name is not None:
output["response_model"] = response_model_name
lease.host.run_in_session(
lease.session,
lease.host.relay.scope.pop,
handle,
output={"outcome": outcome},
output=output,
metadata={
relay_runtime.RUNTIME_SCHEMA_KEY: relay_runtime.RUNTIME_SCHEMA_VERSION,
relay_runtime.RUNTIME_INSTANCE_KEY: lease.host.runtime_id,
@@ -845,7 +892,14 @@ def _is_cancellation(error: BaseException) -> bool:
)
def complete_logical_call(api_request_id: str, *, outcome: str) -> None:
def complete_logical_call(
api_request_id: str,
*,
outcome: str,
model_name: str | None = None,
provider_name: str | None = None,
response_model_name: str | None = None,
) -> None:
"""Complete the active turn's logical LLM call after caller validation."""
turn = relay_runtime.active_turn()
if turn is None or not api_request_id:
@@ -853,7 +907,22 @@ def complete_logical_call(api_request_id: str, *, outcome: str) -> None:
with turn.logical_llm_lock:
handle = turn.logical_llm_calls.get(api_request_id)
if handle is not None:
_complete_logical((turn, handle, api_request_id), outcome=outcome)
_complete_logical(
(turn, handle, api_request_id),
outcome=outcome,
model_name=model_name,
provider_name=provider_name,
response_model_name=response_model_name,
)
def _response_model_name(response: Any) -> str | None:
"""Return a provider-reported model name when one is available."""
if isinstance(response, dict):
value = response.get("model")
else:
value = getattr(response, "model", None)
return value if isinstance(value, str) and value.strip() else None
def _provider_request(
+2 -2
View File
@@ -25,9 +25,9 @@ _INLINE_SHELL_MAX_OUTPUT = 4000
def load_skills_config() -> dict:
"""Load the ``skills`` section of config.yaml (best-effort)."""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config() or {}
cfg = load_config_readonly() or {}
skills_cfg = cfg.get("skills")
if isinstance(skills_cfg, dict):
return skills_cfg
+62
View File
@@ -49,6 +49,55 @@ EXCLUDED_SKILL_DIRS = frozenset(
# archive workflow preserves a complete old skill package under references/.
SKILL_SUPPORT_DIRS = frozenset(("references", "templates", "assets", "scripts"))
# ── Org-shared skills (sync contract) ───────────────────────────
# Org mirrors live under ~/.hermes/skills/_org/<org_id>/. Resolution is
# TOKEN-GATED via a marker file the sync client writes after verifying the
# token (skills_sync_client.pull_org_skills): only the marked org's mirror is
# scanned. No marker ⇒ no org skills load. The marker is plain data (org_id
# string) so this module stays import-light; the VERIFICATION lives in the
# sync client, which is the only writer. Offline grace: the marker persists,
# so already-pulled org skills keep working without connectivity; a VERIFIED
# org change (or personal-org token) rewrites/removes it.
ORG_MIRROR_DIR_NAME = "_org"
ORG_ACTIVE_MARKER = ".active_org"
ORG_PROVENANCE_FILE = ".org-provenance.json"
# Records the fingerprint of each skill exactly as upstream sent it, so a
# later local edit is detectable and an org pull can refuse to clobber it.
ORG_BASELINE_FILE = ".org-baseline.json"
def read_active_org_id(skills_dir: Path) -> Optional[str]:
"""The org id whose mirror may resolve, or None (no org skills load)."""
try:
marker = skills_dir / ORG_MIRROR_DIR_NAME / ORG_ACTIVE_MARKER
if not marker.exists():
return None
val = marker.read_text(encoding="utf-8").strip()
return val or None
except OSError:
return None
def is_org_mirror_path(path, skills_dir: Path) -> bool:
"""True when *path* is inside the org mirror (``_org/``)."""
try:
rel = Path(path).resolve().relative_to(Path(skills_dir).resolve())
except (OSError, ValueError):
return False
return bool(rel.parts) and rel.parts[0] == ORG_MIRROR_DIR_NAME
def org_id_of_path(path, skills_dir: Path) -> Optional[str]:
"""The ``<org_id>`` segment for a path under ``_org/<org_id>/...``."""
try:
rel = Path(path).resolve().relative_to(Path(skills_dir).resolve())
except (OSError, ValueError):
return None
if len(rel.parts) >= 2 and rel.parts[0] == ORG_MIRROR_DIR_NAME:
return rel.parts[1]
return None
def is_excluded_skill_path(path, *, root: Optional[Path] = None) -> bool:
"""True if *path* should be skipped by active skill scanners.
@@ -817,11 +866,24 @@ def iter_skill_index_files(skills_dir: Path, filename: str):
scripts) can contain arbitrary markdown and even archived package
``SKILL.md`` files, but they are progressive-disclosure data loaded through
``skill_view(..., file_path=...)`` rather than active skill roots.
M2 org mirrors (``_org/``): TOKEN-GATED resolution. Only the active org's
subdir (per the sync-client-written ``.active_org`` marker) is walked;
every other ``_org/<id>/`` (stale mirror from a previous org, or no
marker at all) is pruned — leave an org and its skills stop resolving,
without any manual cleanup.
"""
skills_dir_str = str(skills_dir)
active_org = read_active_org_id(skills_dir)
org_root = os.path.join(skills_dir_str, ORG_MIRROR_DIR_NAME)
matches: list[str] = []
for root, dirs, files in os.walk(skills_dir_str, followlinks=True):
has_skill_md = "SKILL.md" in files
if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None:
dirs.remove(ORG_MIRROR_DIR_NAME)
elif root == org_root:
# Inside _org/: descend ONLY into the active org's mirror.
dirs[:] = [d for d in dirs if d == active_org]
dirs[:] = [
d
for d in dirs
+1 -1
View File
@@ -18,7 +18,7 @@ import secrets
import threading
import time
from contextlib import contextmanager
from concurrent.futures import Future, ThreadPoolExecutor, TimeoutError
from concurrent.futures import Future, TimeoutError
from typing import Any, Callable, Mapping, Optional
+2 -2
View File
@@ -43,10 +43,10 @@ _TITLE_PROMPT_PINNED_LANGUAGE = (
def _title_language() -> str:
"""Return configured title language, or empty string to match the user."""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
return str(
((load_config() or {}).get("auxiliary") or {})
((load_config_readonly() or {}).get("auxiliary") or {})
.get("title_generation", {})
.get("language", "")
).strip()
-4
View File
@@ -32,7 +32,6 @@ from agent.display import (
redact_tool_args_for_display as _redact_tool_args_for_display,
_detect_tool_failure,
)
from agent.tool_guardrails import ToolGuardrailDecision
from agent.tool_dispatch_helpers import (
_is_destructive_command,
_is_multimodal_tool_result,
@@ -2057,9 +2056,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
return
break
if agent.tool_delay > 0 and i < len(assistant_message.tool_calls):
time.sleep(agent.tool_delay)
# ── Per-turn aggregate budget enforcement ─────────────────────────
num_tools_seq = len(assistant_message.tool_calls)
if finalize and num_tools_seq > 0:
-2
View File
@@ -378,7 +378,6 @@ class ChatCompletionsTransport(ProviderTransport):
ephemeral = params.get("ephemeral_max_output_tokens")
max_tokens = params.get("max_tokens")
anthropic_max_out = params.get("anthropic_max_output")
is_nvidia_nim = params.get("is_nvidia_nim", False)
is_kimi = params.get("is_kimi", False)
is_tokenhub = params.get("is_tokenhub", False)
reasoning_config = _reasoning_config_for_model(model, params.get("reasoning_config"))
@@ -436,7 +435,6 @@ class ChatCompletionsTransport(ProviderTransport):
extra_body: dict[str, Any] = {}
is_openrouter = params.get("is_openrouter", False)
is_nous = params.get("is_nous", False)
is_github_models = params.get("is_github_models", False)
provider_name = str(params.get("provider_name") or "").strip().lower()
base_url = params.get("base_url")
+40
View File
@@ -349,6 +349,46 @@ def finalize_turn(
_apply_override = getattr(agent, "_apply_persist_user_message_override", None)
if callable(_apply_override):
_apply_override(messages)
# ── Post-turn micro-compaction ────────────────────────────
# After the assistant response is finalized but before the session is
# persisted, run micro-compaction to absorb the oldest uncompacted
# exchange into the rolling summary. This amortizes compression
# across turns rather than batching it into one big pause.
if not interrupted and not failed:
try:
_compressor = getattr(agent, "context_compressor", None)
# Strict `is True` + isinstance gates: plugin context engines
# (and MagicMock compressors in tests) satisfy getattr/duck
# checks with truthy auto-attributes — a bare truthiness check
# here called _micro_compact on a mock and spliced its (empty-
# iterating) return value over the transcript, wiping it.
if (
_compressor
and getattr(_compressor, '_micro_compact_enabled', False) is True
and callable(getattr(_compressor, '_micro_compact', None))
and final_response
# Persistence-isolated agents (background review fork)
# must not micro-compact: the pass burns a real aux-LLM
# call on a throwaway replay transcript, and if the
# compressor ever holds a session_db binding it would
# archive_and_compact the CANONICAL session rows — the
# exact write class _persist_disabled exists to stop.
and not getattr(agent, "_persist_disabled", False)
):
_before = len(messages)
_compacted = _compressor._micro_compact(messages)
if isinstance(_compacted, list) and _compacted:
messages[:] = _compacted
_after = len(messages)
if _before != _after:
logger.info(
"Micro-compaction: %d -> %d messages",
_before, _after,
)
except Exception as _mc_err:
logger.info("Micro-compaction failed: %s", _mc_err)
agent._persist_session(messages, conversation_history)
except Exception as _persist_err:
_cleanup_errors.append(f"persist_session: {_persist_err}")
+2 -2
View File
@@ -1244,8 +1244,8 @@ def normalize_usage(
output_tokens = _to_int(getattr(response_usage, "completion_tokens", 0))
details = getattr(response_usage, "prompt_tokens_details", None)
# Primary: OpenAI-style prompt_tokens_details. Fallback: Anthropic-style
# top-level fields that some OpenAI-compatible proxies (OpenRouter, Cline)
# expose when routing Claude models — without this
# top-level fields that some OpenAI-compatible proxies (OpenRouter, Vercel
# AI Gateway, Cline) expose when routing Claude models — without this
# fallback, cache writes are undercounted as 0 and cache reads can be
# missed when the proxy only surfaces them at the top level.
# Port of cline/cline#10266.
+14 -54
View File
@@ -72,64 +72,24 @@ def _filter_verifiable_paths(paths: Iterable[str]) -> list[str]:
return [p for p in paths if p and not _is_non_code_path(p)]
# Session identities (platform or source) that are NOT human conversational
# messaging surfaces: interactive coding surfaces (CLI, TUI, desktop, codex,
# local, gateway) and programmatic callers (API server, webhooks, tools).
# Verify-on-stop stays ON by default for these. Any other resolved gateway
# platform is a conversational messaging surface (Telegram, Discord, WhatsApp,
# Signal, Slack, etc.) where the verification narrative would reach a human as
# chat noise, so it defaults OFF. Mirrors LOCAL_SESSION_SOURCE_IDS in
# apps/desktop/src/lib/session-source.ts; keep roughly in sync when adding a
# local or programmatic surface. Default-deny by design: an unrecognized
# identity is treated as messaging (OFF) so a new chat platform never leaks the
# verification receipt before this set is updated.
_NON_MESSAGING_SESSION_SURFACES = frozenset(
{
"",
"cli",
"codex",
"desktop",
"gateway",
"local",
"tui",
"tool",
"api_server",
"webhook",
"msgraph_webhook",
}
)
def _session_is_messaging_surface() -> bool:
"""Return whether this turn is delivered over a human messaging channel.
"""Whether this turn is delivered over a human messaging channel.
The gateway binds the platform value (e.g. ``telegram``) to
``HERMES_SESSION_PLATFORM``; the CLI and TUI set ``HERMES_SESSION_SOURCE``
(e.g. ``cli``, ``tui``) instead. Both are consulted via the session-context
helper (with an ``os.environ`` fallback), alongside the ``HERMES_PLATFORM``
override, matching the sibling platform resolution in
``agent/skill_commands.py`` and ``agent/prompt_builder.py``. A turn is a
messaging surface when a resolved identity is present and is not a known
non-messaging surface.
Verify-on-stop defaults ON for the interactive coding surfaces and
programmatic callers, and OFF on a conversational platform (Telegram,
Discord, Slack, ...) where the verification narrative reaches a human as
chat noise. The surface classification itself is shared with the other
consumers of this distinction — see
``gateway.session_context.session_is_messaging_surface``.
"""
try:
from gateway.session_context import get_session_env
from gateway.session_context import session_is_messaging_surface
platform = (
os.getenv("HERMES_PLATFORM")
or get_session_env("HERMES_SESSION_PLATFORM", "")
)
source = get_session_env("HERMES_SESSION_SOURCE", "")
return session_is_messaging_surface()
except Exception:
platform = os.getenv("HERMES_PLATFORM", "") or os.environ.get(
"HERMES_SESSION_PLATFORM", ""
)
source = os.environ.get("HERMES_SESSION_SOURCE", "")
for identity in (platform, source):
identity = str(identity or "").strip().lower()
if identity and identity not in _NON_MESSAGING_SESSION_SURFACES:
return True
return False
# The gateway package is unreachable, so there is no messaging channel
# to be on. Reporting a local surface keeps verify-on-stop enabled.
return False
def verify_on_stop_enabled(config: dict[str, Any] | None = None) -> bool:
@@ -149,9 +109,9 @@ def verify_on_stop_enabled(config: dict[str, Any] | None = None) -> bool:
return env.strip().lower() not in {"0", "false", "no", "off"}
if config is None:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
config = load_config()
config = load_config_readonly()
except Exception:
config = {}
agent_cfg = (config or {}).get("agent") if isinstance(config, dict) else None
+2 -2
View File
@@ -84,9 +84,9 @@ def get_active_provider() -> Optional[VideoGenProvider]:
"""
configured: Optional[str] = None
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
section = cfg.get("video_gen") if isinstance(cfg, dict) else None
if isinstance(section, dict):
raw = section.get("provider")
+2 -2
View File
@@ -98,9 +98,9 @@ def get_provider(name: str) -> Optional[WebSearchProvider]:
def _read_config_key(*path: str) -> Optional[str]:
"""Resolve a dotted config key from ``config.yaml``. Returns None on miss."""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
cur = cfg
for segment in path:
if not isinstance(cur, dict):
@@ -66,6 +66,10 @@ windows-sys = { version = "0.59", features = [
"Win32_UI_WindowsAndMessaging",
] }
# Signal-0 liveness probe for the update-lock marker owner (update.rs).
[target.'cfg(unix)'.dependencies]
libc = "0.2"
[profile.release]
# A 5-10MB signed installer is the goal. LTO + size-opt + single codegen unit.
panic = "abort"
+362 -14
View File
@@ -107,16 +107,110 @@ pub async fn start_update(app: AppHandle) -> Result<(), String> {
/// future desktop launches. The marker payload is `{pid}\n{started_at_unix}`
/// so the desktop's launch gate can detect a stale marker (dead PID / past a
/// hard ceiling) and self-heal rather than wait forever.
///
/// The marker is also the cross-process update lock: `hermes update` claims
/// the same file (see `hermes_cli/update_lock.py`) so a dashboard-spawned
/// update and this updater can't mutate one checkout at the same time.
/// `acquire` therefore REFUSES when a live foreign owner holds it rather than
/// overwriting — the pre-fix clobber is what let a dashboard `hermes update`
/// keep running while install-mode bootstrap rewrote the tree underneath it.
struct UpdateMarkerGuard {
path: PathBuf,
/// False when a live foreign updater already owns the marker: we hold no
/// claim, so `Drop` must not delete their marker.
owned: bool,
}
/// Never treat a marker older than this as a live update. Mirrors
/// UPDATE_MARKER_MAX_AGE_MS in apps/desktop/electron/update-marker.ts and
/// UPDATE_MARKER_MAX_AGE_SECONDS in hermes_cli/update_lock.py — all three read
/// this one file, so a shorter ceiling in any of them would steal a lock the
/// others still consider live.
const UPDATE_MARKER_MAX_AGE_SECS: u64 = 20 * 60;
/// The pid + age of a confirmed-live update holding the marker.
struct MarkerOwner {
pid: u32,
age_secs: u64,
}
/// Read the marker and report a live *foreign* owner, if any. `None` for every
/// "no live update" case — absent, unreadable, malformed, dead pid, past the
/// ceiling, or a marker whose pid is **this** process — matching
/// `readLiveUpdateMarker` in the Electron gate. Never panics.
///
/// Self-PID is treated as non-ownership on purpose (#74761): since #50238 the
/// desktop pre-writes this marker with the spawned updater's pid before the
/// updater reaches `acquire`. Without the exclusion, `acquire` sees a live
/// owner that is itself and aborts ("Another Hermes update is already
/// running"), then the desktop relaunches and retries forever. A foreign live
/// pid (e.g. a dashboard-spawned `hermes update`) still blocks.
fn live_marker_owner(path: &Path) -> Option<MarkerOwner> {
let raw = std::fs::read_to_string(path).ok()?;
let mut lines = raw.lines();
let pid: u32 = lines.next()?.trim().parse().ok()?;
let started_at: u64 = lines.next().unwrap_or("").trim().parse().unwrap_or(0);
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0);
let age_secs = now.saturating_sub(started_at);
if age_secs > UPDATE_MARKER_MAX_AGE_SECS || !pid_is_alive(pid) {
return None;
}
// Desktop `writeUpdateMarker(hermesHome, child.pid)` races ahead of us;
// adopt that pre-claim rather than refusing our own marker.
if pid == std::process::id() {
return None;
}
Some(MarkerOwner { pid, age_secs })
}
/// True when a process with `pid` currently exists.
#[cfg(windows)]
fn pid_is_alive(pid: u32) -> bool {
use windows_sys::Win32::Foundation::{CloseHandle, STILL_ACTIVE};
use windows_sys::Win32::System::Threading::{
GetExitCodeProcess, OpenProcess, PROCESS_QUERY_LIMITED_INFORMATION,
};
unsafe {
let handle = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, 0, pid);
if handle.is_null() {
// Either the pid is gone or we lack rights to open it. A pid we
// can't inspect is treated as dead so an unopenable straggler
// can't wedge every future update.
return false;
}
let mut code: u32 = 0;
let ok = GetExitCodeProcess(handle, &mut code);
CloseHandle(handle);
ok != 0 && code == STILL_ACTIVE as u32
}
}
#[cfg(not(windows))]
fn pid_is_alive(pid: u32) -> bool {
// signal 0 delivers nothing; it only probes existence/permission.
// ESRCH => dead. EPERM => alive but owned by another user.
let rc = unsafe { libc::kill(pid as libc::pid_t, 0) };
if rc == 0 {
return true;
}
std::io::Error::last_os_error().raw_os_error() == Some(libc::EPERM)
}
impl UpdateMarkerGuard {
/// Write the marker. Best-effort: a write failure must NOT abort the
/// update (the gate degrades to "no marker => proceed", i.e. exactly the
/// pre-fix behavior), so we log and carry on with a guard that still
/// attempts cleanup of whatever may exist at the path.
fn acquire(path: PathBuf) -> Self {
/// Claim the marker, or report the live updater that already owns it.
///
/// Writing is best-effort: a write failure must NOT abort the update (the
/// gate degrades to "no marker => proceed", i.e. exactly the pre-marker
/// behavior), so we log and carry on with a guard that still attempts
/// cleanup of whatever may exist at the path.
fn acquire(path: PathBuf) -> Result<Self, MarkerOwner> {
if let Some(owner) = live_marker_owner(&path) {
return Err(owner);
}
let pid = std::process::id();
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
@@ -128,17 +222,32 @@ impl UpdateMarkerGuard {
if let Err(err) = std::fs::write(&path, format!("{pid}\n{started_at}")) {
tracing::warn!(?path, %err, "could not write update-in-progress marker");
}
Self { path }
Ok(Self { path, owned: true })
}
/// Release the marker as soon as every mutating stage has completed.
///
/// The updater still owns a Tauri/Cocoa event loop while it relaunches the
/// desktop, and that loop can outlive `app.exit(0)`. Relying on `Drop`
/// alone therefore leaves a *successful* update looking active — a live
/// pid holding a fresh marker — which blocks desktop startup and every
/// other updater for the full age ceiling. Idempotent: `Drop` still runs
/// and tolerates an already-removed marker.
fn complete(&self) {
if !self.owned {
return;
}
if let Err(err) = std::fs::remove_file(&self.path) {
if err.kind() != std::io::ErrorKind::NotFound {
tracing::warn!(path = ?self.path, %err, "could not remove completed update marker");
}
}
}
}
impl Drop for UpdateMarkerGuard {
fn drop(&mut self) {
if let Err(err) = std::fs::remove_file(&self.path) {
if err.kind() != std::io::ErrorKind::NotFound {
tracing::warn!(path = ?self.path, %err, "could not remove update-in-progress marker");
}
}
self.complete();
}
}
@@ -152,7 +261,39 @@ async fn run_update(app: AppHandle) -> Result<()> {
// it, that backend re-locks the venv shim, our `force_kill_other_hermes`
// straggler-cleanup kills it, and the relaunch/kill cycle loops. The guard
// removes the marker on every exit path (incl. early returns / panics).
let _update_marker = UpdateMarkerGuard::acquire(crate::paths::update_in_progress_marker());
//
// The same marker is the cross-process update lock (hermes_cli/
// update_lock.py claims it too), so a live foreign owner means another
// updater — most often a dashboard-spawned `hermes update` — is already
// mutating this checkout. Refuse instead of running a second one over it.
let _update_marker = match UpdateMarkerGuard::acquire(
crate::paths::update_in_progress_marker(),
) {
Ok(guard) => guard,
Err(owner) => {
let mins = owner.age_secs / 60;
let secs = owner.age_secs % 60;
let elapsed = if mins > 0 {
format!("{mins}m {secs}s")
} else {
format!("{secs}s")
};
let msg = format!(
"Another Hermes update is already running (PID {}, started {} ago). \
Wait for it to finish, or close the window or dashboard tab that \
started it, then try again.",
owner.pid, elapsed
);
emit(
&app,
BootstrapEvent::Failed {
stage: None,
error: msg.clone(),
},
);
return Err(anyhow!(msg));
}
};
let update_branch = update_branch_from_args(std::env::args().skip(1))
.or_else(|| option_env_string("BUILD_PIN_BRANCH"))
@@ -453,6 +594,12 @@ async fn run_update(app: AppHandle) -> Result<()> {
marker: None,
},
);
// Every install-tree mutation is finished. Release the lock BEFORE the
// relaunch: this process can stay wedged in its native event loop even
// after a successful app.exit(), and a live pid on a fresh marker would
// make a completed update look active — blocking desktop startup and
// every other updater until the age ceiling expires.
_update_marker.complete();
if let Some(target_app) = launch_target {
if let Err(err) = launch_macos_app_and_exit(&app, &target_app).await {
@@ -477,9 +624,26 @@ async fn run_update(app: AppHandle) -> Result<()> {
);
}
// The launch helpers normally request exit themselves, but their failure
// paths must still close a successful updater. A native event loop can
// ignore that graceful request, so arm a process-exit fallback now that
// all update state and the marker have been settled.
exit_after_success(&app);
Ok(())
}
/// Ask the app to exit, with a hard `process::exit` fallback for a native
/// event loop that ignores the graceful request. Without it a finished updater
/// can linger as a live pid forever.
fn exit_after_success(app: &AppHandle) {
std::thread::spawn(|| {
std::thread::sleep(std::time::Duration::from_secs(3));
tracing::warn!("graceful updater exit timed out; forcing process exit");
std::process::exit(0);
});
app.exit(0);
}
/// Poll until the venv shim AND packaged desktop app bundle are no longer locked
/// (Windows) or a bounded timeout elapses. On non-Windows this is a short fixed
/// grace since file locking isn't the failure mode there.
@@ -744,6 +908,17 @@ fn update_child_env(install_root: &Path) -> Vec<(String, OsString)> {
// a frozen stage, and users cancel a healthy update. Force line-by-line
// output instead.
envs.push(("PYTHONUNBUFFERED".to_string(), OsString::from("1")));
// We hold the update-in-progress marker for this whole run, and the
// `hermes update` child claims that SAME lock (hermes_cli/update_lock.py).
// Name our pid so the child recognizes the live holder as its own
// orchestrator and runs under our claim — without this every GUI update
// refuses its parent's marker with exit 2 ("Hermes is still running")
// and no number of retries can ever succeed. Keep the variable name in
// sync with HANDOFF_PID_ENV in hermes_cli/update_lock.py.
envs.push((
"HERMES_UPDATE_HANDOFF_PID".to_string(),
OsString::from(std::process::id().to_string()),
));
if let Some(path) = path_with_prepended_entries(&[
hermes_home.join("node").join("bin"),
venv_bin_dir(install_root),
@@ -1067,6 +1242,17 @@ mod tests {
);
}
#[test]
fn update_child_env_names_our_pid_for_the_lock_handoff() {
let envs = update_child_env(Path::new("/x/hermes-agent"));
assert!(
envs.iter().any(|(k, v)| k == "HERMES_UPDATE_HANDOFF_PID"
&& v.to_str() == Some(std::process::id().to_string().as_str())),
"the hermes update child claims the same marker we hold; without our pid \
it refuses its own parent's lock and every GUI update dead-ends on exit 2"
);
}
#[test]
fn lock_probe_paths_include_desktop_app_payload() {
let root = Path::new("/x/hermes-agent");
@@ -1102,7 +1288,8 @@ mod tests {
let marker = dir.join(".hermes-update-in-progress");
{
let _g = UpdateMarkerGuard::acquire(marker.clone());
let _g = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
assert!(marker.exists(), "marker must exist while the guard is held");
let body = std::fs::read_to_string(&marker).unwrap();
let pid_line = body.lines().next().unwrap();
@@ -1127,7 +1314,8 @@ mod tests {
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
let guard = UpdateMarkerGuard::acquire(marker.clone());
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
// Simulate an external cleanup (e.g. the desktop pruned a marker it
// judged stale) before our guard drops — Drop must not panic.
std::fs::remove_file(&marker).unwrap();
@@ -1137,6 +1325,166 @@ mod tests {
let _ = std::fs::remove_dir_all(&dir);
}
/// Spawn a short-lived sibling process whose pid stands in for a foreign
/// updater. Same-process double-acquire no longer models contention: since
/// #74761 `live_marker_owner` treats our own pid as adoptable (desktop
/// pre-writes it), so a second acquire in *this* process would succeed.
fn spawn_foreign_holder() -> std::process::Child {
#[cfg(windows)]
{
std::process::Command::new("timeout")
.args(["/t", "30", "/nobreak"])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.spawn()
.expect("spawn foreign marker holder")
}
#[cfg(not(windows))]
{
std::process::Command::new("sleep")
.arg("30")
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.spawn()
.expect("spawn foreign marker holder")
}
}
#[test]
fn acquire_refuses_while_a_live_updater_owns_the_marker() {
let dir = unique_tmp_dir("marker-contended");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
// A live *foreign* updater holds it. We must NOT clobber the marker and
// run concurrently over the same checkout — that race is what let a
// dashboard `hermes update` and install-mode bootstrap mutate one tree
// at once. Own-pid markers are adoptable (#74761), so the foreign pid
// must be a real sibling process.
let mut foreign = spawn_foreign_holder();
let foreign_pid = foreign.id();
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0);
std::fs::write(&marker, format!("{foreign_pid}\n{started_at}")).unwrap();
let owner = UpdateMarkerGuard::acquire(marker.clone())
.err()
.expect("acquire must be refused while a foreign updater is live");
assert_eq!(owner.pid, foreign_pid);
// The refused guard must not delete the live owner's marker.
assert!(marker.exists(), "refused acquire must leave the marker intact");
let _ = foreign.kill();
let _ = foreign.wait();
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn acquire_adopts_a_marker_prewritten_with_our_own_pid() {
// #74761: desktop writeUpdateMarker(hermesHome, child.pid) races ahead
// of UpdateMarkerGuard::acquire. The marker names US; refusing it made
// every in-app desktop update loop forever. Adopt and rewrite.
let dir = unique_tmp_dir("marker-own-pid");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0)
.saturating_sub(2);
std::fs::write(&marker, format!("{}\n{started_at}", std::process::id())).unwrap();
let guard = UpdateMarkerGuard::acquire(marker.clone()).unwrap_or_else(|owner| {
panic!(
"own-pid pre-write must be adoptable, got foreign owner pid={}",
owner.pid
)
});
assert!(marker.exists(), "adopted guard must own the marker");
let body = std::fs::read_to_string(&marker).unwrap();
assert_eq!(
body.lines().next().unwrap().trim().parse::<u32>().unwrap(),
std::process::id(),
"acquire rewrites the marker with our pid + fresh started_at"
);
drop(guard);
assert!(
!marker.exists(),
"Drop must still clear the marker we adopted"
);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn acquire_reclaims_a_marker_owned_by_a_dead_pid() {
let dir = unique_tmp_dir("marker-dead-pid");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
// pid 1 exists everywhere, so fabricate a dead one: a very large pid
// that no live process owns. A crashed updater must never wedge every
// future update.
let started_at = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0);
std::fs::write(&marker, format!("4294967294\n{started_at}")).unwrap();
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("a dead owner must not block acquisition"));
let body = std::fs::read_to_string(&marker).unwrap();
assert_eq!(
body.lines().next().unwrap().trim().parse::<u32>().unwrap(),
std::process::id(),
"reclaiming rewrites the marker with our pid"
);
drop(guard);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn acquire_reclaims_a_marker_past_the_age_ceiling() {
let dir = unique_tmp_dir("marker-stale-age");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
// Our own (live) pid, but started well past the ceiling: a wedged
// updater must not hold the lock forever.
let long_ago = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs())
.unwrap_or(0)
.saturating_sub(UPDATE_MARKER_MAX_AGE_SECS + 60);
std::fs::write(&marker, format!("{}\n{long_ago}", std::process::id())).unwrap();
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("a marker past the ceiling must be reclaimable"));
drop(guard);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn completed_update_releases_marker_before_guard_drop() {
let dir = unique_tmp_dir("marker-complete");
std::fs::create_dir_all(&dir).unwrap();
let marker = dir.join(".hermes-update-in-progress");
let guard = UpdateMarkerGuard::acquire(marker.clone())
.unwrap_or_else(|_| panic!("no live owner => acquire must succeed"));
guard.complete();
assert!(
!marker.exists(),
"a successful update must unblock desktop startup before relaunch/exit"
);
drop(guard);
assert!(!marker.exists(), "Drop stays idempotent after completion");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn parses_update_branch_from_space_or_equals_args() {
assert_eq!(
+10 -1
View File
@@ -91,10 +91,11 @@ for call-site shadow or border inventions.
| Token | Use |
| --- | --- |
| `--ui-stroke-primary…quaternary` | hairlines, in descending strength |
| `--ui-stroke-tertiary` | the default in-panel divider / list hairline |
| `--ui-stroke-tertiary` | the default in-panel divider / list hairline — and every bordered surface in the transcript |
| `--stroke-nous` | the overlay hairline (pairs with `shadow-nous`) |
| `--ui-text-primary / -secondary / -tertiary` | text hierarchy |
| `--ui-bg-quaternary` | soft control fill (secondary button) |
| `--ui-widget-surface-background` | fill for inline chat widgets (`WIDGET_SHELL_CLASS`) |
| `--chrome-action-hover` | hover fill for quiet controls |
| `--theme-primary`, `--ui-accent` | brand/accent |
@@ -196,6 +197,14 @@ Notes:
existing components under `src/components/assistant-ui` and
`src/app/chat/composer`; do not fork a second markdown, message, tool-call, or
approval renderer for one feature.
- **Inline widgets** — a tool result that renders as a panel the user reads or
acts on (clarify, artifact card) wears `WIDGET_SHELL_CLASS`
(`src/components/chat/widget-shell.ts`): shared radius, the
`--ui-widget-surface-background` fill, no border. Its actions sit *outside*
the panel, below it. Don't give one widget its own radius or fill.
- Bordered surfaces in the transcript (tables, fences, callouts, attachments)
use `--ui-stroke-tertiary`. Not `border-border` — that's the app-wide
default and reads too hot against the thread.
- A tool result may expose an inline action that opens a preview. It must not
open the rail automatically.
- Install, onboarding, connecting, boot failure, and reauthentication are
+7 -1
View File
@@ -9,7 +9,13 @@ export function linkTitleWindowOptions(partitionSession) {
width: 1280,
height: 800,
webPreferences: {
backgroundThrottling: false,
// Deliberately throttled: this hidden window loads arbitrary user-linked
// pages, and an unthrottled heavy page burns full CPU for the window's
// whole lifetime. Title resolution rides load events
// (page-title-updated / did-finish-load) plus main-process timers, none
// of which the renderer clamp touches — hidden-page throttling only
// slows the page's own timer-driven JS, and the grace window already
// absorbs that.
contextIsolation: true,
javascript: true,
nodeIntegration: false,
+92 -18
View File
@@ -183,6 +183,7 @@ import {
redactSecrets,
SshConnection
} from './ssh-connection'
import { createStreamThrottle } from './stream-throttle'
import { nativeOverlayWidth as computeNativeOverlayWidth, macTitleBarOverlayHeight } from './titlebar-overlay-width'
import { resolveBehindCount, shouldCountCommits } from './update-count'
import { waitForUpdateClearance } from './update-gate'
@@ -421,18 +422,24 @@ if (IS_WINDOWS) {
ipcMain.handle('hermes:get-remote-display-reason', () => REMOTE_DISPLAY_REASON)
// Keep the renderer running at full speed while the window is in the background
// or occluded. The chat transcript streams to screen through a bounded timer
// flush; Chromium clamps timers for backgrounded/occluded renderers, so without
// these the live answer stalls
// whenever the window loses focus (switching to your editor mid-turn, detached
// devtools, another window covering it) and only paints on refocus or refresh.
// `backgroundThrottling: false` on the BrowserWindow covers the blurred case;
// these process-level switches additionally stop Chromium from backgrounding or
// occlusion-throttling the renderer. Must run before app `ready`.
// Keep the renderer's PROCESS priority normal while its windows are hidden —
// a deprioritized renderer streams a live answer visibly slower once the
// window is minimized. This switch only affects scheduling priority; it does
// not exempt timers from throttling and costs nothing at idle.
//
// The timer/rAF throttling story is deliberately NOT handled here anymore.
// The old process-wide `disable-background-timer-throttling` /
// `disable-backgrounding-occluded-windows` switches (plus a static
// `backgroundThrottling: false` on every chat window) pinned every renderer's
// `document.visibilityState` to 'visible' forever — which silently turned all
// the renderer's visibility-gated backstop polls and clock ticks into
// always-on timers. A completely idle, minimized Hermes burned ~20% CPU
// around the clock. Throttling is now a runtime dial scoped to streaming:
// see createStreamThrottle() — chat windows are unthrottled while any turn is
// in flight (so a live answer keeps painting while blurred, occluded, or
// minimized, exactly as before) and return to Chromium's default throttling
// once the work settles.
app.commandLine.appendSwitch('disable-renderer-backgrounding')
app.commandLine.appendSwitch('disable-backgrounding-occluded-windows')
app.commandLine.appendSwitch('disable-background-timer-throttling')
const SOURCE_REPO_ROOT = path.resolve(APP_ROOT, '../..')
@@ -5155,6 +5162,20 @@ function sendClosePreviewRequested() {
webContents.send('hermes:close-preview-requested')
}
function sendOpenFolderRequested() {
if (!mainWindow || mainWindow.isDestroyed()) {
return
}
const webContents = mainWindow.webContents
if (!webContents || webContents.isDestroyed()) {
return
}
webContents.send('hermes:open-folder-requested')
}
// Tell the renderer the machine just woke. Sleep silently drops the
// renderer's WebSocket to the local backend; the renderer reconnects on this
// signal so the chat composer doesn't stay stuck on "Starting Hermes...".
@@ -5174,6 +5195,31 @@ function sendPowerResume() {
let powerResumeRegistered = false
// Mirror of powerMonitor's AC/battery state, broadcast to every window so
// renderer backstop polls can slow down on battery (see store/power.ts).
// `null` until the first powerMonitor read after app ready.
let onBatteryPower: boolean | null = null
// Renderer-side battery gating seeds from this and stays current via the
// 'hermes:power-battery' push below.
ipcMain.handle('hermes:power-battery:get', () => onBatteryPower === true)
function broadcastBatteryState(next: boolean) {
if (onBatteryPower === next) {
return
}
onBatteryPower = next
for (const win of BrowserWindow.getAllWindows()) {
const { webContents } = win
if (webContents && !webContents.isDestroyed()) {
webContents.send('hermes:power-battery', next)
}
}
}
function registerPowerResumeListeners() {
if (powerResumeRegistered) {
return
@@ -5186,6 +5232,9 @@ function registerPowerResumeListeners() {
// full suspend. Either can drop an idle socket.
powerMonitor.on('resume', sendPowerResume)
powerMonitor.on('unlock-screen', sendPowerResume)
powerMonitor.on('on-battery', () => broadcastBatteryState(true))
powerMonitor.on('on-ac', () => broadcastBatteryState(false))
onBatteryPower = powerMonitor.isOnBatteryPower()
} catch {
// powerMonitor is unavailable before app 'ready' on some platforms; the
// caller registers after 'ready', so this should not normally throw.
@@ -5272,6 +5321,10 @@ function buildApplicationMenu() {
// a menu accelerator would fight the rebind panel and (on macOS) be
// swallowed before the renderer sees it. Here purely for discoverability.
{ click: () => createInstanceWindow(), label: 'New Window' },
// Same no-accelerator rationale: ⌘O is the rebindable renderer keybind
// (workspace.openFolder). Clicking runs the same open-folder-as-project
// flow through the renderer.
{ click: () => sendOpenFolderRequested(), label: 'Open Folder…' },
{ type: 'separator' },
IS_MAC
? {
@@ -5624,8 +5677,12 @@ function installContextMenu(window) {
}
}
// Bare right-click on non-editable, non-selected, non-media content (a pane
// body, the sidebar, chrome): the renderer's own context menus own those
// surfaces, and anywhere without one shows nothing — not a lone, useless
// "Select All" from the native fallback.
if (!template.length) {
template.push({ role: 'selectAll' })
return
}
Menu.buildFromTemplate(template).popup({ window })
@@ -8675,6 +8732,7 @@ function spawnSecondaryWindow({ sessionId, watch }: { sessionId?: string; watch?
win.on('enter-full-screen', () => sendWindowStateChanged(true))
win.on('leave-full-screen', () => sendWindowStateChanged(false))
streamThrottle.register(win)
wireCommonWindowHandlers(win, zoomWiringForWindowKind('chat'))
loadWindowUrl(
@@ -8717,7 +8775,7 @@ function nextInstanceBounds() {
}
// Open a new full-chrome instance window. Mirrors createWindow()'s window
// options (shared chatWindowWebPreferences keeps backgroundThrottling:false so a
// options (shared chatWindowWebPreferences + streamThrottle registration so a
// streamed answer never stalls in the background) but is a peer, not the
// primary: it never overwrites the mainWindow global, doesn't start the backend
// (the renderer's getConnection() joins the already-running one), and loads the
@@ -8758,6 +8816,7 @@ function createInstanceWindow() {
win.on('enter-full-screen', () => sendWindowStateChanged(true, win))
win.on('leave-full-screen', () => sendWindowStateChanged(false, win))
streamThrottle.register(win)
wireCommonWindowHandlers(win, zoomWiringForWindowKind('chat'))
win.on('closed', () => {
@@ -9140,10 +9199,11 @@ function createWindow() {
// material before the renderer paints the app theme. See createSessionWindow.
show: false,
backgroundColor: getWindowBackgroundColor(),
// Shared with the secondary session windows (chatWindowWebPreferences) so
// both keep `backgroundThrottling: false` — the chat transcript uses a
// bounded timer flush that Chromium clamps for blurred windows, stalling
// the live answer until refocus. See session-windows.ts.
// Shared with the secondary session windows (chatWindowWebPreferences);
// stream-aware throttling is applied per-window via streamThrottle so a
// live answer keeps painting while the window is blurred or minimized,
// without pinning visibilityState to 'visible' at idle. See
// session-windows.ts and stream-throttle.ts.
webPreferences: chatWindowWebPreferences(PRELOAD_PATH)
})
@@ -9236,6 +9296,7 @@ function createWindow() {
}
})
streamThrottle.register(mainWindow)
wireCommonWindowHandlers(mainWindow, zoomWiringForWindowKind('chat'))
mainWindow.webContents.on('render-process-gone', (_event, details) => {
@@ -10405,14 +10466,27 @@ ipcMain.handle('hermes:stopPreviewFileWatch', (_event, id) => stopPreviewFileWat
// merged picture. Keyed by webContents id so a closed window stops counting.
const activeWorkByWebContents = new Map<number, ActiveWork>()
// The same merged picture drives background throttling: chat windows run
// unthrottled while any turn is in flight (streaming must paint while hidden)
// and fall back to Chromium's default throttling at idle. See stream-throttle.ts.
const streamThrottle = createStreamThrottle()
function updateStreamThrottleFromActiveWork() {
streamThrottle.update(mergeActiveWork(activeWorkByWebContents.values()).count > 0)
}
ipcMain.on('hermes:active-work', (event, payload) => {
const id = event.sender.id
if (!activeWorkByWebContents.has(id)) {
event.sender.once('destroyed', () => activeWorkByWebContents.delete(id))
event.sender.once('destroyed', () => {
activeWorkByWebContents.delete(id)
updateStreamThrottleFromActiveWork()
})
}
activeWorkByWebContents.set(id, normalizeActiveWork(payload))
updateStreamThrottleFromActiveWork()
})
ipcMain.on('hermes:titlebar-theme', (_event, payload) => {
+14
View File
@@ -212,6 +212,12 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
return () => ipcRenderer.removeListener('hermes:close-preview-requested', listener)
},
onOpenFolderRequested: callback => {
const listener = () => callback()
ipcRenderer.on('hermes:open-folder-requested', listener)
return () => ipcRenderer.removeListener('hermes:open-folder-requested', listener)
},
onOpenUpdatesRequested: callback => {
const listener = () => callback()
ipcRenderer.on('hermes:open-updates', listener)
@@ -269,6 +275,14 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
return () => ipcRenderer.removeListener('hermes:power-resume', listener)
},
// AC ↔ battery transitions; renderers slow their backstop polls on battery.
getOnBattery: () => ipcRenderer.invoke('hermes:power-battery:get'),
onBatteryChanged: callback => {
const listener = (_event, onBattery) => callback(Boolean(onBattery))
ipcRenderer.on('hermes:power-battery', listener)
return () => ipcRenderer.removeListener('hermes:power-battery', listener)
},
onBootProgress: callback => {
const listener = (_event, payload) => callback(payload)
ipcRenderer.on('hermes:boot-progress', listener)
@@ -191,13 +191,16 @@ test('registry trims the session id before keying', () => {
assert.equal(registry.has('s1'), true)
})
test('chatWindowWebPreferences disables background throttling so streaming paints while blurred', () => {
// Regression: secondary session windows used to omit this flag, so a streamed
// answer stalled until the window regained focus (Chromium clamps the
// transcript flush timer for backgrounded windows).
test('chatWindowWebPreferences leaves background throttling to the runtime stream dial', () => {
// Regression (both directions): a static `backgroundThrottling: false` here
// pinned document.visibilityState to 'visible' forever, turning every
// visibility-gated poll into an always-on timer (~20% CPU at idle,
// minimized). Streaming's "paint while blurred" need is served by
// stream-throttle.ts flipping setBackgroundThrottling at turn boundaries —
// so the static flag must stay absent.
const prefs = chatWindowWebPreferences('/tmp/preload.cjs')
assert.equal(prefs.backgroundThrottling, false)
assert.equal('backgroundThrottling' in prefs, false)
})
test('chatWindowWebPreferences passes the preload path through and keeps the hardened defaults', () => {
+13 -8
View File
@@ -13,14 +13,20 @@ const SESSION_WINDOW_MIN_HEIGHT = 620
// Shared webPreferences for every window that renders the chat transcript — the
// primary window AND the secondary session windows. Keeping it in one place is
// the whole point: the two BrowserWindow definitions in main.ts used to be
// hand-copied, and the secondary windows silently lost `backgroundThrottling:
// false`, so a streamed answer stalled until the window regained focus.
// hand-copied, and the secondary windows silently drifted apart (a streamed
// answer stalled until the window regained focus because one of them lost the
// throttling opt-out).
//
// `backgroundThrottling: false` is load-bearing: the transcript streams to the
// screen through a bounded timer flush, which Chromium clamps for blurred/
// occluded windows. A streaming chat app must keep painting in the
// background, so every chat window opts out. The preload path is injected
// because it depends on the Electron entry's __dirname.
// Background throttling is deliberately NOT set here. It is managed at runtime
// by main.ts (`setBackgroundThrottling` driven by the merged `hermes:active-work`
// reports): while any turn is in flight every chat window is unthrottled so the
// transcript's bounded timer flush keeps painting while blurred, occluded, or
// minimized — and once all turns finish, Chromium's default throttling returns
// so an idle hidden window costs ~nothing. A static `backgroundThrottling:
// false` here would pin `document.visibilityState` to 'visible' forever,
// turning every visibility-gated poll in the renderer into an always-on timer
// (the "Hermes idles at 20% CPU while minimized" bug). The preload path is
// injected because it depends on the Electron entry's __dirname.
//
// `autoplayPolicy: 'no-user-gesture-required'` is load-bearing for voice:
// Chromium's default autoplay policy suspends audio (HTMLAudioElement.play()
@@ -39,7 +45,6 @@ function chatWindowWebPreferences(preloadPath: string) {
sandbox: true,
nodeIntegration: false,
devTools: true,
backgroundThrottling: false,
autoplayPolicy: 'no-user-gesture-required' as const
}
}
@@ -0,0 +1,152 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { createStreamThrottle, type ThrottleWindowLike } from './stream-throttle'
function makeTimers() {
const pending = new Map<number, () => void>()
let nextId = 1
return {
clearTimeout: (handle: unknown) => {
pending.delete(handle as number)
},
fire() {
const jobs = [...pending.values()]
pending.clear()
for (const job of jobs) {
job()
}
},
get pendingCount() {
return pending.size
},
setTimeout: (fn: () => void, _ms: number) => {
const id = nextId++
pending.set(id, fn)
return id
}
}
}
function makeWindow() {
const calls: boolean[] = []
const listeners = new Map<string, () => void>()
let destroyed = false
const win = {
calls,
close() {
destroyed = true
listeners.get('closed')?.()
},
isDestroyed: () => destroyed,
on(event: string, fn: () => void) {
listeners.set(event, fn)
},
webContents: {
isDestroyed: () => destroyed,
setBackgroundThrottling(allowed: boolean) {
calls.push(allowed)
}
}
}
return win
}
test('registering a window applies the current throttle state immediately', () => {
const timers = makeTimers()
const throttle = createStreamThrottle(timers)
const idle = makeWindow()
throttle.register(idle)
// Idle default: throttling allowed.
assert.deepEqual(idle.calls, [true])
throttle.update(true)
const late = makeWindow()
throttle.register(late)
// A window created mid-stream starts unthrottled.
assert.deepEqual(late.calls, [false])
})
test('a turn in flight unthrottles every chat window; settling re-throttles after the trailing delay', () => {
const timers = makeTimers()
const throttle = createStreamThrottle(timers)
const win = makeWindow()
throttle.register(win)
throttle.update(true)
assert.deepEqual(win.calls, [true, false])
assert.equal(throttle.isUnthrottled(), true)
// Turn ends: not re-throttled synchronously — the tail flush needs full
// cadence — only after the trailing timer fires.
throttle.update(false)
assert.deepEqual(win.calls, [true, false])
assert.equal(throttle.isUnthrottled(), true)
timers.fire()
assert.deepEqual(win.calls, [true, false, true])
assert.equal(throttle.isUnthrottled(), false)
})
test('a new turn during the trailing window cancels the pending re-throttle', () => {
const timers = makeTimers()
const throttle = createStreamThrottle(timers)
const win = makeWindow()
throttle.register(win)
throttle.update(true)
throttle.update(false)
assert.equal(timers.pendingCount, 1)
// Busy again before the delay elapses: stay unthrottled, timer cancelled.
throttle.update(true)
assert.equal(timers.pendingCount, 0)
assert.equal(throttle.isUnthrottled(), true)
// The cancelled timer firing late must be a no-op.
timers.fire()
assert.equal(throttle.isUnthrottled(), true)
})
test('repeated busy reports do not re-apply or stack timers', () => {
const timers = makeTimers()
const throttle = createStreamThrottle(timers)
const win = makeWindow()
throttle.register(win)
throttle.update(true)
throttle.update(true)
throttle.update(true)
assert.deepEqual(win.calls, [true, false])
throttle.update(false)
throttle.update(false)
assert.equal(timers.pendingCount, 1)
})
test('closed and destroyed windows drop out without throwing', () => {
const timers = makeTimers()
const throttle = createStreamThrottle(timers)
const closedWin = makeWindow()
throttle.register(closedWin)
closedWin.close()
const gone: ThrottleWindowLike & { on?: never } = {
isDestroyed: () => true,
webContents: null
}
throttle.register(gone)
throttle.update(true)
// Only the registration-time call landed; nothing after close.
assert.deepEqual(closedWin.calls, [true])
})
+119
View File
@@ -0,0 +1,119 @@
// Stream-aware background throttling for chat windows.
//
// Chat windows must paint the live transcript while blurred, occluded, or
// minimized — but a static `backgroundThrottling: false` in webPreferences
// costs far more than that feature needs: it pins the renderer's
// `document.visibilityState` to 'visible' for the life of the window, which
// turns every visibility-gated poll and clock tick in the renderer into an
// always-on timer. An idle, hidden Hermes burned ~20% CPU forever.
//
// So throttling is a runtime dial instead: the renderers already report
// "which chats are mid-turn" for the quit guard (`hermes:active-work`), and
// this controller rides the merged edge of those reports. Any turn in flight →
// every registered chat window gets `setBackgroundThrottling(false)`, exactly
// the streaming behavior the static flag used to provide. All turns done →
// after a short trailing delay (so tail flushes land at full cadence) Chromium's
// default throttling returns and hidden windows go quiet.
//
// Pure and Electron-free (timers + the WebContents surface are injected) so it
// can be unit-tested, mirroring session-windows.ts.
/** How long after the last turn ends before throttling is restored. Covers the
* stream queue's final coalesced flush and the settle writes that trail a
* turn's completion, so re-throttling never strands a visible delta. */
const RETHROTTLE_DELAY_MS = 5_000
export interface ThrottleWindowLike {
isDestroyed(): boolean
webContents?: {
isDestroyed(): boolean
setBackgroundThrottling(allowed: boolean): void
} | null
}
interface TimersLike {
clearTimeout(handle: unknown): void
setTimeout(fn: () => void, ms: number): unknown
}
export interface StreamThrottle {
/** True while windows are currently unthrottled (streaming or trailing). */
isUnthrottled(): boolean
/** Track a chat window; applies the current state immediately and stops
* tracking on close. */
register(win: ThrottleWindowLike & { on?: (event: string, fn: () => void) => void }): void
/** Report whether any turn is in flight across all renderers. */
update(busy: boolean): void
}
export function createStreamThrottle(
timers: TimersLike = { clearTimeout: handle => clearTimeout(handle as never), setTimeout },
delayMs: number = RETHROTTLE_DELAY_MS
): StreamThrottle {
const windows = new Set<ThrottleWindowLike>()
let unthrottled = false
let trailing: unknown = null
function apply(win: ThrottleWindowLike) {
if (win.isDestroyed()) {
windows.delete(win)
return
}
const contents = win.webContents
if (!contents || contents.isDestroyed()) {
return
}
try {
contents.setBackgroundThrottling(!unthrottled)
} catch {
// A window mid-teardown can throw; it's about to leave the set anyway.
}
}
function applyAll() {
for (const win of windows) {
apply(win)
}
}
return {
isUnthrottled: () => unthrottled,
register(win) {
windows.add(win)
win.on?.('closed', () => windows.delete(win))
apply(win)
},
update(busy) {
if (busy) {
if (trailing !== null) {
timers.clearTimeout(trailing)
trailing = null
}
if (!unthrottled) {
unthrottled = true
applyAll()
}
return
}
if (!unthrottled || trailing !== null) {
return
}
// Trailing edge: keep full cadence briefly so the final flush paints.
trailing = timers.setTimeout(() => {
trailing = null
unthrottled = false
applyAll()
}, delayMs)
}
}
}
+2
View File
@@ -101,7 +101,9 @@
"d3-force": "^3.0.0",
"dnd-core": "^14.0.1",
"dompurify": "^3.4.11",
"emojibase-data": "^16.0.3",
"fflate": "^0.8.3",
"frimousse": "^0.3.0",
"hast-util-from-html-isomorphic": "^2.0.0",
"hast-util-to-text": "^4.0.2",
"ignore": "^7.0.5",
@@ -0,0 +1,146 @@
// ⌘K open latency, measured in-page (no CDP round-trip in the number).
//
// node scripts/probe-command-palette.mjs [--port 9222] [--rounds 8]
//
// Reports, per round, the time from the keydown the app actually receives to:
// frame_ms — the dialog frame + input in the DOM and painted (what "instant"
// means: the overlay owes you a frame immediately)
// rows_ms — the row list painted (may lag frame_ms; rows are deferred)
// plus any long tasks in the window, so a slow open is attributable.
import { CDP, sleep } from './perf/lib/cdp.mjs'
const args = process.argv.slice(2)
const flag = name => {
const i = args.indexOf(`--${name}`)
return i >= 0 ? args[i + 1] : undefined
}
const port = Number(flag('port') ?? 9222)
const rounds = Number(flag('rounds') ?? 8)
const cdp = await CDP.connect({ port })
await cdp.send('Runtime.enable')
const INSTALL = `
(() => {
if (window.__CMDK__) window.__CMDK__.stop()
const state = { t0: null, frame: null, rows: 0, rowsAt: null, tasks: [], armed: false }
// Time from the keydown the APP receives — excludes CDP transport, so the
// number is what a user's finger actually experiences.
const onKey = e => {
if (state.armed && (e.metaKey || e.ctrlKey) && e.key.toLowerCase() === 'k') {
state.t0 = performance.now()
state.armed = false
}
}
window.addEventListener('keydown', onKey, true)
const obs = new MutationObserver(() => {
if (state.t0 === null) return
if (state.frame === null && document.querySelector('[cmdk-input]')) {
state.frame = performance.now() - state.t0
}
const n = document.querySelectorAll('[cmdk-item]').length
if (n > state.rows) { state.rows = n; state.rowsAt = performance.now() - state.t0 }
})
obs.observe(document.body, { childList: true, subtree: true })
const po = new PerformanceObserver(list => {
for (const e of list.getEntries()) state.tasks.push({ start: e.startTime, dur: Math.round(e.duration) })
})
try { po.observe({ entryTypes: ['longtask'] }) } catch {}
window.__CMDK__ = {
arm: () => { state.t0 = null; state.frame = null; state.rows = 0; state.rowsAt = null; state.tasks = []; state.armed = true },
read: () => ({
frame_ms: state.frame === null ? -1 : Math.round(state.frame),
rows_ms: state.rowsAt === null ? -1 : Math.round(state.rowsAt),
rows: state.rows,
longtask_ms: state.t0 === null ? 0 : state.tasks.filter(t => t.start >= state.t0).reduce((s, t) => s + t.dur, 0)
}),
stop: () => { window.removeEventListener('keydown', onKey, true); obs.disconnect(); po.disconnect() }
}
return true
})()
`
// Settle: frame painted AND rows stopped growing for two frames.
const WAIT = `
new Promise(resolve => {
let stable = 0
let last = -1
const started = performance.now()
const tick = () => {
const r = window.__CMDK__.read()
if (r.frame_ms >= 0 && r.rows === last && r.rows > 0) {
if (++stable >= 2) { resolve(r); return }
} else { stable = 0 }
last = r.rows
if (performance.now() - started > 8000) { resolve(window.__CMDK__.read()); return }
requestAnimationFrame(tick)
}
requestAnimationFrame(tick)
})
`
const key = async type =>
cdp.send('Input.dispatchKeyEvent', {
type,
key: 'k',
code: 'KeyK',
windowsVirtualKeyCode: 75,
nativeVirtualKeyCode: 75,
modifiers: 4
})
const esc = async () => {
for (const type of ['keyDown', 'keyUp']) {
await cdp.send('Input.dispatchKeyEvent', { type, key: 'Escape', code: 'Escape', windowsVirtualKeyCode: 27 })
}
await sleep(400)
}
await cdp.eval(INSTALL)
await esc()
const samples = []
for (let i = 0; i < rounds; i++) {
await sleep(250)
await cdp.eval('window.__CMDK__.arm()')
await key('rawKeyDown')
await key('keyUp')
const r = await cdp.eval(WAIT)
samples.push(r)
console.log(`round ${i}:`, r)
await esc()
}
await cdp.eval('window.__CMDK__.stop()')
const stat = k => {
const v = samples.map(s => s[k]).filter(n => n >= 0).sort((a, b) => a - b)
if (!v.length) return null
return {
min: v[0],
median: v[Math.floor(v.length / 2)],
max: v[v.length - 1]
}
}
console.log('\nkeydown → dialog frame painted (ms):', stat('frame_ms'))
console.log('keydown → rows painted (ms):', stat('rows_ms'))
console.log('long-task time in window (ms):', stat('longtask_ms'))
cdp.close()
+99 -13
View File
@@ -1,9 +1,30 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const closeFocusedSessionTab = vi.fn(() => false)
const nextSessionTileForWorkspace = vi.fn<() => null | string>(() => null)
const closeSessionTile = vi.fn()
const requestFreshSession = vi.fn()
vi.mock('@/components/pane-shell/tree/store', () => ({
closeFocusedSessionTab: () => closeFocusedSessionTab()
}))
vi.mock('@/store/session-states', () => ({
closeSessionTile: (...args: unknown[]) => closeSessionTile(...args),
nextSessionTileForWorkspace: () => nextSessionTileForWorkspace()
}))
vi.mock('@/store/profile', () => ({
requestFreshSession: () => requestFreshSession()
}))
import { $rightRailActiveTabId } from '@/store/layout'
import { $previewTabs, closeRightRail, openPreview, type PreviewTarget } from '@/store/preview'
import { $activeSessionId, $selectedStoredSessionId } from '@/store/session'
import { closeActiveTab } from './close-tab'
import { $workspaceIsPage } from '../routes'
import { closeActiveTab, closeWorkspaceTab } from './close-tab'
function fileTarget(path: string): PreviewTarget {
return {
@@ -16,19 +37,31 @@ function fileTarget(path: string): PreviewTarget {
}
}
/** Main is holding a loaded chat and nothing else is stacked with it. */
function loadedMainOnly() {
$selectedStoredSessionId.set('stored-a')
$activeSessionId.set('runtime-a')
}
beforeEach(() => {
vi.stubGlobal('document', { activeElement: null })
closeRightRail()
window.localStorage.clear()
$selectedStoredSessionId.set(null)
$activeSessionId.set(null)
$workspaceIsPage.set(false)
closeFocusedSessionTab.mockReturnValue(false)
nextSessionTileForWorkspace.mockReturnValue(null)
vi.clearAllMocks()
})
afterEach(() => {
vi.unstubAllGlobals()
closeRightRail()
window.localStorage.clear()
})
describe('closeActiveTab', () => {
beforeEach(() => {
vi.stubGlobal('document', { activeElement: null })
closeRightRail()
window.localStorage.clear()
})
afterEach(() => {
vi.unstubAllGlobals()
closeRightRail()
window.localStorage.clear()
})
it('closes the active file preview tab (⌘W happy path)', () => {
openPreview(fileTarget('/work/notes.md'), 'manual')
@@ -50,3 +83,56 @@ describe('closeActiveTab', () => {
expect($previewTabs.get()).toHaveLength(0)
})
})
/**
* The main tab's own close. The workspace pane can never leave the tree, so
* every answer here is about what FILLS it — a stacked session, or an empty
* draft. The gesture used to dead-end whenever main was the only tab.
*/
describe('closeWorkspaceTab', () => {
it('shifts the next stacked session into main', () => {
loadedMainOnly()
nextSessionTileForWorkspace.mockReturnValue('stored-b')
const load = vi.fn()
expect(closeWorkspaceTab(load)).toBe(true)
expect(closeSessionTile).toHaveBeenCalledWith('stored-b')
expect(load).toHaveBeenCalledWith('stored-b')
// Promotion refills main — it must not ALSO blank it.
expect(requestFreshSession).not.toHaveBeenCalled()
})
it('drops a lone loaded main to a fresh draft', () => {
loadedMainOnly()
expect(closeWorkspaceTab(vi.fn())).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
it('empties main even with no session loader wired', () => {
loadedMainOnly()
expect(closeWorkspaceTab()).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
it('is a no-op on a blank draft — that IS the post-close state', () => {
expect(closeWorkspaceTab(vi.fn())).toBe(false)
expect(requestFreshSession).not.toHaveBeenCalled()
})
it('is a no-op over a full-page view, which owns no chat tab', () => {
loadedMainOnly()
$workspaceIsPage.set(true)
expect(closeWorkspaceTab(vi.fn())).toBe(false)
expect(requestFreshSession).not.toHaveBeenCalled()
})
it('⌘W reaches it once the terminal, rail and zone tabs pass', () => {
loadedMainOnly()
expect(closeActiveTab(vi.fn())).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
})
+51 -26
View File
@@ -1,27 +1,68 @@
import { mainChatOccupied } from '@/app/open-session'
import { closeActiveTerminal } from '@/app/right-sidebar/terminal/terminals'
import { $workspaceIsPage } from '@/app/routes'
import { closeFocusedSessionTab } from '@/components/pane-shell/tree/store'
import { isFocusWithin } from '@/lib/keybinds/combo'
import { $previewTabs, closeActiveRightRailTab } from '@/store/preview'
import { requestFreshSession } from '@/store/profile'
import { $activeSessionId, $selectedStoredSessionId } from '@/store/session'
import { closeSessionTile, nextSessionTileForWorkspace } from '@/store/session-states'
/**
* Close the MAIN tab. The workspace pane itself can't leave the tree, so
* "closing" it means emptying it, and what fills the hole depends on what's
* stacked beside it:
*
* - session tabs stacked with it → the next one shifts INTO main (drop its
* tile, load it as the primary — the session stays alive, no busy prompt),
* - nothing stacked → main drops to a fresh "New session" draft.
*
* The second half is what makes the gesture honest when main is the ONLY tab:
* ⌘W / ⌘-click / middle-click used to be a dead key there, since the only
* available answer was "remove the pane", which this app never does.
*
* Returns false when there is nothing to close — a blank draft (already the
* post-close state) or a full-page view (skills / artifacts, which isn't a
* chat and owns no tab). ⌘W then stays a no-op; it never closes the window.
*
* `loadSessionIntoWorkspace` carries the app's route-based "load this session
* into main"; omitting it disables the promotion half.
*/
export function closeWorkspaceTab(loadSessionIntoWorkspace?: (storedSessionId: string) => void): boolean {
// Order matters — close the tile FIRST so the selection homes to the
// workspace instead of re-fronting the tile.
if (loadSessionIntoWorkspace) {
const next = nextSessionTileForWorkspace()
if (next) {
closeSessionTile(next)
loadSessionIntoWorkspace(next)
return true
}
}
if ($workspaceIsPage.get() || !mainChatOccupied($activeSessionId.get(), $selectedStoredSessionId.get())) {
return false
}
requestFreshSession()
return true
}
/**
* ⌘W — close the tab of the context you're in, by precedence:
* 1. a focused terminal → its active terminal tab,
* 2. right-rail tabs (live preview and/or file peeks),
* 3. the FOCUSED chat zone → its active tab (a session tile stacked into it).
* 4. the workspace tab itself, when session tabs are stacked with it:
* the workspace can't close, so ⌘W shifts the NEXT session tab into main
* (loads it as the primary + drops its now-redundant tile).
* 4. the workspace tab itself — see `closeWorkspaceTab`.
* Returns false when nothing closes, so ⌘W is a no-op — it never closes the
* window (a bare workspace stays put). Shared by the keyboard path (Win/Linux)
* and the macOS menu-accelerator IPC.
* window. Shared by the keyboard path (Win/Linux) and the macOS
* menu-accelerator IPC.
*
* Steps 3-4 follow the same focused zone ⌘1…⌘9 indexes, so a second chat zone
* with its own tab strip closes ITS tab instead of main's.
*
* `loadSessionIntoWorkspace` carries the app's route-based "load this session
* into main" (the two call sites have router access); omitting it disables the
* step-4 promotion (⌘W stays the pre-existing no-op on the main tab).
*/
export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: string) => void): boolean {
if (isFocusWithin('[data-terminal]')) {
@@ -43,21 +84,5 @@ export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: stri
return true
}
// The main (workspace) tab is active and can't be closed — but if session
// tabs are stacked with it, ⌘W shifts the next one into the main tab: drop
// its tile (the session stays alive, no busy-close prompt) and load it into
// main. Order matters — close the tile FIRST so the selection homes to the
// workspace instead of re-fronting the tile.
if (loadSessionIntoWorkspace) {
const next = nextSessionTileForWorkspace()
if (next) {
closeSessionTile(next)
loadSessionIntoWorkspace(next)
return true
}
}
return false
return closeWorkspaceTab(loadSessionIntoWorkspace)
}
@@ -0,0 +1,131 @@
import type { Unstable_TriggerItem } from '@assistant-ui/core'
import { act, renderHook } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { hermesDirectiveFormatter } from '@/components/assistant-ui/directive-text'
import { classify } from './hooks/use-at-completions'
import { useComposerTrigger } from './hooks/use-composer-trigger'
import { composerPlainText, RICH_INPUT_SLOT } from './rich-editor'
/** A row exactly as tui_gateway's complete.path emits it, run through the
* real classify() the popover uses. */
function backendRow(text: string, display: string, meta: string): Unstable_TriggerItem {
const c = classify({ text, display, meta })
return {
id: `${text}|0`,
type: c.type,
label: c.display,
metadata: { icon: c.type, display: c.display, meta: c.meta, rawText: text, insertId: c.insertId }
}
}
function typed(text: string) {
const editor = document.createElement('div')
editor.contentEditable = 'true'
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
editor.append(document.createTextNode(text))
const range = document.createRange()
range.selectNodeContents(editor)
range.collapse(false)
const sel = window.getSelection()
sel?.removeAllRanges()
sel?.addRange(range)
const editorRef = { current: editor as HTMLDivElement | null }
const { result } = renderHook(() =>
useComposerTrigger({
at: { adapter: null, loading: false },
draftRef: { current: text },
editorRef,
requestMainFocus: vi.fn(),
setComposerText: vi.fn(),
slash: { adapter: null, loading: false }
})
)
act(() => result.current.refreshTrigger())
return { editor, result }
}
/** The label the sent message renders for a committed draft. */
function sentLabel(draft: string) {
return hermesDirectiveFormatter
.parse(draft)
.filter((s): s is Extract<typeof s, { kind: 'mention' }> => s.kind === 'mention')
.map(s => s.label)
.join(',')
}
describe('one label per reference, on every surface', () => {
it('the popover row, the committed chip, and the sent chip all read the same', () => {
const cases = [
{ text: '@folder:apps/desktop/', display: 'desktop/', meta: 'dir' },
{ text: '@file:apps/desktop/src/main.tsx', display: 'main.tsx', meta: 'apps/desktop/src' },
{ text: '@folder:apps/desktop/src/', display: 'src/', meta: 'dir' }
]
for (const entry of cases) {
const item = backendRow(entry.text, entry.display, entry.meta)
const { editor, result } = typed('@desk')
act(() => result.current.replaceTriggerWithChip(item))
const row = String((item.metadata as { display: string }).display)
const chip = editor.querySelector('[data-ref-text]')?.textContent ?? ''
expect(chip).toBe(row)
expect(sentLabel(composerPlainText(editor))).toBe(row)
}
})
it('a folder pick reads as its path, not a bare basename', () => {
// `src` and `desktop` repeat all over a repo — the row you picked said
// where it was, and the chip has to keep saying it.
const item = backendRow('@folder:apps/desktop/', 'desktop/', 'dir')
expect(item.label).toBe('apps/desktop/')
const { editor, result } = typed('@desk')
act(() => result.current.replaceTriggerWithChip(item))
expect(editor.querySelector('[data-ref-text]')?.textContent).toBe('apps/desktop/')
})
it('Tab-descend leaves the live query, and the scope when there is one', () => {
const { editor, result } = typed('@folder:desk')
act(() =>
result.current.replaceTriggerWithChip(backendRow('@folder:apps/desktop/', 'desktop/', 'dir'), {
descend: true
})
)
// Mid-browse the editor holds the live query, scope included — that's the
// path being typed, not a label, and it's what the next completion reads.
expect(composerPlainText(editor)).toBe('@folder:apps/desktop/')
})
it('a url still reads host + path on every surface', () => {
const item = backendRow('@url:https://github.com/NousResearch/hermes-agent/pull/74533', '', '')
const { editor, result } = typed('@gith')
act(() => result.current.replaceTriggerWithChip(item))
const expected = 'github.com/NousResearch/hermes-agent/pull/74533'
expect(item.label).toBe(expected)
expect(editor.querySelector('[data-ref-text]')?.textContent).toBe(expected)
expect(sentLabel(composerPlainText(editor))).toBe(expected)
})
})
@@ -0,0 +1,164 @@
import type { Unstable_TriggerItem } from '@assistant-ui/core'
import { act, renderHook } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { useComposerTrigger } from './hooks/use-composer-trigger'
import { pathifyRefs } from './path-refs'
import { composerPlainText, insertComposerContentsAtCaret, RICH_INPUT_SLOT } from './rich-editor'
import { detectTrigger, openDirectiveScope, textBeforeCaret } from './text-utils'
import { linkifyUrls } from './url-refs'
function folderItem(rel: string): Unstable_TriggerItem {
const rawText = `@folder:${rel}/`
return {
id: `${rawText}|0`,
type: 'folder',
label: rel.split('/').filter(Boolean).pop() ?? rel,
metadata: { icon: 'folder', display: `${rel}/`, meta: 'dir', rawText, insertId: `${rel}/` }
}
}
/** Literally-typed text, caret `fromEnd` characters before the end. */
function typed(text: string, fromEnd = 0) {
const editor = document.createElement('div')
editor.contentEditable = 'true'
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
const node = document.createTextNode(text)
editor.append(node)
const range = document.createRange()
range.setStart(node, text.length - fromEnd)
range.collapse(true)
const sel = window.getSelection()
sel?.removeAllRanges()
sel?.addRange(range)
return editor
}
function withTrigger(editor: HTMLDivElement, draft: string) {
const editorRef = { current: editor as HTMLDivElement | null }
const { result } = renderHook(() =>
useComposerTrigger({
at: { adapter: null, loading: false },
draftRef: { current: draft },
editorRef,
requestMainFocus: vi.fn(),
setComposerText: vi.fn(),
slash: { adapter: null, loading: false }
})
)
act(() => result.current.refreshTrigger())
return result
}
/** The composer's paste handler, minus the clipboard plumbing. */
function paste(editor: HTMLDivElement, text: string) {
insertComposerContentsAtCaret(editor, pathifyRefs(linkifyUrls(text)), openDirectiveScope(editor))
}
describe('directive scope is a browse mode, not text to maintain', () => {
it('Tab-descend carries the scope down instead of dropping to a bare path', () => {
const editor = typed('@folder:apps/deskt')
const result = withTrigger(editor, '@folder:apps/deskt')
expect(result.current.trigger).toMatchObject({ kind: '@', scope: 'folder', value: 'apps/deskt' })
act(() => result.current.replaceTriggerWithChip(folderItem('apps/desktop'), { descend: true }))
expect(composerPlainText(editor)).toBe('@folder:apps/desktop/')
})
it('Backspace climbs the path, then drops the whole scope', () => {
const editor = typed('@folder:apps/desktop/')
const result = withTrigger(editor, '@folder:apps/desktop/')
act(() => result.current.ascendTriggerPath())
expect(composerPlainText(editor)).toBe('@folder:apps/')
act(() => result.current.refreshTrigger())
act(() => result.current.ascendTriggerPath())
expect(composerPlainText(editor)).toBe('@folder:')
// The scope is one unit: Backspace drops it whole rather than nibbling
// back through `:`, `r`, `e`, `d`, `l`, `o`, `f`.
act(() => result.current.refreshTrigger())
act(() => result.current.ascendTriggerPath())
expect(composerPlainText(editor)).toBe('@')
})
it('leaves Backspace alone when there is no scope and no path', () => {
const editor = typed('@apps')
const result = withTrigger(editor, '@apps')
let handled = true
act(() => {
handled = result.current.ascendTriggerPath()
})
expect(handled).toBe(false)
})
it('a pick mid-message keeps the trailing prose and consumes the whole token', () => {
const editor = typed('@folder:apps/deskt and some trailing words', 24)
const result = withTrigger(editor, '@folder:apps/deskt and some trailing words')
act(() => result.current.replaceTriggerWithChip(folderItem('apps/desktop')))
expect(composerPlainText(editor)).toBe('@folder:`apps/desktop/` and some trailing words')
expect(editor.querySelector('[data-ref-kind="folder"]')).not.toBeNull()
})
it('pasting into an open @url: scope consumes it instead of stacking', () => {
const editor = typed('refer to @url:')
paste(editor, 'https://github.com/NousResearch/hermes-agent/pull/74533')
expect(composerPlainText(editor)).toBe('refer to @url:`https://github.com/NousResearch/hermes-agent/pull/74533`')
expect(editor.textContent).not.toContain('@url:@url:')
})
it('a normal paste with no open scope is untouched', () => {
const editor = typed('look at ')
paste(editor, 'https://example.com/x')
expect(composerPlainText(editor)).toBe('look at @url:`https://example.com/x`')
})
it('scope parsing leaves an unscoped @ query alone', () => {
expect(detectTrigger('@apps/desk')).toMatchObject({ kind: '@', value: 'apps/desk' })
expect(detectTrigger('@apps/desk')?.scope).toBeUndefined()
})
it('openDirectiveScope only fires on an EMPTY scope', () => {
// The count is what a paste consumes: `@url:` is 5 characters of syntax
// the user never typed and shouldn't be left holding.
expect(openDirectiveScope(typed('@url:'))).toBe(5)
expect(openDirectiveScope(typed('@url:https://x.com'))).toBe(0)
expect(openDirectiveScope(typed('plain text'))).toBe(0)
})
it('chips stay atomic to scope detection', () => {
const editor = typed('@folder:apps/desktop/')
const result = withTrigger(editor, '@folder:apps/desktop/')
act(() => result.current.replaceTriggerWithChip(folderItem('apps/desktop')))
// A committed chip is one object-replacement char, so a fresh `@` typed
// after it opens an unscoped browse rather than inheriting the old scope.
expect(detectTrigger(`${textBeforeCaret(editor)}@`)?.scope).toBeUndefined()
})
})
@@ -0,0 +1,139 @@
import { describe, expect, it } from 'vitest'
import {
composerPlainText,
normalizeComposerEditorDom,
renderComposerContents,
RICH_INPUT_SLOT
} from './rich-editor'
function editor(): HTMLDivElement {
const el = document.createElement('div')
el.dataset.slot = RICH_INPUT_SLOT
el.contentEditable = 'true'
document.body.append(el)
return el
}
/** Whatever emptied it — Delete, cut, Chromium's own selection-delete — the
* normalizer lands on the same DOM. */
function emptied(): HTMLDivElement {
const el = editor()
el.append(document.createTextNode('hello'))
el.replaceChildren()
normalizeComposerEditorDom(el)
return el
}
describe('an emptied composer reads as empty', () => {
it('keeps the placeholder <br> so the contenteditable holds its height', () => {
// The scaffolding is deliberate: a childless contenteditable collapses to a
// sliver in Chromium. It just must not read as content.
expect(emptied().innerHTML).toBe('<br>')
})
it('reads that editor as empty, not as a newline', () => {
expect(composerPlainText(emptied())).toBe('')
})
it('reads a truly childless editor as empty', () => {
expect(composerPlainText(editor())).toBe('')
})
it('still reads a real Shift+Enter line break as a newline', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'), document.createTextNode('two'))
expect(composerPlainText(el)).toBe('one\ntwo')
})
it('still reads a trailing break after text as a newline', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'))
expect(composerPlainText(el)).toBe('one\n')
})
it('only treats the EDITOR\u2019s lone <br> as scaffolding, not a nested one', () => {
// A lone <br> inside some other element is a real line break; the exemption
// is scoped to the editor root by its slot marker. (The block wrapper adds
// its own trailing newline — unchanged behavior, asserted so the exemption
// can't quietly widen to nested nodes.)
const el = editor()
const inner = document.createElement('div')
inner.append(document.createElement('br'))
el.append(document.createTextNode('one'), inner)
expect(composerPlainText(el)).toBe('one\n\n')
})
})
/** The rule the stylesheet paints the placeholder with. `:empty` alone goes
* false the instant the scaffolding <br> lands. */
const PLACEHOLDER_SHOWS = ':is(:empty, [data-empty])'
describe('an emptied composer shows its placeholder again', () => {
it('advertises emptiness once the scaffolding break is in place', () => {
expect(emptied().matches(PLACEHOLDER_SHOWS)).toBe(true)
})
it('advertises emptiness for a truly childless editor', () => {
expect(editor().matches(PLACEHOLDER_SHOWS)).toBe(true)
})
it('stops advertising it once something is typed', () => {
const el = emptied()
el.replaceChildren(document.createTextNode('hi'))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
// A text node is invisible to selectors, so `one<br>` and `<br>` are the same
// shape to any pure-CSS rule (`:has(> br:only-child)` matches both and paints
// the placeholder straight over the user's text). The DOM writer has to say.
it('does not advertise emptiness for a trailing break after text', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
it('does not advertise emptiness for a Shift+Enter break between text', () => {
const el = editor()
el.append(document.createTextNode('one'), document.createElement('br'), document.createTextNode('two'))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
// Repainting from text (restored draft, undo, completion rebuild) is the
// other writer that reshapes the editor root — it must not strand the marker.
it('drops the marker when a draft is painted back in', () => {
const el = emptied()
renderComposerContents(el, 'restored draft')
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
it('re-advertises emptiness when a draft is painted back out', () => {
const el = editor()
renderComposerContents(el, 'temporary')
renderComposerContents(el, '')
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true)
})
})
@@ -1,12 +1,16 @@
import { afterEach, describe, expect, it } from 'vitest'
import { $hoveredTreeGroup } from '@/components/pane-shell/tree/store'
import {
blurComposerInput,
getActiveComposer,
markActiveComposer,
onComposerFocusRequest,
onComposerModelMenuRequest,
releaseActiveComposer,
requestComposerFocus
requestComposerFocus,
requestModelMenuToggle
} from './focus'
import { RICH_INPUT_SLOT } from './rich-editor'
@@ -45,6 +49,7 @@ afterEach(() => {
// `activeTarget` is module-level — a case that leaves a stale claim behind
// would otherwise decide the next one.
markActiveComposer('main')
$hoveredTreeGroup.set(null)
})
describe('blurComposerInput', () => {
@@ -216,3 +221,65 @@ describe('resolveActive / keep-alive tab heal', () => {
expect(getActiveComposer()).toBe('edit')
})
})
/** A chat surface inside a layout zone, mirroring ChatView-in-tree-group. */
function mountZonedSurface(target: string, zone: string, hidden = false) {
const group = document.createElement('div')
group.dataset.treeGroup = zone
const layer = document.createElement('div')
layer.toggleAttribute('data-pane-hidden', hidden)
const surface = document.createElement('div')
surface.dataset.composerTarget = target
layer.append(surface)
group.append(layer)
document.body.append(group)
return surface
}
const collectModelMenuTargets = async (): Promise<string[]> => {
const saw: string[] = []
const off = onComposerModelMenuRequest(target => saw.push(target))
await new Promise(resolve => window.setTimeout(resolve, 0))
off()
return saw
}
describe('requestModelMenuToggle', () => {
it('targets the pane under the pointer over the focused one (#74447 convention)', async () => {
mountZonedSurface('main', 'zone-a')
mountZonedSurface('tile:hovered', 'zone-b')
markActiveComposer('main')
$hoveredTreeGroup.set('zone-b')
expect(requestModelMenuToggle()).toBe(true)
expect(await collectModelMenuTargets()).toEqual(['tile:hovered'])
})
it('falls back to the active composer when the pointer is off every zone', async () => {
mountZonedSurface('main', 'zone-a')
mountZonedSurface('tile:other', 'zone-b')
markActiveComposer('tile:other')
expect(requestModelMenuToggle()).toBe(true)
expect(await collectModelMenuTargets()).toEqual(['tile:other'])
})
it('skips a hidden keep-alive tab in the hovered zone (targets its visible sibling)', async () => {
mountZonedSurface('main', 'zone-a', true)
mountZonedSurface('tile:front', 'zone-a')
markActiveComposer('main')
$hoveredTreeGroup.set('zone-a')
expect(requestModelMenuToggle()).toBe(true)
expect(await collectModelMenuTargets()).toEqual(['tile:front'])
})
it('returns false with no chat surface on screen so the caller can open the dialog', async () => {
// Settings/profiles routes: no [data-composer-target] anywhere.
expect(requestModelMenuToggle()).toBe(false)
expect(await collectModelMenuTargets()).toEqual([])
})
})
+41 -1
View File
@@ -10,7 +10,8 @@
* steal focus from the composer effect.
*/
import { queryVisible } from '@/components/pane-shell/pane-visibility'
import { queryAllVisible, queryVisible } from '@/components/pane-shell/pane-visibility'
import { $hoveredTreeGroup } from '@/components/pane-shell/tree/store'
import type { InlineRefInput } from './inline-refs'
import { RICH_INPUT_SLOT } from './rich-editor'
@@ -42,6 +43,7 @@ const INSERT_EVENT = 'hermes:composer-insert'
const INSERT_REFS_EVENT = 'hermes:composer-insert-refs'
const SUBMIT_EVENT = 'hermes:composer-submit'
const VOICE_TOGGLE_EVENT = 'hermes:composer-voice-toggle'
const MODEL_MENU_EVENT = 'hermes:composer-model-menu'
/** Inline edit composer root — mounted only while a user bubble is being edited. */
const EDIT_COMPOSER_ROOT = '[data-slot="aui_edit-composer-root"]'
@@ -258,6 +260,44 @@ export const requestVoiceToggle = (target: ComposerTarget | 'active' = 'active')
export const onComposerVoiceToggleRequest = (handler: (target: ComposerTarget) => void) =>
subscribe<{ target: ComposerTarget }>(VOICE_TOGGLE_EVENT, ({ target }) => handler(target))
/** The chat surface inside the zone the pointer is over, if any. Mirrors the
* tab verbs' hover-first targeting (`tabTargetGroupId`, #74447): the model
* hotkey lands in the pane you're pointing at without clicking into it first.
* Hidden keep-alive tabs are skipped like every document-wide lookup. */
const composerTargetInHoveredZone = (): ComposerTarget | null => {
const zone = $hoveredTreeGroup.get()
if (!zone || typeof document === 'undefined') {
return null
}
const surface = queryAllVisible<HTMLElement>('[data-composer-target]').find(
el => el.closest<HTMLElement>('[data-tree-group]')?.dataset.treeGroup === zone
)
return (surface?.dataset.composerTarget as ComposerTarget | undefined) ?? null
}
/** Toggle ONE composer's model menu — the `composer.modelPicker` hotkey.
* Targets the pane under the pointer first (the tab-verb convention), then
* the active composer. Returns false when no chat surface is on screen at
* all (settings, profiles…), so the caller can fall back to the full
* model-picker dialog instead of dispatching into the void. */
export const requestModelMenuToggle = (): boolean => {
if (typeof document !== 'undefined' && !queryVisible('[data-composer-target]')) {
return false
}
dispatch<{ target: ComposerTarget }>(MODEL_MENU_EVENT, {
target: composerTargetInHoveredZone() ?? resolveActive()
})
return true
}
export const onComposerModelMenuRequest = (handler: (target: ComposerTarget) => void) =>
subscribe<{ target: ComposerTarget }>(MODEL_MENU_EVENT, ({ target }) => handler(target))
/**
* Focus a composer input across React commit + browser focus restore.
*
@@ -0,0 +1,107 @@
import { act, renderHook } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { queryClient } from '@/lib/query-client'
import { useAtCompletions } from './use-at-completions'
function gatewayStub(latencyMs = 40) {
const calls: string[] = []
const gateway = {
request: vi.fn(async (_method: string, params: { word: string }) => {
calls.push(params.word)
await new Promise(r => setTimeout(r, latencyMs))
return { items: [{ text: `@folder:${params.word.slice(1)}x/`, display: 'x/', meta: 'dir' }] }
})
}
return { calls, gateway }
}
function setup(latencyMs = 40) {
const { calls, gateway } = gatewayStub(latencyMs)
const { result } = renderHook(() => useAtCompletions({ gateway: gateway as never, sessionId: 's1', cwd: '/repo' }))
return { calls, result }
}
/** Type a burst of keystrokes `gapMs` apart, like a person. */
async function type(
result: { current: { adapter: { search?: (q: string) => unknown } } },
queries: string[],
gapMs: number
) {
for (const q of queries) {
act(() => {
result.current.adapter.search?.(q)
})
await act(async () => {
await vi.advanceTimersByTimeAsync(gapMs)
})
}
}
describe('PERF: @ path completions are cached and skip the debounce', () => {
it('serves a repeated query with no round trip and no spinner', async () => {
vi.useFakeTimers()
queryClient.clear()
const { calls, result } = setup()
// First visit to `apps/` pays the round trip.
await type(result, ['apps/'], 0)
await act(async () => {
await vi.advanceTimersByTimeAsync(200)
})
const afterFirst = calls.length
expect(afterFirst).toBe(1)
// Walk away and come back — Tab in, Backspace out, retype. Every one of
// these used to be a fresh git ls-files + rank on the backend.
await type(result, ['apps/desktop/', 'apps/', 'apps/desktop/', 'apps/'], 0)
await act(async () => {
await vi.advanceTimersByTimeAsync(200)
})
// Two distinct paths, so exactly two round trips total — the repeats are free.
expect(calls.length).toBe(2)
expect(result.current.loading).toBe(false)
vi.useRealTimers()
})
it('a cached query paints without waiting out the debounce', async () => {
vi.useFakeTimers()
queryClient.clear()
const { calls, result } = setup()
await type(result, ['apps/'], 0)
await act(async () => {
await vi.advanceTimersByTimeAsync(200)
})
expect(calls.length).toBe(1)
// Re-ask for the cached query and advance by far less than the 60ms
// debounce. A cached answer resolves in a microtask, so it must paint
// without the timer and without ever flipping the spinner on.
act(() => {
result.current.adapter.search?.('apps/')
})
await act(async () => {
await vi.advanceTimersByTimeAsync(1)
})
expect(result.current.loading).toBe(false)
expect(calls.length).toBe(1)
vi.useRealTimers()
})
})
@@ -1,7 +1,9 @@
import type { Unstable_TriggerAdapter, Unstable_TriggerItem } from '@assistant-ui/core'
import { useCallback } from 'react'
import { refChipLabel } from '@/components/assistant-ui/directive-text'
import type { HermesGateway } from '@/hermes'
import { cachedPathCompletion, hasCachedPathCompletion } from '@/lib/slash-completion-cache'
import { normalize } from '@/lib/text'
import type { CompletionEntry, CompletionPayload } from './use-live-completion-adapter'
@@ -60,7 +62,14 @@ function classify(entry: CompletionEntry): {
return {
type: kind,
insertId: rest,
display: textValue(entry.display, rest || `@${kind}:`),
// The row shows exactly what picking it produces. Upstream keeps one
// label per item and hands it to the chip verbatim (DirectiveNode's
// `__label = item.label`); our wire format is `@kind:value`, which can't
// carry a label the way their `:type[label]{name=id}` does, so the same
// invariant is held by deriving both ends from refChipLabel. Without
// this the list said `desktop/`, the editor said `apps/desktop/`, and
// the chip said `desktop` — three names for one folder.
display: rest ? refChipLabel(kind, rest) : textValue(entry.display, `@${kind}:`),
meta: textValue(entry.meta)
}
}
@@ -82,6 +91,11 @@ export function useAtCompletions(options: {
const { gateway, sessionId, cwd } = options
const enabled = Boolean(gateway)
// Cache key: the completion depends on the query AND the directory it's
// resolved against, so a cwd or session change can't serve another tree's
// listing.
const cacheKey = useCallback((query: string) => `${cwd ?? ''}|${sessionId ?? ''}|${query}`, [cwd, sessionId])
const fetcher = useCallback(
async (query: string): Promise<CompletionPayload> => {
const starters = starterEntries(query)
@@ -102,7 +116,15 @@ export function useAtCompletions(options: {
}
try {
const result = await gateway.request<{ items?: CompletionEntry[] }>('complete.path', params)
// De-duplicated the same way `/` completions are. Walking a path is
// inherently repetitive — Tab into a folder, Backspace out, retype a
// segment — and every one of those steps used to be a fresh
// `git ls-files` + rank on the backend (~40ms of the ~50ms round trip
// measured on this repo's 8k files).
const result = await cachedPathCompletion(cacheKey(query), () =>
gateway.request<{ items?: CompletionEntry[] }>('complete.path', params)
)
const items = result.items ?? []
return { items: items.length > 0 ? items : starters, query }
@@ -110,7 +132,7 @@ export function useAtCompletions(options: {
return { items: starters, query }
}
},
[gateway, sessionId, cwd]
[cacheKey, gateway, sessionId, cwd]
)
const toItem = useCallback((entry: CompletionEntry, index: number): Unstable_TriggerItem => {
@@ -135,7 +157,13 @@ export function useAtCompletions(options: {
}
}, [])
return useLiveCompletionAdapter({ enabled, fetcher, toItem })
// A query already in cache skips both the debounce and the loading state.
// This is what makes walking a tree feel instant rather than merely fast:
// the 60ms debounce exists to avoid a request per keystroke, and it buys
// nothing when the answer is already in hand.
const isCached = useCallback((query: string) => hasCachedPathCompletion(cacheKey(query)), [cacheKey])
return useLiveCompletionAdapter({ enabled, fetcher, isCached, toItem })
}
/** Re-export `classify` for use by the formatter (insertion side). */
@@ -2,6 +2,7 @@ import { useAui, useAuiState, useComposerRuntime } from '@assistant-ui/react'
import { type RefObject, useCallback, useEffect, useLayoutEffect, useRef, useState } from 'react'
import { SLASH_COMMAND_RE } from '@/lib/chat-runtime'
import { sanitizeComposerInput } from '@/lib/composer-input-sanitize'
import { type ComposerAttachment, stashSessionDraft, takeSessionDraft } from '@/store/composer'
import { isBrowsingHistory } from '@/store/composer-input-history'
@@ -21,7 +22,13 @@ import {
releaseActiveComposer
} from '../focus'
import { type InlineRefInput, insertInlineRefsIntoEditor } from '../inline-refs'
import { composerPlainText, placeCaretEnd, REF_RE, renderComposerContents } from '../rich-editor'
import {
composerPlainText,
normalizeComposerEditorDom,
placeCaretEnd,
REF_RE,
renderComposerContents
} from '../rich-editor'
import { useComposerScope } from '../scope'
import type { ChatBarProps } from '../types'
@@ -121,7 +128,7 @@ export function useComposerDraft({
const editor = editorRef.current
if (editor) {
renderComposerContents(editor, next)
renderComposerContents(editor, next, { trailingCommitted: true })
placeCaretEnd(editor)
}
@@ -237,7 +244,13 @@ export function useComposerDraft({
return draftRef.current
}
const text = composerPlainText(editor)
// Same normalize-then-sanitize the rAF flush does. An emptied editor still
// holds the placeholder <br> that keeps the contenteditable from collapsing
// to a sliver, and that serializes as "\n" — so an editor the user just
// cleared would otherwise stash a one-newline draft and come back non-empty.
normalizeComposerEditorDom(editor)
const text = sanitizeComposerInput(composerPlainText(editor))
if (text !== draftRef.current) {
draftRef.current = text
@@ -265,7 +278,7 @@ export function useComposerDraft({
const editor = editorRef.current
if (editor && document.activeElement !== editor && composerPlainText(editor) !== text) {
renderComposerContents(editor, text)
renderComposerContents(editor, text, { trailingCommitted: true })
}
if (isBrowsingHistory(sessionIdRef.current) || queueEditRef.current) {
@@ -14,6 +14,7 @@ import { useResizeObserver } from '@/hooks/use-resize-observer'
import { COMPOSER_COMPACT_PILL_PX, COMPOSER_SINGLE_LINE_MAX_PX, COMPOSER_STACK_BREAKPOINT_PX } from '../composer-utils'
interface UseComposerMetricsArgs {
composerDockRef: RefObject<HTMLDivElement | null>
composerRef: RefObject<HTMLFormElement | null>
composerSurfaceRef: RefObject<HTMLDivElement | null>
editorRef: RefObject<HTMLDivElement | null>
@@ -28,7 +29,13 @@ interface UseComposerMetricsArgs {
* tree's computed style, and `tight` only flips when it crosses the breakpoint.
* Returns `stacked` (the only value the render needs).
*/
export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef, poppedOut }: UseComposerMetricsArgs): {
export function useComposerMetrics({
composerDockRef,
composerRef,
composerSurfaceRef,
editorRef,
poppedOut
}: UseComposerMetricsArgs): {
compactPill: boolean
stacked: boolean
} {
@@ -89,8 +96,11 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
const syncComposerMetrics = useCallback(() => {
const composer = composerRef.current
// The dock is the full docked footprint — strips, status stack, composer —
// so it, not the composer alone, is what the thread has to clear.
const dock = composerDockRef.current
if (!composer) {
if (!composer || !dock) {
return
}
@@ -108,7 +118,8 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
return
}
const { height, width } = composer.getBoundingClientRect()
const { height } = dock.getBoundingClientRect()
const { width } = composer.getBoundingClientRect()
const surfaceHeight = composerSurfaceRef.current?.getBoundingClientRect().height
if (width > 0) {
@@ -156,9 +167,9 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
setSurfaceVar(composer, COMPOSER_SURFACE_HEIGHT_VAR, `${bucket}px`)
}
}
}, [composerRef, composerSurfaceRef, editorRef])
}, [composerDockRef, composerRef, composerSurfaceRef, editorRef])
useResizeObserver(syncComposerMetrics, composerRef, composerSurfaceRef, editorRef)
useResizeObserver(syncComposerMetrics, composerDockRef, composerRef, composerSurfaceRef, editorRef)
// Toggling pop-out changes whether the composer reserves thread clearance.
// The ResizeObserver may not fire (the box can keep the same box size), so
@@ -170,10 +181,8 @@ export function useComposerMetrics({ composerRef, composerSurfaceRef, editorRef,
useEffect(() => {
// Resolve the owning surface while the composer is still attached; the
// unmount cleanup runs after React detached the node, where closest()
// can no longer find [data-chat-surface] and would clear the document
// root instead of this surface (same class of bug as the status stack's
// stale-clearance leak).
// unmount cleanup runs after React detached the node, where closest() can
// no longer find [data-chat-surface].
const root = chatSurfaceRoot(composerRef.current)
return () => {
@@ -223,3 +223,72 @@ describe('useComposerTrigger — free-text slash arguments', () => {
expect(editor.querySelector('[data-slash-kind]')?.getAttribute('data-ref-text')).toBe('/personality creative')
})
})
describe('useComposerTrigger — chip survival (the plaintext-demotion bug class)', () => {
it('keeps a leading command pill through a Backspace path-ascend', () => {
// The reported repro: `/work @folder…` then Backspace — both chips went
// plaintext because ascend re-rendered the whole editor from text.
const editor = mountEditor('/work @Desktop/')
const { hook } = mountTrigger(editor, [])
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
act(() => hook.result.current.refreshTrigger())
expect(hook.result.current.trigger).toMatchObject({ kind: '@', query: 'Desktop/' })
let ran = false
act(() => {
ran = hook.result.current.ascendTriggerPath()
})
expect(ran).toBe(true)
expect(composerPlainText(editor)).toBe('/work @')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
})
it('keeps a leading command pill when a folder pick commits its ref chip', () => {
const editor = mountEditor('/work @Desk')
const folder: Unstable_TriggerItem = {
id: 'folder:Desktop',
type: 'folder',
label: 'Desktop',
metadata: { rawText: '@folder:Desktop', insertId: 'Desktop' }
}
const { hook } = mountTrigger(editor, [folder])
act(() => hook.result.current.refreshTrigger())
act(() => hook.result.current.replaceTriggerWithChip(folder))
expect(composerPlainText(editor)).toBe('/work @folder:`Desktop` ')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
expect(editor.querySelector('[data-ref-kind="folder"]')).not.toBeNull()
})
it('commits in place when Chromium has split the token across text nodes', () => {
// Chromium fragments text nodes around contenteditable=false chips; the
// commit path must span the fragments instead of bailing to a full
// re-render.
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
editor.contentEditable = 'true'
document.body.append(editor)
editor.append(document.createTextNode('please run /c'), document.createTextNode('le'))
const caret = document.createRange()
caret.setStart(editor.lastChild!, 2)
caret.collapse(true)
const selection = window.getSelection()!
selection.removeAllRanges()
selection.addRange(caret)
const { hook } = mountTrigger(editor, [item('/clean')])
act(() => hook.result.current.refreshTrigger())
act(() => hook.result.current.replaceTriggerWithChip(item('/clean')))
expect(composerPlainText(editor)).toBe('please run /clean ')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
})
})
@@ -12,14 +12,59 @@ import {
slashCommandToken
} from '../composer-utils'
import {
appendComposerContents,
caretOffsetInEditor,
composerPlainText,
placeCaretEnd,
placeCaretAtOffset,
refChipElement,
renderComposerContents,
replaceBeforeCaret,
RICH_INPUT_SLOT,
slashChipElement
} from '../rich-editor'
import { detectTrigger, textBeforeCaret, type TriggerState } from '../text-utils'
/**
* Rebuild-from-text fallback for carets the range walk can't anchor (a
* non-collapsed selection, a caret not preceded by contiguous text). It
* re-renders the whole editor from serialized text, so it only runs when the
* in-place path reports failure — never as the default.
*
* The split is around the CARET, not the end of the draft. Slicing
* `length - tokenLength` off the end assumed the trigger token was the last
* thing in the editor: a completion picked mid-message chopped the trailing
* prose off and stranded a partial `folder:` in front of the chip, because the
* window it removed wasn't the token the user was typing.
*/
export function rebuildAroundCaret(editor: HTMLDivElement, tokenLength: number, insert: DocumentFragment | string) {
const current = composerPlainText(editor)
const caret = caretOffsetInEditor(editor)
const prefix = current.slice(0, Math.max(0, caret - tokenLength))
const suffix = current.slice(caret)
if (typeof insert === 'string') {
renderComposerContents(editor, `${prefix}${insert}${suffix}`)
placeCaretAtOffset(editor, prefix.length + insert.length)
return
}
// Measure before appending — moving a fragment empties it. Appending the
// element rather than re-serializing keeps mid-message slash pills alive:
// they have no text hydration, unlike `@` refs and the leading command.
const scratch = document.createElement('div')
scratch.dataset.slot = RICH_INPUT_SLOT
scratch.append(insert.cloneNode(true))
const inserted = composerPlainText(scratch)
renderComposerContents(editor, prefix)
editor.append(insert)
appendComposerContents(editor, suffix)
placeCaretAtOffset(editor, prefix.length + inserted.length)
}
interface CompletionSource {
adapter: Unstable_TriggerAdapter | null
loading: boolean
@@ -29,6 +74,10 @@ interface UseComposerTriggerOptions {
at: CompletionSource
draftRef: MutableRefObject<string>
editorRef: RefObject<HTMLDivElement | null>
/** `:joy` emoji completions — inserts the emoji character, never a chip. */
emoji?: CompletionSource
/** Bank the pre-commit state so a popover pick is a single undo step. */
recordUndoPoint?: () => void
requestMainFocus: () => void
setComposerText: (text: string) => void
slash: CompletionSource
@@ -47,6 +96,8 @@ export function useComposerTrigger({
at,
draftRef,
editorRef,
emoji,
recordUndoPoint,
requestMainFocus,
setComposerText,
slash
@@ -87,7 +138,7 @@ export function useComposerTrigger({
// is present do we pay the cost of the full walk + DOM range work.
const rawText = editor.textContent ?? ''
if (!rawText.includes('@') && !rawText.includes('/')) {
if (!rawText.includes('@') && !rawText.includes('/') && !rawText.includes(':')) {
if (trigger) {
setTrigger(null)
resetTriggerActive()
@@ -124,7 +175,13 @@ export function useComposerTrigger({
}, [editorRef, resetTriggerActive, trigger])
const triggerAdapter: Unstable_TriggerAdapter | null =
trigger?.kind === '@' ? at.adapter : trigger?.kind === '/' ? slash.adapter : null
trigger?.kind === '@'
? at.adapter
: trigger?.kind === '/'
? slash.adapter
: trigger?.kind === ':'
? (emoji?.adapter ?? null)
: null
useEffect(() => {
if (!trigger || !triggerAdapter?.search) {
@@ -142,7 +199,14 @@ export function useComposerTrigger({
setTriggerItems(trigger.inline ? items.filter(isSkillItem) : items)
}, [trigger, triggerAdapter])
const triggerLoading = trigger?.kind === '@' ? at.loading : trigger?.kind === '/' ? slash.loading : false
const triggerLoading =
trigger?.kind === '@'
? at.loading
: trigger?.kind === '/'
? slash.loading
: trigger?.kind === ':'
? (emoji?.loading ?? false)
: false
// Suppress the "No matches" empty state once a slash command is past its name:
// a no-arg command has nothing to offer, and a fully-typed arg commits on
@@ -214,17 +278,22 @@ export function useComposerTrigger({
return
}
// Bank the pre-commit state first — every path below mutates the editor,
// and a pick must be exactly one undo step.
recordUndoPoint?.()
const rebuildAround = (insert: DocumentFragment | string) => rebuildAroundCaret(editor, trigger.tokenLength, insert)
// Action items (e.g. "Browse all sessions…") run a side effect instead of
// inserting a chip: strip the typed trigger token, then fire the action.
const completionAction = (item.metadata as { action?: unknown } | undefined)?.action
const runAction = typeof completionAction === 'string' ? COMPLETION_ACTIONS[completionAction] : undefined
if (runAction) {
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
if (!replaceBeforeCaret(editor, trigger.tokenLength, document.createDocumentFragment())) {
rebuildAround('')
}
renderComposerContents(editor, prefix)
placeCaretEnd(editor)
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
closeTrigger()
@@ -247,19 +316,29 @@ export function useComposerTrigger({
? String((item.metadata as { insertId?: unknown } | undefined)?.insertId ?? '')
: ''
if (descendInto) {
const path = descendInto.endsWith('/') ? descendInto : `${descendInto}/`
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
renderComposerContents(editor, `${prefix}@${path}`)
placeCaretEnd(editor)
const finish = (keepOpen: boolean) => {
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
requestMainFocus()
window.setTimeout(refreshTrigger, 0)
keepOpen ? window.setTimeout(refreshTrigger, 0) : closeTrigger()
}
return
if (descendInto) {
const path = descendInto.endsWith('/') ? descendInto : `${descendInto}/`
// Carry the browse scope down with the path. Dropping it turned an
// explicit `@folder:` browse into a bare `@apps/desktop/` token halfway
// through, so the next completion silently widened back to files and the
// committed chip had to re-guess the kind from a trailing slash.
const scope = trigger.scope ? `${trigger.scope}:` : ''
const fragment = document.createDocumentFragment()
fragment.append(document.createTextNode(`@${scope}${path}`))
if (!replaceBeforeCaret(editor, trigger.tokenLength, fragment)) {
rebuildAround(`@${scope}${path}`)
}
return finish(true)
}
// Picking a bare arg-taking command (e.g. `/personality`) shouldn't commit
@@ -280,90 +359,78 @@ export function useComposerTrigger({
const slashKind = !expandsToArgs && trigger.kind === '/' ? slashChipKindForItem(item) : null
const keepTriggerOpen = starter || (expandsToArgs && argumentMode !== 'text')
const finish = () => {
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
requestMainFocus()
keepTriggerOpen ? window.setTimeout(refreshTrigger, 0) : closeTrigger()
}
const sel = window.getSelection()
const range = sel?.rangeCount ? sel.getRangeAt(0) : null
const node = range?.startContainer
const offset = range?.startOffset ?? 0
if (!sel || !range || node?.nodeType !== Node.TEXT_NODE || offset < trigger.tokenLength) {
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
if (slashKind) {
// Two-step arg picks (e.g. `/handoff` pill already inserted, now picking
// the platform) land here because the caret sits past a contenteditable
// chip. Rebuild the prefix and re-emit a single pill for the full command.
renderComposerContents(editor, prefix)
editor.append(slashChipElement(serialized, slashKind), document.createTextNode(' '))
placeCaretEnd(editor)
return finish()
}
renderComposerContents(editor, `${prefix}${text}`)
placeCaretEnd(editor)
return finish()
}
const replaceRange = document.createRange()
replaceRange.setStart(node, offset - trigger.tokenLength)
replaceRange.setEnd(node, offset)
replaceRange.deleteContents()
const chip = slashKind
? slashChipElement(serialized, slashKind)
: directive
? refChipElement(directive[1], directive[2])
? // Carry the picked row's own label into the chip rather than letting
// it re-derive one from the value. Upstream's DirectiveNode does the
// same (`__label = item.label`), and it's what makes the list and the
// chip agree: you get the string you just read, not a second guess at
// it. Falls back to the shared deriver for callers with no label.
refChipElement(directive[1], directive[2], (item.metadata as { display?: string })?.display || item.label)
: null
if (chip) {
const space = document.createTextNode(' ')
const fragment = document.createDocumentFragment()
fragment.append(chip, space)
replaceRange.insertNode(fragment)
// The trailing space is a convenience for "keep typing after the chip", so
// it's wrong when the caret already has whitespace in front of it — a pick
// made mid-sentence would leave a double space in the prose.
const followedBySpace = /^\s/.test(composerPlainText(editor).slice(caretOffsetInEditor(editor)))
const fragment = document.createDocumentFragment()
const caret = document.createRange()
caret.setStart(space, 1)
caret.collapse(true)
sel.removeAllRanges()
sel.addRange(caret)
chip
? fragment.append(chip, ...(followedBySpace ? [] : [document.createTextNode(' ')]))
: fragment.append(document.createTextNode(followedBySpace ? text.trimEnd() : text))
return finish()
if (!replaceBeforeCaret(editor, trigger.tokenLength, fragment)) {
// The failed in-place attempt never consumed the fragment, so the chip +
// trailing space are re-inserted around the caret here. Moving the
// element (rather than re-serializing) keeps mid-message slash pills
// alive — they have no text hydration, unlike `@` refs and the leading
// command.
rebuildAround(chip ? fragment : text)
}
document.execCommand('insertText', false, text)
finish()
finish(keepTriggerOpen)
}
/** Backspace inside an `@` path drops the last segment (`a/b/` → `a/`)
* instead of one character. Descending is one Tab per level, so climbing
* back out should cost one key too rather than a held delete. Returns
* instead of one character, and once the path is empty it drops the browse
* scope (`@folder:` → `@`) rather than nibbling `:`, `r`, `e`, `d`… back
* through the directive syntax the user never typed. Descending is one Tab
* per level, so climbing back out costs one key per level too. Returns
* false when the caret isn't in a path, so keydown falls through. */
const ascendTriggerPath = () => {
const editor = editorRef.current
if (!editor || trigger?.kind !== '@' || !trigger.query.includes('/')) {
if (!editor || trigger?.kind !== '@') {
return false
}
const scope = trigger.scope ? `${trigger.scope}:` : ''
if (!trigger.value.includes('/') && !scope) {
return false
}
// Trailing slash means we're listing a folder's children: drop that
// folder. Otherwise a partial segment is typed — drop just that.
const trimmed = trigger.query.replace(/\/$/, '')
// folder. Otherwise a partial segment is typed — drop just that. With the
// value already empty, the only thing left to drop is the scope itself.
const trimmed = trigger.value.replace(/\/$/, '')
const parent = trimmed.slice(0, trimmed.lastIndexOf('/') + 1)
const next = trigger.value ? `${scope}${parent}` : ''
const current = composerPlainText(editor)
const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength))
recordUndoPoint?.()
const fragment = document.createDocumentFragment()
fragment.append(document.createTextNode(`@${next}`))
// In place first: the destructive re-render fallback rebuilds the editor
// from text, which is exactly what used to demote a leading command pill
// to plaintext on every Backspace inside a path.
if (!replaceBeforeCaret(editor, trigger.tokenLength, fragment)) {
rebuildAroundCaret(editor, trigger.tokenLength, `@${next}`)
}
renderComposerContents(editor, `${prefix}@${parent}`)
placeCaretEnd(editor)
draftRef.current = composerPlainText(editor)
setComposerText(draftRef.current)
window.setTimeout(refreshTrigger, 0)
@@ -27,6 +27,9 @@ interface UseComposerVoiceArgs {
focusInput: () => void
insertText: (text: string) => void
maxRecordingSeconds: number
/** Interrupt the in-flight agent turn (Stop-button seam) — fired when the
* user speaks over the model while it is still generating. */
onInterrupt?: () => Promise<void> | void
onSubmit: ChatBarProps['onSubmit']
onTranscribeAudio: ChatBarProps['onTranscribeAudio']
sessionId: string | null | undefined
@@ -48,6 +51,7 @@ export function useComposerVoice({
focusInput,
insertText,
maxRecordingSeconds,
onInterrupt,
onSubmit,
onTranscribeAudio,
sessionId,
@@ -129,6 +133,10 @@ export function useComposerVoice({
consumePendingResponse,
enabled: voiceConversationActive,
onFatalError: () => setVoiceConversationActive(false),
// Speaking over the model mid-generation interrupts the in-flight turn —
// the same seam as the Stop button — so the interjection becomes the next
// turn instead of waiting behind a reply the user already rejected.
onInterrupt,
// A spoken stop command ("stop", "never mind", "goodbye", …) ends the
// hands-free conversation. Flipping the flag is the authoritative off
// switch — the enabled=false prop + effect below drive conversation.end()
@@ -0,0 +1,122 @@
import { useCallback } from 'react'
import { type CompletionEntry, type CompletionPayload, useLiveCompletionAdapter } from './use-live-completion-adapter'
/**
* `:shortcode:` completions for the composers, Slack-style (`:joy` → 😂).
*
* Draws from the same bundled emojibase-data the reaction picker uses (served
* at ./emojibase by the `hermes:emojibase-assets` vite plugin — offline, no
* CDN). The index lazy-loads on the first `:` trigger, then every query is
* answered from memory, so `isCached` skips the debounce and loading state
* after that first load.
*
* A pick inserts the emoji CHARACTER as plain text — not a chip. Directive
* chips exist to carry machine-readable references the backend resolves
* (@file:, /skill); a picked emoji is just text, so it rides the formatter's
* `rawText` path and lands inline.
*/
interface EmojiEntry {
emoji: string
/** Primary shortcode, e.g. "joy". */
code: string
/** Every shortcode, tag, and label that should match a search. */
haystack: string[]
}
let indexPromise: Promise<EmojiEntry[]> | null = null
let indexLoaded = false
async function loadIndex(): Promise<EmojiEntry[]> {
const [dataRes, codesRes] = await Promise.all([
fetch('./emojibase/en/data.json'),
fetch('./emojibase/en/shortcodes/emojibase.json')
])
const data: { emoji: string; hexcode: string; label: string; tags?: string[] }[] = await dataRes.json()
const codes: Record<string, string | string[]> = await codesRes.json()
const entries: EmojiEntry[] = []
for (const item of data) {
const raw = codes[item.hexcode]
if (!raw) {
continue
}
const shortcodes = Array.isArray(raw) ? raw : [raw]
entries.push({
emoji: item.emoji,
code: shortcodes[0],
haystack: [...shortcodes, ...(item.tags ?? []), item.label.toLowerCase()]
})
}
indexLoaded = true
return entries
}
/** Prefix matches on shortcodes rank first, then tag/label substring hits. */
async function searchEmoji(query: string, limit = 8): Promise<EmojiEntry[]> {
const index = await (indexPromise ??= loadIndex())
const q = query.toLowerCase()
const prefix: EmojiEntry[] = []
const loose: EmojiEntry[] = []
for (const entry of index) {
if (entry.code.startsWith(q) || entry.haystack.some(h => h.startsWith(q))) {
prefix.push(entry)
} else if (entry.haystack.some(h => h.includes(q))) {
loose.push(entry)
}
if (prefix.length >= limit) {
break
}
}
return [...prefix, ...loose].slice(0, limit)
}
export function useEmojiCompletions() {
const fetcher = useCallback(async (query: string): Promise<CompletionPayload> => {
const entries = await searchEmoji(query)
return {
query,
items: entries.map(entry => ({
text: entry.emoji,
display: `${entry.emoji} :${entry.code}:`,
meta: ''
}))
}
}, [])
const toItem = useCallback(
(entry: CompletionEntry, index: number) => ({
id: `${entry.text}|${index}`,
type: 'emoji',
label: typeof entry.display === 'string' ? entry.display : entry.text,
metadata: {
display: typeof entry.display === 'string' ? entry.display : entry.text,
// The formatter's serialize() returns rawText verbatim → the emoji
// character lands as plain inline text, no chip.
rawText: entry.text,
meta: '',
group: '',
action: ''
}
}),
[]
)
return useLiveCompletionAdapter({
enabled: true,
fetcher,
isCached: () => indexLoaded,
toItem
})
}
@@ -49,15 +49,7 @@ function gestureTargetOk(target: EventTarget | null) {
return false
}
// `composer-no-drag`: chrome that lives inside the composer root but isn't
// part of the draggable frame — the floating pill strips. The pills are
// `button`s and already excluded, but the strip's own box (the gaps between
// pills) isn't, so without this a press landing between two badges still
// drags. The strips are `w-fit`, so this costs the grab band only the width
// of the badges themselves.
return !target.closest(
'button, a, input, textarea, select, [role="menuitem"], [data-radix-popper-content-wrapper], [data-slot="composer-no-drag"]'
)
return !target.closest('button, a, input, textarea, select, [role="menuitem"], [data-radix-popper-content-wrapper]')
}
/** Floating composer's 5px outer frame — grab here to drag without long-press. */
@@ -0,0 +1,258 @@
import { act, cleanup, renderHook, waitFor } from '@testing-library/react'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { $voicePlayback } from '@/store/voice-playback'
import { useVoiceConversation } from './use-voice-conversation'
const mocks = vi.hoisted(() => {
let deferStreamStart = false
let onSilence: null | (() => void) = null
let resolveStreamStart: null | (() => void) = null
let resolveSpeech: null | ((outcome: 'done' | 'fallback') => void) = null
let streamAvailable = true
const stopVoicePlayback = vi.fn(() => {
const current = $voicePlayback.get()
$voicePlayback.set({ ...current, sequence: current.sequence + 1, status: 'idle' })
})
const playSpeechText = vi.fn(() => {
stopVoicePlayback()
return Promise.resolve(true)
})
const handle = {
cancel: vi.fn(),
start: vi.fn(async (options: { onSilence: () => void }) => {
onSilence = options.onSilence
}),
stop: vi.fn(async () => ({
audio: new Blob(['voice'], { type: 'audio/webm' }),
heardSpeech: true
}))
}
return {
continueStreamStart() {
resolveStreamStart?.()
resolveStreamStart = null
},
deferStreamStart() {
deferStreamStart = true
},
finishSpeech(outcome: 'done' | 'fallback') {
resolveSpeech?.(outcome)
},
handle,
playSpeechText,
resetSpeechMocks() {
deferStreamStart = false
resolveStreamStart = null
resolveSpeech = null
streamAvailable = true
},
startSpeechStream: vi.fn(async () => {
if (deferStreamStart) {
await new Promise<void>(resolve => {
resolveStreamStart = resolve
})
}
if (!streamAvailable) {
return null
}
const current = $voicePlayback.get()
$voicePlayback.set({ ...current, sequence: current.sequence + 1, status: 'preparing' })
return {
append: vi.fn(),
done: new Promise<'done' | 'fallback'>(resolve => {
resolveSpeech = resolve
}),
finish: vi.fn()
}
}),
stopVoicePlayback,
triggerSilence() {
onSilence?.()
},
useFallbackSpeech() {
streamAvailable = false
}
}
})
vi.mock('./use-mic-recorder', () => ({
useMicRecorder: () => ({ handle: mocks.handle, level: 0 })
}))
vi.mock('@/lib/voice-barge-in', () => ({
monitorSpeechDuringPlayback: () => vi.fn()
}))
vi.mock('@/lib/voice-playback', () => ({
markVoicePlaybackInterrupted: vi.fn(),
playSpeechText: mocks.playSpeechText,
startSpeechStream: mocks.startSpeechStream,
stopVoicePlayback: mocks.stopVoicePlayback
}))
vi.mock('@/lib/thinking-sound', () => ({
startThinkingSound: vi.fn(),
stopThinkingSound: vi.fn()
}))
vi.mock('@/store/notifications', () => ({
notify: vi.fn(),
notifyError: vi.fn()
}))
vi.mock('@/i18n', () => ({
useI18n: () => ({
t: {
notifications: {
voice: {
configureSpeechToText: '',
couldNotStartSession: '',
microphoneFailed: '',
playbackFailed: '',
transcriptionFailed: '',
unavailable: ''
}
}
}
})
}))
function renderRearmConversation(responseId: string, responseText: string) {
let response: null | { id: string; pending: boolean; text: string } = null
return renderHook(
({ enabled }) =>
useVoiceConversation({
busy: false,
consumePendingResponse: vi.fn(),
enabled,
onSubmit: async () => {
response = { id: responseId, pending: false, text: responseText }
},
onTranscribeAudio: async () => 'Hello',
pendingResponse: () => response
}),
{ initialProps: { enabled: false } }
)
}
async function beginReply(hook: ReturnType<typeof renderRearmConversation>) {
hook.rerender({ enabled: true })
await waitFor(() => expect(mocks.handle.start).toHaveBeenCalledTimes(1))
await act(async () => {
mocks.triggerSilence()
})
}
describe('useVoiceConversation playback rearm', () => {
afterEach(() => {
cleanup()
vi.clearAllMocks()
mocks.resetSpeechMocks()
$voicePlayback.set({
audioElement: null,
messageId: null,
sequence: 0,
source: null,
status: 'idle'
})
})
it('re-arms the microphone after normal streaming playback completes', async () => {
$voicePlayback.set({
audioElement: null,
messageId: null,
sequence: 7,
source: null,
status: 'idle'
})
const hook = renderRearmConversation('reply-1', 'Hello back')
await beginReply(hook)
await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled())
expect($voicePlayback.get().sequence).toBeGreaterThan(7)
await act(async () => {
mocks.finishSpeech('done')
})
await waitFor(() => expect(mocks.handle.start).toHaveBeenCalledTimes(2))
expect(hook.result.current.status).toBe('listening')
})
it('honors Stop while streaming playback is still preparing', async () => {
mocks.deferStreamStart()
const hook = renderRearmConversation('reply-preparing', 'Do not play this')
await beginReply(hook)
await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled())
mocks.stopVoicePlayback()
await act(async () => {
mocks.continueStreamStart()
})
await waitFor(() => expect(hook.result.current.status).toBe('idle'))
expect(mocks.stopVoicePlayback).toHaveBeenCalledTimes(2)
expect(mocks.handle.start).toHaveBeenCalledTimes(1)
})
it('does not start fallback playback after Stop during stream discovery', async () => {
mocks.deferStreamStart()
mocks.useFallbackSpeech()
const hook = renderRearmConversation('reply-no-stream', 'Do not fall back')
await beginReply(hook)
await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled())
mocks.stopVoicePlayback()
await act(async () => {
mocks.continueStreamStart()
})
await waitFor(() => expect(hook.result.current.status).toBe('idle'))
expect(mocks.playSpeechText).not.toHaveBeenCalled()
expect(mocks.handle.start).toHaveBeenCalledTimes(1)
})
it('does not re-arm after an external Stop during streaming playback', async () => {
const hook = renderRearmConversation('reply-stopped', 'Playing now')
await beginReply(hook)
await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled())
mocks.stopVoicePlayback()
await act(async () => {
mocks.finishSpeech('done')
})
await waitFor(() => expect(hook.result.current.status).toBe('idle'))
expect(mocks.handle.start).toHaveBeenCalledTimes(1)
})
it('re-arms the microphone after normal fallback playback completes', async () => {
mocks.useFallbackSpeech()
const hook = renderRearmConversation('reply-fallback', 'Fallback reply')
await beginReply(hook)
await waitFor(() =>
expect(mocks.playSpeechText).toHaveBeenCalledWith('Fallback reply', {
source: 'voice-conversation'
})
)
await waitFor(() => expect(mocks.handle.start).toHaveBeenCalledTimes(2))
expect(hook.result.current.status).toBe('listening')
})
})
@@ -0,0 +1,266 @@
import { act, cleanup, renderHook, waitFor } from '@testing-library/react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type { BargeMonitorCallbacks } from '@/lib/voice-barge-in'
import type { MicRecording } from './use-mic-recorder'
import { useVoiceConversation } from './use-voice-conversation'
// The full-duplex contract: the barge monitor is live across the WHOLE agent
// turn — generation (thinking) and playback (speaking) — so speaking over the
// model interrupts it mid-generation instead of the mic being deaf until TTS
// starts (the Windows report: interruption "never works" because the deaf
// window covered generation, and playback bleed made the old monitor's
// trigger unreachable).
const monitorCalls: BargeMonitorCallbacks[] = []
const stopMonitor = vi.fn()
vi.mock('@/lib/voice-barge-in', () => ({
monitorSpeechDuringPlayback: (callbacks: BargeMonitorCallbacks) => {
monitorCalls.push(callbacks)
return stopMonitor
}
}))
const markVoicePlaybackInterrupted = vi.fn()
const stopVoicePlayback = vi.fn()
vi.mock('@/lib/voice-playback', () => ({
markVoicePlaybackInterrupted: () => markVoicePlaybackInterrupted(),
playSpeechText: vi.fn(async () => true),
startSpeechStream: vi.fn(async () => null),
stopVoicePlayback: () => stopVoicePlayback()
}))
vi.mock('@/lib/thinking-sound', () => ({
startThinkingSound: vi.fn(),
stopThinkingSound: vi.fn()
}))
const micHandle = {
cancel: vi.fn(),
start: vi.fn(async () => undefined),
stop: vi.fn<() => Promise<MicRecording | null>>(async () => null)
}
vi.mock('./use-mic-recorder', () => ({
useMicRecorder: () => ({ handle: micHandle, level: 0, recording: false })
}))
vi.mock('@/i18n', () => ({
useI18n: () => ({
t: {
notifications: {
voice: {
configureSpeechToText: 'configure STT',
couldNotStartSession: 'could not start',
microphoneFailed: 'mic failed',
playbackFailed: 'playback failed',
transcriptionFailed: 'transcription failed',
unavailable: 'unavailable'
}
}
}
})
}))
vi.mock('@/store/notifications', () => ({
notify: vi.fn(),
notifyError: vi.fn()
}))
interface HookProps {
busy: boolean
}
function renderConversation(overrides: { onInterrupt?: () => void; transcript?: string } = {}) {
const onInterrupt = overrides.onInterrupt ?? vi.fn()
// Mirrors the real app: submitting a turn makes the agent busy.
const onBusyChange: { current: (busy: boolean) => void } = { current: () => undefined }
const onSubmit = vi.fn(async () => {
onBusyChange.current(true)
})
const onStopWord = vi.fn()
// First transcription is the turn that starts the conversation; subsequent
// ones are barge captures (the overridable transcript).
let transcriptions = 0
const onTranscribeAudio = vi.fn(async () =>
transcriptions++ === 0 ? 'kick off the task' : (overrides.transcript ?? 'and another thing')
)
const hook = renderHook(
({ busy }: HookProps) =>
useVoiceConversation({
busy,
consumePendingResponse: vi.fn(),
enabled: true,
onInterrupt,
onStopWord,
onSubmit,
onTranscribeAudio,
pendingResponse: () => null
}),
{ initialProps: { busy: false } }
)
onBusyChange.current = busy => hook.rerender({ busy })
return { hook, onInterrupt, onStopWord, onSubmit, onTranscribeAudio }
}
/** Drive the hook into the generation phase (turn submitted, model working). */
async function enterThinking(hook: ReturnType<typeof renderConversation>['hook']) {
await act(async () => {
await hook.result.current.start()
})
await waitFor(() => expect(hook.result.current.status).toBe('listening'))
micHandle.stop.mockResolvedValueOnce({
audio: new Blob(['q'], { type: 'audio/webm' }),
durationMs: 900,
heardSpeech: true
})
await act(async () => {
hook.result.current.stopTurn()
})
await waitFor(() => expect(hook.result.current.status).toBe('thinking'))
}
describe('useVoiceConversation full-duplex barge-in', () => {
beforeEach(() => {
monitorCalls.length = 0
vi.clearAllMocks()
micHandle.start.mockResolvedValue(undefined)
micHandle.stop.mockResolvedValue(null)
})
afterEach(cleanup)
it('arms the barge monitor during generation (before any reply audio exists)', async () => {
const { hook } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(hook.result.current.status).toBe('thinking'))
// busy=true + thinking → the full-duplex monitor must be live.
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
})
it('interrupts the in-flight turn when speech trips mid-generation', async () => {
const { hook, onInterrupt } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
act(() => {
monitorCalls.at(-1)?.onSpeech()
})
expect(onInterrupt).toHaveBeenCalledTimes(1)
expect(markVoicePlaybackInterrupted).toHaveBeenCalled()
expect(stopVoicePlayback).toHaveBeenCalled()
})
it('submits the captured interruption once the interrupt settles (busy clears)', async () => {
const { hook, onSubmit } = renderConversation({ transcript: 'no, do it differently' })
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
const monitor = monitorCalls.at(-1)
act(() => {
monitor?.onSpeech()
})
// Interrupt lands → the turn ends → busy flips false.
hook.rerender({ busy: false })
await act(async () => {
monitor?.onUtterance?.(new Blob(['x'], { type: 'audio/webm' }))
})
await waitFor(() => expect(onSubmit).toHaveBeenCalledWith('no, do it differently'))
})
it('does not interrupt when speech trips during playback (turn already done)', async () => {
const { hook, onInterrupt } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
// Turn finished; playback phase.
hook.rerender({ busy: false })
act(() => {
monitorCalls.at(-1)?.onSpeech()
})
expect(onInterrupt).not.toHaveBeenCalled()
expect(stopVoicePlayback).toHaveBeenCalled()
})
it('a spoken stop command in the barge capture ends the conversation instead of submitting', async () => {
const { hook, onStopWord, onSubmit } = renderConversation({ transcript: 'stop' })
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
const monitor = monitorCalls.at(-1)
act(() => {
monitor?.onSpeech()
})
hook.rerender({ busy: false })
await act(async () => {
monitor?.onUtterance?.(new Blob(['s'], { type: 'audio/webm' }))
})
await waitFor(() => expect(onStopWord).toHaveBeenCalledTimes(1))
// Only the kickoff turn was submitted — the "stop" capture never was.
expect(onSubmit).toHaveBeenCalledTimes(1)
expect(onSubmit).not.toHaveBeenCalledWith('stop')
})
it('re-arms a single monitor per turn (idempotent ensure)', async () => {
const { hook } = renderConversation()
await act(async () => {
await hook.result.current.start()
})
await enterThinking(hook)
await waitFor(() => expect(monitorCalls.length).toBeGreaterThan(0))
const armed = monitorCalls.length
// Effect re-runs (busy toggles, status changes) must not open more mics.
hook.rerender({ busy: true })
hook.rerender({ busy: true })
expect(monitorCalls.length).toBe(armed)
})
})
@@ -28,6 +28,9 @@ interface VoiceConversationOptions {
busy: boolean
enabled: boolean
onFatalError?: () => void
/** Interrupt the in-flight agent turn (the same seam as the Stop button).
* Fired when the user speaks while the model is still generating. */
onInterrupt?: () => Promise<void> | void
onStopWord?: () => void
onSubmit: (text: string) => Promise<void> | void
onTranscribeAudio?: (audio: Blob) => Promise<string>
@@ -38,10 +41,15 @@ interface VoiceConversationOptions {
beforeMicOpen?: () => Promise<void> | void
}
/** How long a barge-triggered interrupt may take to settle before we submit
* the captured utterance anyway. */
const INTERRUPT_SETTLE_TIMEOUT_MS = 5_000
export function useVoiceConversation({
busy,
enabled,
onFatalError,
onInterrupt,
onStopWord,
onSubmit,
onTranscribeAudio,
@@ -63,6 +71,7 @@ export function useVoiceConversation({
const speechSessionRef = useRef<null | SpeechStreamSession>(null)
const stopBargeMonitorRef = useRef<(() => void) | null>(null)
const bargeCapturePendingRef = useRef(false)
const bargedRef = useRef(false)
const speechStartSequenceRef = useRef(0)
const enabledRef = useRef(enabled)
const mutedRef = useRef(muted)
@@ -70,6 +79,12 @@ export function useVoiceConversation({
const statusRef = useRef<ConversationStatus>('idle')
const wasEnabledRef = useRef(enabled)
const onStopWordRef = useRef(onStopWord)
const onInterruptRef = useRef(onInterrupt)
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
onInterruptRef.current = onInterrupt
}, [onInterrupt])
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
@@ -114,6 +129,7 @@ export function useVoiceConversation({
stopBargeMonitorRef.current?.()
stopBargeMonitorRef.current = null
bargeCapturePendingRef.current = false
bargedRef.current = false
speechSessionRef.current = null
responseIdRef.current = null
spokenSourceLengthRef.current = 0
@@ -246,7 +262,7 @@ export function useVoiceConversation({
}, [handle, handleTurn, onFatalError, voiceCopy.couldNotStartSession, voiceCopy.microphoneFailed])
const settleAfterSpeech = useCallback(
(barged: boolean) => {
(barged: boolean, stoppedDuringSetup = false) => {
if (barged || !awaitingSpokenResponseRef.current) {
awaitingSpokenResponseRef.current = false
consumePendingResponse()
@@ -270,7 +286,9 @@ export function useVoiceConversation({
// voice-playback sequence has advanced past what we captured at speech
// start — don't auto-start the next sentence, the user chose to stop.
const stoppedByUser =
speechStartSequenceRef.current > 0 && $voicePlayback.get().sequence > speechStartSequenceRef.current
stoppedDuringSetup ||
(speechStartSequenceRef.current > 0 &&
$voicePlayback.get().sequence > speechStartSequenceRef.current)
speechStartSequenceRef.current = 0
@@ -315,6 +333,25 @@ export function useVoiceConversation({
return
}
// A spoken stop command while barging means "stop everything" — the
// turn/playback was already cut at trip time; now end the conversation
// instead of submitting "stop" as a new prompt.
if (isVoiceStopCommand(transcript)) {
dropSpeechSession()
setStatus('idle')
onStopWordRef.current?.()
return
}
// A generation-phase barge interrupted the in-flight turn; the submit
// path refuses while `busy`, so wait for the interrupt to settle.
const deadline = Date.now() + INTERRUPT_SETTLE_TIMEOUT_MS
while (busyRef.current && Date.now() < deadline) {
await new Promise(resolve => window.setTimeout(resolve, 100))
}
awaitingSpokenResponseRef.current = true
dropSpeechSession()
consumePendingResponse()
@@ -328,24 +365,46 @@ export function useVoiceConversation({
[consumePendingResponse, onSubmit, onTranscribeAudio, voiceCopy.transcriptionFailed]
)
/** Barge-in monitor wiring shared by the live and fallback speech paths. */
const openBargeMonitor = useCallback(
(onBarge: () => void) =>
monitorSpeechDuringPlayback({
onSpeech: () => {
bargeCapturePendingRef.current = true
onBarge()
markVoicePlaybackInterrupted()
stopVoicePlayback()
},
onUtterance: audio => {
bargeCapturePendingRef.current = false
stopBargeMonitorRef.current = null
void submitCapturedUtterance(audio)
/**
* Full-duplex barge-in monitor for the WHOLE agent turn: armed at submit,
* live through generation (thinking) AND playback (speaking).
*
* - generation phase (`busy`): speech interrupts the in-flight turn via
* `onInterrupt` — the same seam as the Stop button — and cuts any TTS that
* managed to start, so the stale reply never speaks.
* - playback phase: speech cuts playback and the captured interruption is
* transcribed and submitted as the next turn.
*
* Idempotent — one monitor owns the mic per turn; re-arming while one is
* live is a no-op (the live/fallback speech paths and the turn-drive effect
* all call this).
*/
const ensureBargeMonitor = useCallback(() => {
if (stopBargeMonitorRef.current) {
return
}
stopBargeMonitorRef.current = monitorSpeechDuringPlayback({
isPlaying: () => $voicePlayback.get().status === 'speaking',
onSpeech: () => {
bargeCapturePendingRef.current = true
bargedRef.current = true
markVoicePlaybackInterrupted()
stopVoicePlayback()
if (busyRef.current) {
// Mid-generation: stop the in-flight turn so the captured utterance
// becomes the next one instead of queueing behind a stale reply.
void onInterruptRef.current?.()
}
}),
[submitCapturedUtterance]
)
},
onUtterance: audio => {
bargeCapturePendingRef.current = false
stopBargeMonitorRef.current = null
void submitCapturedUtterance(audio)
}
})
}, [submitCapturedUtterance])
/** Push any new reply text into the live session; finish when complete. */
const feedSpeechSession = useCallback(
@@ -397,28 +456,29 @@ export function useVoiceConversation({
return
}
let barged = false
stopBargeMonitorRef.current?.()
stopBargeMonitorRef.current = openBargeMonitor(() => {
barged = true
})
// The full-duplex monitor is normally already live (armed at submit);
// this is a safety net for read-aloud-style entries into the loop.
ensureBargeMonitor()
const playback = playSpeechText(response.text, { source: 'voice-conversation' })
// playSpeechText performs its normal cleanup synchronously before
// returning. Capture the sequence after that internal increment so
// only a later, external stop suppresses the next listen cycle.
speechStartSequenceRef.current = $voicePlayback.get().sequence
void playSpeechText(response.text, { source: 'voice-conversation' })
void playback
.catch(error => notifyError(error, voiceCopy.playbackFailed))
.finally(() => {
if (responseIdRef.current === responseId) {
awaitingSpokenResponseRef.current = false
settleAfterSpeech(barged)
settleAfterSpeech(bargedRef.current)
}
})
}
poll()
},
[openBargeMonitor, pendingResponse, settleAfterSpeech, voiceCopy.playbackFailed]
[ensureBargeMonitor, pendingResponse, settleAfterSpeech, voiceCopy.playbackFailed]
)
/**
@@ -428,20 +488,17 @@ export function useVoiceConversation({
*/
const openLiveSpeech = useCallback(
(responseId: string) => {
const sequenceBeforeStart = $voicePlayback.get().sequence
responseIdRef.current = responseId
spokenSourceLengthRef.current = 0
speechStartSequenceRef.current = $voicePlayback.get().sequence
setStatus('speaking')
let barged = false
// VAD barge-in: the user talking over the reply cuts playback, drops
// the not-yet-spoken remainder, AND keeps capturing — the interruption
// is transcribed from its first syllable instead of losing the opening
// words to a mic re-open.
stopBargeMonitorRef.current = openBargeMonitor(() => {
barged = true
})
// words to a mic re-open. Usually already live (armed at submit).
ensureBargeMonitor()
void (async () => {
const session = await startSpeechStream({ source: 'voice-conversation' })
@@ -456,6 +513,16 @@ export function useVoiceConversation({
}
if (!session) {
// Stream discovery can also fail after an explicit Stop landed
// during its async URL lookup. In that case, do not turn the stopped
// live attempt into fresh fallback playback.
if ($voicePlayback.get().sequence > sequenceBeforeStart) {
awaitingSpokenResponseRef.current = false
settleAfterSpeech(false, true)
return
}
// No streaming backend/provider: speak the whole reply once it lands.
speechSessionRef.current = null
awaitFallbackSpeech(responseId)
@@ -463,8 +530,24 @@ export function useVoiceConversation({
return
}
// startSpeechStream calls stopVoicePlayback once after its async URL
// lookup. A second sequence bump means the user pressed Stop while
// setup was still pending. Do not absorb that explicit stop into the
// post-start baseline or allow the new session to play.
const sequenceAfterStart = $voicePlayback.get().sequence
const stoppedDuringStart = sequenceAfterStart > sequenceBeforeStart + 1
speechStartSequenceRef.current = sequenceAfterStart
speechSessionRef.current = session
if (stoppedDuringStart) {
stopVoicePlayback()
awaitingSpokenResponseRef.current = false
settleAfterSpeech(false, true)
return
}
// Timer-driven feed: reply text flows into the session at delta rate
// regardless of React render cadence.
const feedTimer = window.setInterval(() => feedSpeechSession(responseId), 150)
@@ -484,10 +567,10 @@ export function useVoiceConversation({
}
awaitingSpokenResponseRef.current = false
settleAfterSpeech(barged)
settleAfterSpeech(bargedRef.current)
})()
},
[awaitFallbackSpeech, feedSpeechSession, openBargeMonitor, settleAfterSpeech]
[awaitFallbackSpeech, ensureBargeMonitor, feedSpeechSession, settleAfterSpeech]
)
const start = useCallback(async () => {
@@ -601,6 +684,13 @@ export function useVoiceConversation({
}
if (awaitingSpokenResponseRef.current && status !== 'speaking') {
// Generation phase: the turn is in flight but no reply audio exists
// yet. Keep the mic live so speech can interrupt the model mid-
// generation (full-duplex) instead of going deaf until playback.
if (status === 'thinking' && (busy || bargeCapturePendingRef.current)) {
ensureBargeMonitor()
}
const response = pendingResponse()
if (response) {
@@ -609,8 +699,9 @@ export function useVoiceConversation({
return
}
if (!busy && status === 'thinking') {
// Turn finished without any speakable reply (tool-only, error).
if (!busy && status === 'thinking' && !bargeCapturePendingRef.current) {
// Turn finished without any speakable reply (tool-only, error). A
// live barge capture owns the loop instead — it submits or resumes.
awaitingSpokenResponseRef.current = false
dropSpeechSession()
pendingStartRef.current = true
@@ -627,7 +718,7 @@ export function useVoiceConversation({
if (pendingStartRef.current) {
void startListening()
}
}, [busy, enabled, muted, openLiveSpeech, pendingResponse, startListening, status])
}, [busy, enabled, muted, ensureBargeMonitor, openLiveSpeech, pendingResponse, startListening, status])
// eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment)
useEffect(() => {
+202 -153
View File
@@ -49,12 +49,15 @@ import { useComposerTrigger } from './hooks/use-composer-trigger'
import { useComposerUndo } from './hooks/use-composer-undo'
import { useComposerUrlDialog } from './hooks/use-composer-url-dialog'
import { useComposerVoice } from './hooks/use-composer-voice'
import { useEmojiCompletions } from './hooks/use-emoji-completions'
import { useComposerMicroActions } from './hooks/use-micro-actions'
import { useSlashCompletions } from './hooks/use-slash-completions'
import { useSessionStatusPresence } from './hooks/use-status-presence'
import { ActionBadges } from './micro-actions'
import { chipTypedPathOnSpace, pathifyRefs } from './path-refs'
import { QueuePanel } from './queue-panel'
import {
COMPOSER_PLACEHOLDER_CLASS,
composerPlainText,
deleteChipBeforeCaret,
deleteSelectionInEditor,
@@ -65,7 +68,7 @@ import {
import { useComposerScope } from './scope'
import { ComposerStatusStack } from './status-stack'
import { CodingStatusRow } from './status-stack/coding-row'
import { extractClipboardImageBlobs } from './text-utils'
import { extractClipboardImageBlobs, openDirectiveScope } from './text-utils'
import { ComposerTriggerPopover } from './trigger-popover'
import type { ChatBarProps } from './types'
import { isRedoShortcut, isUndoShortcut } from './undo-history'
@@ -161,6 +164,9 @@ export function ChatBar({
useComposerMicroActions(statusSessionId, busy)
const composerRef = useRef<HTMLFormElement | null>(null)
// The dock wraps the strips + status stack + composer; the thread's bottom
// clearance measures this, while the pop-out drag still tracks the composer.
const composerDockRef = useRef<HTMLDivElement | null>(null)
const composerSurfaceRef = useRef<HTMLDivElement | null>(null)
// Pop-out engine: docked↔floating state, dock/float/toggle, drag gestures, and
@@ -184,6 +190,7 @@ export function ChatBar({
const { availableThemes, themeName } = useTheme()
const at = useAtCompletions({ gateway: gateway ?? null, sessionId: sessionId ?? null, cwd: cwd ?? null })
const slash = useSlashCompletions({ activeSkin: themeName, gateway: gateway ?? null, skinThemes: availableThemes })
const emoji = useEmojiCompletions()
const { t } = useI18n()
const gatewayState = useStore($gatewayState)
@@ -277,7 +284,14 @@ export function ChatBar({
return onCancel()
}, [activeQueueSessionKeyRef, onCancel])
const { compactPill, stacked } = useComposerMetrics({ composerRef, composerSurfaceRef, editorRef, poppedOut })
const { compactPill, stacked } = useComposerMetrics({
composerDockRef,
composerRef,
composerSurfaceRef,
editorRef,
poppedOut
})
const hasComposerPayload = hasText || attachments.length > 0
const canSubmit = busy || hasComposerPayload
@@ -346,7 +360,7 @@ export function ChatBar({
triggerItems,
triggerKeyConsumedRef,
triggerLoading
} = useComposerTrigger({ at, draftRef, editorRef, requestMainFocus, setComposerText, slash })
} = useComposerTrigger({ at, draftRef, editorRef, emoji, recordUndoPoint, requestMainFocus, setComposerText, slash })
// Pull the live contentEditable text into draftRef + the AUI composer state
// (which drives `hasComposerPayload` → the send button). Shared by the input
@@ -478,8 +492,13 @@ export function ChatBar({
// Links in the paste land as `@url:` chips rather than a wall of URL text —
// the same reference the "Add URL" dialog inserts, parsed in place so a link
// mid-sentence keeps its position. Bare `@path` tokens promote the same way.
// A paste into an open `@url:`/`@file:` scope CONSUMES that scope instead of
// stacking on it — the scope is the browse mode the user is pasting into,
// not text they typed and want to keep (`@url:@url:\`https://…\``).
const scope = openDirectiveScope(event.currentTarget)
recordUndoPoint()
insertComposerContentsAtCaret(event.currentTarget, pathifyRefs(linkifyUrls(pastedText)))
insertComposerContentsAtCaret(event.currentTarget, pathifyRefs(linkifyUrls(pastedText)), scope)
scheduleFlushEditorToDraft(event.currentTarget)
}
@@ -569,6 +588,17 @@ export function ChatBar({
return
}
// The popover is open but its items are still in flight (debounce + RPC).
// Tab must not fall through to the browser — it would move focus out of
// the composer mid-completion, which reads as the popover "eating" the
// keypress. Swallow it; the refresh lands with the items.
if (trigger && triggerLoading && triggerItems.length === 0 && event.key === 'Tab') {
event.preventDefault()
triggerKeyConsumedRef.current = true
return
}
if (trigger && triggerItems.length > 0) {
if (event.key === 'ArrowDown') {
event.preventDefault()
@@ -856,6 +886,8 @@ export function ChatBar({
focusInput,
insertText,
maxRecordingSeconds,
// Voice barge-in mid-generation halts the run like the Stop button.
onInterrupt: haltRun,
onSubmit,
onTranscribeAudio,
sessionId,
@@ -915,7 +947,7 @@ export function ChatBar({
autoCorrect="off"
className={cn(
'min-h-[1.625rem] min-h-(--composer-input-min-height) max-h-(--composer-input-max-height) cursor-text overflow-y-auto whitespace-pre-wrap break-words [overflow-wrap:anywhere] bg-transparent pb-1 pr-1 pt-1 leading-normal text-foreground outline-none disabled:cursor-not-allowed',
'empty:before:content-[attr(data-placeholder)] empty:before:text-muted-foreground/60',
COMPOSER_PLACEHOLDER_CLASS,
'**:data-ref-text:cursor-default',
stacked && 'pl-3',
stacked ? 'w-full' : 'min-w-(--composer-input-inline-min-width) flex-1'
@@ -1005,36 +1037,27 @@ export function ChatBar({
/>
)}
<ComposerPrimitive.Unstable_TriggerPopoverRoot>
<ComposerPrimitive.Root
{/* Dock column: owns the composer's POSITION and stacks, bottom-up,
[micro actions] · [status stack] · [composer] · [underside].
Anchored at the bottom, so in-flow children grow upward and still
overlay the thread — no absolute lane needed.
The strips are siblings of the composer, not children: the pop-out
drag region is `absolute inset-0` INSIDE the composer, so anything
rendered in there is inside the grab area by construction. Keeping
them out here is what makes that impossible rather than excluded. */}
<div
className={cn(
'group/composer z-30 overflow-visible rounded-2xl',
poppedOut
? // Floating: the composer (with its own border) floats with an even
// 5px transparent grab margin around it — drag that to move it.
'fixed w-[var(--composer-popout-width)] max-w-[calc(100vw-1.5rem)] bg-transparent p-[5px]'
: 'absolute bottom-0 left-1/2 w-[min(var(--composer-width),calc(100%-2rem))] max-w-full -translate-x-1/2 pt-2 pb-[var(--composer-shell-pad-block-end)]',
dragging && 'cursor-grabbing select-none touch-none'
'z-30 flex flex-col',
poppedOut ? 'fixed max-w-[calc(100vw-1.5rem)]' : 'absolute bottom-0 left-1/2 max-w-full -translate-x-1/2'
)}
data-drag-active={dragActive ? '' : undefined}
data-popped-out={poppedOut ? '' : undefined}
data-slot="composer-root"
data-status-stack={statusStackVisible ? '' : undefined}
data-slot="composer-dock"
data-thread-scrolled-up={scrolledUp ? '' : undefined}
onDragEnter={handleDragEnter}
onDragLeave={handleDragLeave}
onDragOver={handleDragOver}
onDrop={handleDrop}
onPointerDown={popoutAllowed ? onComposerGesturePointerDown : undefined}
onSubmit={e => {
e.preventDefault()
if (composingRef.current) {
return
}
submitDraft()
}}
ref={composerRef}
// Measured for the thread's bottom clearance: the dock is the box
// that contains the strips, the status stack, AND the composer, so
// one measurement covers everything the thread must clear.
ref={composerDockRef}
style={
poppedOut
? {
@@ -1046,22 +1069,16 @@ export function ChatBar({
: undefined
}
>
{isHelpHint && <HelpHint />}
{trigger && !argStageEmpty && (
<ComposerTriggerPopover
activeIndex={triggerActive}
items={triggerItems}
kind={trigger.kind}
loading={triggerLoading}
onHover={setTriggerActive}
onPick={replaceTriggerWithChip}
/>
)}
{/* Aligned to the composer SURFACE, which sits inside the composer's
5px transparent grab margin — so both strips carry the same inset
and share one left edge with it. */}
<div className={cn(composerFloatingStrip, 'px-[5px] pb-1.5 empty:hidden')}>
<ActionBadges sessionId={statusSessionId} />
</div>
{/* Session-scoped status stack (todos, subagents, background tasks,
queue). Out of flow so it never inflates the composer's measured
height; it overlays the chat instead of pushing it, and publishes
its own --status-stack-measured-height so the thread's clearance
accounts for it. Collapses to nothing when every status is empty. */}
queue). An in-flow dock child: the dock is bottom-anchored, so it
grows upward over the thread and the dock's own measurement covers
it. Collapses to nothing when every status is empty. */}
<ComposerStatusStack
queue={
activeQueueSessionKey && queuedPrompts.length > 0 ? (
@@ -1091,134 +1108,166 @@ export function ChatBar({
}
sessionId={statusSessionId}
/>
{!poppedOut && (
<div
className="pointer-events-none absolute inset-0 rounded-[inherit]"
style={{ background: COMPOSER_FADE_BACKGROUND }}
/>
)}
{/* Drag region: covers the transparent grab margin around the surface.
<ComposerPrimitive.Root
className={cn(
'group/composer relative w-full overflow-visible rounded-2xl',
poppedOut && 'bg-transparent',
dragging && 'cursor-grabbing select-none touch-none'
)}
data-drag-active={dragActive ? '' : undefined}
data-popped-out={poppedOut ? '' : undefined}
data-slot="composer-root"
data-status-stack={statusStackVisible ? '' : undefined}
data-thread-scrolled-up={scrolledUp ? '' : undefined}
onDragEnter={handleDragEnter}
onDragLeave={handleDragLeave}
onDragOver={handleDragOver}
onDrop={handleDrop}
onPointerDown={popoutAllowed ? onComposerGesturePointerDown : undefined}
onSubmit={e => {
e.preventDefault()
if (composingRef.current) {
return
}
submitDraft()
}}
ref={composerRef}
>
{isHelpHint && <HelpHint />}
{trigger && !argStageEmpty && (
<ComposerTriggerPopover
activeIndex={triggerActive}
items={triggerItems}
kind={trigger.kind}
loading={triggerLoading}
onHover={setTriggerActive}
onPick={replaceTriggerWithChip}
scope={trigger.scope}
/>
)}
{!poppedOut && (
<div
className="pointer-events-none absolute inset-0 rounded-[inherit]"
style={{ background: COMPOSER_FADE_BACKGROUND }}
/>
)}
{/* Drag region: covers the transparent grab margin around the surface.
The surface sits on top (z-4) so only the exposed ring receives this
element's hover/cursor — grab cursor + a diagonal hatch (/////)
appear when you hover the draggable margin, never over the input.
The hatch pattern + opacity ladder live in styles.css. */}
{popoutAllowed && (
<div
aria-hidden
className={cn('pointer-events-auto absolute inset-0', dragging ? 'cursor-grabbing' : 'cursor-grab')}
data-dragging={dragging ? '' : undefined}
data-slot="composer-drag-region"
onDoubleClick={event => {
// The pill strips paint above this region; a double-click that
// lands on one must not float the composer. onPointerDown goes
// through gestureTargetOk, but this handler doesn't.
if (!(event.target as Element).closest('[data-slot="composer-no-drag"]')) {
handleComposerToggle()
}
}}
/>
)}
<div className="relative w-full rounded-[inherit]">
<div
className={cn(
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
COMPOSER_DROP_FADE_CLASS,
dragActive && COMPOSER_DROP_ACTIVE_CLASS
)}
data-slot="composer-surface"
ref={composerSurfaceRef}
>
{popoutAllowed && (
<div
aria-hidden
className={cn(
'pointer-events-none absolute inset-0 -z-10 rounded-[inherit]',
composerFill,
composerSurfaceGlass
)}
/>
<CodingStatusRow
onBranchOff={handleBranchOff}
onConvertBranch={handleConvertBranch}
onListBranches={handleListBranches}
onOpen={toggleReview}
onOpenWorktree={openInWorktree}
onSwitchBranch={handleSwitchBranch}
repoPath={cwd}
className={cn('pointer-events-auto absolute inset-0', dragging ? 'cursor-grabbing' : 'cursor-grab')}
data-dragging={dragging ? '' : undefined}
data-slot="composer-drag-region"
onDoubleClick={handleComposerToggle}
/>
)}
<div className="relative w-full rounded-[inherit]">
<div
className={cn(
'relative z-1 flex min-h-0 w-full flex-col gap-(--composer-row-gap) overflow-hidden rounded-[inherit] px-(--composer-surface-pad-x) py-(--composer-surface-pad-y) transition-opacity duration-200 ease-out',
scrolledUp
? 'opacity-30 group-hover/composer:opacity-100 group-focus-within/composer-surface:opacity-100'
: 'opacity-100'
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
COMPOSER_DROP_FADE_CLASS,
dragActive && COMPOSER_DROP_ACTIVE_CLASS
)}
data-slot="composer-fade"
data-slot="composer-surface"
ref={composerSurfaceRef}
>
{/* Contribution seams: banners above, a row below, inline
additions beside the "+" menu and before the controls.
All four render nothing until something contributes. */}
<ContribSlot area={COMPOSER_AREAS.top} />
<VoiceActivity state={voiceActivityState} />
<VoicePlaybackActivity />
{queueEdit && editingQueuedPrompt && (
<div className="flex items-center justify-between gap-2 rounded-lg border border-[color-mix(in_srgb,var(--dt-composer-ring)_32%,transparent)] bg-accent/18 px-2 py-1">
<div className="min-w-0 text-[0.7rem] text-muted-foreground/88">
{t.composer.editingQueuedInComposer}
</div>
<div className="flex shrink-0 items-center gap-1">
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('cancel')}
type="button"
variant="ghost"
>
{t.common.cancel}
</Button>
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('save')}
type="button"
>
{t.common.save}
</Button>
</div>
</div>
)}
{attachments.length > 0 && <AttachmentList attachments={attachments} onRemove={onRemoveAttachment} />}
<div
aria-hidden
className={cn(
'pointer-events-none absolute inset-0 -z-10 rounded-[inherit]',
composerFill,
composerSurfaceGlass
)}
/>
<CodingStatusRow
onBranchOff={handleBranchOff}
onConvertBranch={handleConvertBranch}
onListBranches={handleListBranches}
// A tile's rail reviews ITS worktree: pin the pane's scope to
// this surface's cwd. Main keeps the classic follow-the-
// active-session scope (null).
onOpen={() => toggleReview(scope.target === 'main' ? null : (cwd ?? null))}
onOpenWorktree={openInWorktree}
onSwitchBranch={handleSwitchBranch}
repoPath={cwd}
/>
<div
className={cn(
'grid w-full',
stacked
? 'grid-cols-[auto_1fr] gap-(--composer-row-gap) [grid-template-areas:"input_input"_"menu_controls"]'
: 'grid-cols-[auto_1fr_auto] items-center gap-(--composer-control-gap) [grid-template-areas:"menu_input_controls"]'
'relative z-1 flex min-h-0 w-full flex-col gap-(--composer-row-gap) overflow-hidden rounded-[inherit] px-(--composer-surface-pad-x) py-(--composer-surface-pad-y) transition-opacity duration-200 ease-out',
scrolledUp
? 'opacity-30 group-hover/composer:opacity-100 group-focus-within/composer-surface:opacity-100'
: 'opacity-100'
)}
data-slot="composer-fade"
>
<div className="flex translate-y-[3px] items-start gap-(--composer-control-gap) self-start [grid-area:menu]">
{contextMenu}
<ContribSlot area={COMPOSER_AREAS.leading} />
</div>
<div className="min-w-0 [grid-area:input]">{input}</div>
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<ContribSlot area={COMPOSER_AREAS.actions} />
{controls}
{/* Contribution seams: banners above, a row below, inline
additions beside the "+" menu and before the controls.
All four render nothing until something contributes. */}
<ContribSlot area={COMPOSER_AREAS.top} />
<VoiceActivity state={voiceActivityState} />
<VoicePlaybackActivity />
{queueEdit && editingQueuedPrompt && (
<div className="flex items-center justify-between gap-2 rounded-lg border border-[color-mix(in_srgb,var(--dt-composer-ring)_32%,transparent)] bg-accent/18 px-2 py-1">
<div className="min-w-0 text-[0.7rem] text-muted-foreground/88">
{t.composer.editingQueuedInComposer}
</div>
<div className="flex shrink-0 items-center gap-1">
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('cancel')}
type="button"
variant="ghost"
>
{t.common.cancel}
</Button>
<Button
className="h-6 rounded-md px-2 text-[0.68rem]"
onClick={() => exitQueuedEdit('save')}
type="button"
>
{t.common.save}
</Button>
</div>
</div>
)}
{attachments.length > 0 && <AttachmentList attachments={attachments} onRemove={onRemoveAttachment} />}
<div
className={cn(
'grid w-full',
stacked
? 'grid-cols-[auto_1fr] gap-(--composer-row-gap) [grid-template-areas:"input_input"_"menu_controls"]'
: 'grid-cols-[auto_1fr_auto] items-center gap-(--composer-control-gap) [grid-template-areas:"menu_input_controls"]'
)}
>
<div className="flex translate-y-[3px] items-start gap-(--composer-control-gap) self-start [grid-area:menu]">
{contextMenu}
<ContribSlot area={COMPOSER_AREAS.leading} />
</div>
<div className="min-w-0 [grid-area:input]">{input}</div>
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<ContribSlot area={COMPOSER_AREAS.actions} />
{controls}
</div>
</div>
<ContribSlot area={COMPOSER_AREAS.bottom} />
</div>
<ContribSlot area={COMPOSER_AREAS.bottom} />
</div>
</div>
</div>
{/* Underside: a floating strip BELOW the whole composer surface.
Chrome-free by design — contributions bring their own pill/skin,
like the micro-action strip above. In flow (the root is
bottom-anchored, so this grows the composer upward and stays on
screen) but OUTSIDE the surface, so it escapes the surface's
clipping, border, and scroll fade. Shares the micro-action
strip's grid so the two bracket the composer on one vertical
line. Renders nothing until something contributes. */}
<div className={cn(composerFloatingStrip, 'pt-1.5 empty:hidden')} data-slot="composer-no-drag">
</ComposerPrimitive.Root>
{/* Underside: chrome-free strip BELOW the composer. Outside the root
for the same reason as the micro actions — it must not fall inside
the pop-out drag region. Same px as the strip above, so the two
bracket the composer on one vertical line. */}
<div className={cn(composerFloatingStrip, 'px-[5px] pt-1.5 empty:hidden')}>
<ContribSlot area={COMPOSER_AREAS.underside} />
</div>
</ComposerPrimitive.Root>
</div>
</ComposerPrimitive.Unstable_TriggerPopoverRoot>
<UrlDialog
@@ -0,0 +1,72 @@
import { describe, expect, it } from 'vitest'
import { refAttrs, refAttrsHtml } from '@/components/assistant-ui/directive-text'
import { REFERENCE_STYLES, referenceKind, referenceStyle } from '@/components/assistant-ui/reference-kinds'
/**
* There is ONE inline-reference system: `class="ref"` + `data-ref="<kind>"`.
* A pasted link, an `@file:` chip, a `/skill`, a `@session:` the agent wrote —
* all the same markup, styled by the `.ref` rules in styles.css.
*/
describe('the inline reference contract', () => {
it('marks any element as a reference of a given kind', () => {
expect(refAttrs('file')).toEqual({ className: 'ref', 'data-ref': 'file' })
expect(refAttrsHtml('skill')).toBe('class="ref" data-ref="skill"')
})
it('an unkinded reference is a plain link, not a broken one', () => {
// A bare external link has no kind — it keeps the default link colour
// rather than being tagged with a wrong one.
expect(refAttrs()).toEqual({ className: 'ref' })
expect(refAttrsHtml()).toBe('class="ref"')
})
it('normalises an unknown kind instead of emitting it raw', () => {
// A kind CSS has no rule for would silently render unstyled; coercing to
// `other` keeps it inside the system.
expect(refAttrs('wat')['data-ref']).toBe('other')
expect(referenceKind('wat')).toBe('other')
})
it('ships no colour from TypeScript — the theme owns every accent', () => {
// The whole point of keying on `data-ref`: a skin restyles all references
// at once, and no hex or color-mix() is hardcoded in a component.
for (const [kind, style] of Object.entries(REFERENCE_STYLES)) {
expect(style, `${kind} must not carry a colour`).not.toHaveProperty('color')
}
expect(JSON.stringify(refAttrs('url'))).not.toMatch(/color|#[0-9a-f]{3}/i)
})
it('gives every kind a glyph and a label', () => {
for (const [kind, style] of Object.entries(REFERENCE_STYLES)) {
expect(style.codicon, `${kind} codicon`).toBeTruthy()
expect(style.label, `${kind} label`).toBeTruthy()
// Emoji rows render the emoji itself instead of a glyph.
if (kind !== 'emoji') {
expect(style.paths.length, `${kind} paths`).toBeGreaterThan(0)
}
}
})
it('keeps commands and skills visually distinct', () => {
// Different data-ref values, so the stylesheet can accent them apart.
expect(refAttrs('skill')['data-ref']).not.toBe(refAttrs('command')['data-ref'])
expect(referenceStyle('skill').codicon).not.toBe(referenceStyle('command').codicon)
})
})
describe('references are text, not badges', () => {
it('carries no layout, padding, or background of its own', () => {
// Everything visual lives in the stylesheet. If a component starts adding
// its own chrome here, that's the drift this system exists to prevent.
const { className } = refAttrs('file')
expect(className).toBe('ref')
for (const chrome of ['bg-', 'rounded', 'px-', 'py-', 'border', 'inline-flex', 'text-[']) {
expect(className).not.toContain(chrome)
}
})
})
@@ -1,8 +1,9 @@
import { memo, useState } from 'react'
import { Codicon } from '@/components/ui/codicon'
import { useSessionSlice } from '@/lib/use-session-slice'
import { cn } from '@/lib/utils'
import type { ComposerAction } from '@/store/composer-actions'
import { $composerActionsBySession, type ComposerAction } from '@/store/composer-actions'
import { notifyError } from '@/store/notifications'
/**
@@ -31,19 +32,14 @@ const PILL = cn(
* (`composerFloatingStrip`), this owns only the pills, so the strip above the
* surface and the `composer.underside` strip below it can't drift apart.
*/
export const ActionBadges = memo(function ActionBadges({
actions,
sessionId
}: {
actions: ComposerAction[]
sessionId: string
}) {
export const ActionBadges = memo(function ActionBadges({ sessionId }: { sessionId: null | string }) {
const actions = useSessionSlice($composerActionsBySession, sessionId)
// A pill can kick off async work (a gateway call, a submit). Track which one
// is in flight so it can spin and lock instead of double-firing.
const [runningId, setRunningId] = useState<null | string>(null)
const run = async (action: ComposerAction) => {
if (runningId) {
if (runningId || !sessionId) {
return
}
@@ -1,11 +1,12 @@
import { useStore } from '@nanostores/react'
import { useState } from 'react'
import { useEffect, useState } from 'react'
import { useSessionView } from '@/app/chat/session-view'
import { ModelMenuCloseContext } from '@/app/shell/model-menu-panel'
import { Button } from '@/components/ui/button'
import { DropdownMenu, DropdownMenuContent, DropdownMenuTrigger } from '@/components/ui/dropdown-menu'
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
import { releaseTypingFocus } from '@/components/ui/keyboard-first'
import { Tip } from '@/components/ui/tooltip'
import { useI18n } from '@/i18n'
import { ChevronDown } from '@/lib/icons'
@@ -13,6 +14,8 @@ import { formatModelStatusLabel } from '@/lib/model-status-label'
import { cn } from '@/lib/utils'
import { $currentModelSource, $defaultReasoningEffort, setModelPickerOpen } from '@/store/session'
import { onComposerModelMenuRequest } from './focus'
import { useComposerScope } from './scope'
import type { ChatBarState } from './types'
const PILL = cn(
@@ -51,6 +54,28 @@ export function ModelPill({
const defaultEffort = useStore($defaultReasoningEffort)
const runtimeId = useStore(view.$runtimeId)
const [open, setOpen] = useState(false)
const scope = useComposerScope()
const hasLiveMenu = Boolean(model.modelMenuContent)
// The `composer.modelPicker` hotkey, routed to exactly one surface (the pane
// under the pointer, else the active composer — see requestModelMenuToggle).
// Toggles the live dropdown; with no live menu (gateway closed) it opens the
// full picker dialog, same as clicking the pill.
useEffect(
() =>
onComposerModelMenuRequest(target => {
if (target !== scope.target || disabled) {
return
}
if (hasLiveMenu) {
setOpen(prev => !prev)
} else {
setModelPickerOpen(true)
}
}),
[scope.target, disabled, hasLiveMenu]
)
// The composer pick is sticky: a manual selection is pinned and every NEW
// chat uses it instead of the Settings → Model default — silently, which has
@@ -119,8 +144,19 @@ export function ModelPill({
)
}
// Closing the menu ends its claim on the keyboard: Radix restores focus to
// this pill (a toolbar button), so without the release the Enter that
// committed a model also swallows whatever you type next.
const setMenuOpen = (next: boolean) => {
setOpen(next)
if (!next) {
releaseTypingFocus()
}
}
return (
<DropdownMenu onOpenChange={setOpen} open={open}>
<DropdownMenu onOpenChange={setMenuOpen} open={open}>
<Tip label={title} side="top">
<DropdownMenuTrigger asChild>
<Button aria-label={title} className={pillClass} disabled={disabled} type="button" variant="ghost">
@@ -129,7 +165,7 @@ export function ModelPill({
</DropdownMenuTrigger>
</Tip>
<DropdownMenuContent align="end" className="w-64 p-0" side="top" sideOffset={8}>
<ModelMenuCloseContext.Provider value={() => setOpen(false)}>
<ModelMenuCloseContext.Provider value={() => setMenuOpen(false)}>
{model.modelMenuContent}
</ModelMenuCloseContext.Provider>
</DropdownMenuContent>
@@ -35,6 +35,89 @@ describe('renderComposerContents', () => {
expect(editor.textContent).toContain('<b>raw</b>')
expect(composerPlainText(editor)).toBe('@file:`<img src=x onerror=alert(1)>` <b>raw</b>')
})
it('hydrates a committed leading slash command back to its pill', () => {
// Text-hydration parity with @ refs: a re-render from serialized text
// (draft restore, undo, the trigger commit fallback) must not demote a
// committed no-arg command chip to plain text.
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
renderComposerContents(editor, '/some-skill @folder:`Desktop` ')
const pill = editor.querySelector('[data-slash-kind]')
expect(pill?.getAttribute('data-ref-text')).toBe('/some-skill')
expect(editor.querySelector('[data-ref-kind="folder"]')).not.toBeNull()
expect(composerPlainText(editor)).toBe('/some-skill @folder:`Desktop` ')
})
it('keeps a still-typed leading slash token as editable text', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
// No trailing whitespace — not committed yet.
renderComposerContents(editor, '/some-skil')
expect(editor.querySelector('[data-slash-kind]')).toBeNull()
expect(composerPlainText(editor)).toBe('/some-skil')
})
it('keeps an arg-taking command as text — its tail may be uncommitted prose', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
renderComposerContents(editor, '/goal ship the redesign')
expect(editor.querySelector('[data-slash-kind]')).toBeNull()
expect(composerPlainText(editor)).toBe('/goal ship the redesign')
})
})
describe('replaceBeforeCaret across split text nodes', () => {
it('replaces a token that Chromium fragmented into multiple text nodes', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
editor.contentEditable = 'true'
document.body.append(editor)
editor.append(document.createTextNode('see @Desk'), document.createTextNode('top/'))
const caret = document.createRange()
caret.setStart(editor.lastChild!, 4)
caret.collapse(true)
const selection = window.getSelection()!
selection.removeAllRanges()
selection.addRange(caret)
const fragment = document.createDocumentFragment()
fragment.append(refChipElement('folder', '`Desktop`'), document.createTextNode(' '))
// Token `@Desktop/` (9 chars) spans both text nodes.
expect(replaceBeforeCaret(editor, 9, fragment)).toBe(true)
expect(composerPlainText(editor)).toBe('see @folder:`Desktop` ')
editor.remove()
})
it('refuses when a chip interrupts the span — the token is not contiguous text', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
editor.contentEditable = 'true'
document.body.append(editor)
editor.append(document.createTextNode('a'), refChipElement('file', '`x`'), document.createTextNode('bc'))
const caret = document.createRange()
caret.setStart(editor.lastChild!, 2)
caret.collapse(true)
const selection = window.getSelection()!
selection.removeAllRanges()
selection.addRange(caret)
expect(replaceBeforeCaret(editor, 5, document.createDocumentFragment())).toBe(false)
expect(composerPlainText(editor)).toBe('a@file:`x`bc')
editor.remove()
})
})
describe('normalizeComposerEditorDom', () => {
@@ -147,6 +230,82 @@ describe('insertComposerContentsAtCaret', () => {
editor.remove()
})
// A directive typed by hand chips; the same directive pasted has to chip too,
// or copy/pasting a prompt silently drops every command in it.
it('chips a pasted slash command, including one that ends the paste', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
caretIn(editor)
insertComposerContentsAtCaret(editor, '/some-skill')
expect(editor.querySelector('[data-slash-kind]')?.getAttribute('data-ref-text')).toBe('/some-skill')
// Committed pills carry the trailing space the typed path appends, so a
// later full re-render doesn't read the token as half-typed.
expect(composerPlainText(editor)).toBe('/some-skill ')
editor.remove()
})
it('chips a skill named mid-paste alongside a ref', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
caretIn(editor)
insertComposerContentsAtCaret(editor, 'clean @file:`a.ts` with /some-skill then ship')
expect(editor.querySelectorAll('[data-slash-kind]').length).toBe(1)
expect(editor.querySelectorAll('[data-ref-kind="file"]').length).toBe(1)
expect(composerPlainText(editor)).toBe('clean @file:`a.ts` with /some-skill then ship')
editor.remove()
})
it('leaves a pasted path alone — /usr/local is not a command', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
document.body.append(editor)
caretIn(editor)
insertComposerContentsAtCaret(editor, 'see /usr/local/bin and /goal ship it')
expect(editor.querySelector('[data-slash-kind]')).toBeNull()
expect(composerPlainText(editor)).toBe('see /usr/local/bin and /goal ship it')
editor.remove()
})
it('does not chip a command pasted against a word — foo/clean is not a command', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
editor.textContent = 'foo'
document.body.append(editor)
caretIn(editor)
insertComposerContentsAtCaret(editor, '/some-skill')
expect(editor.querySelector('[data-slash-kind]')).toBeNull()
expect(composerPlainText(editor)).toBe('foo/some-skill')
editor.remove()
})
it('chips a command pasted right after an existing chip', () => {
const editor = document.createElement('div')
editor.dataset.slot = RICH_INPUT_SLOT
editor.append(refChipElement('file', '`a.ts`'))
document.body.append(editor)
caretIn(editor)
insertComposerContentsAtCaret(editor, '/some-skill')
expect(editor.querySelector('[data-slash-kind]')).not.toBeNull()
editor.remove()
})
})
describe('replaceBeforeCaret', () => {
+253 -45
View File
@@ -7,19 +7,45 @@
* plain-text round-trip.
*/
import {
DIRECTIVE_CHIP_CLASS,
directiveIconElement,
directiveIconSvg,
formatRefValue,
refAttrsHtml,
refChipLabel,
slashChipClass,
type SlashChipKind,
slashIconElement
} from '@/components/assistant-ui/directive-text'
import { referenceKind, referenceRe } from '@/components/assistant-ui/reference-kinds'
import { slashCommandMatches, type SlashCommandScanOptions } from './slash-refs'
export const RICH_INPUT_SLOT = 'composer-rich-input'
export const REF_RE = /@(file|folder|url|image|tool|line|terminal|session):(`[^`\n]+`|"[^"\n]+"|'[^'\n]+'|\S+)/g
/** Paints `data-placeholder` while the editor is empty.
*
* `:empty` can't be the whole test: a cleared editor keeps a scaffolding <br>
* so the contenteditable doesn't collapse, and that break makes `:empty`
* false. Nor can CSS infer it on its own — a text node is invisible to
* selectors, so `one<br>` and a lone `<br>` are the same shape, and
* `:has(> br:only-child)` would paint the placeholder straight over the
* user's text. The code that empties the editor is what knows, so it marks it.
*
* @see markEditorEmptiness */
export const COMPOSER_PLACEHOLDER_CLASS =
'[&:is(:empty,[data-empty])]:before:content-[attr(data-placeholder)] [&:is(:empty,[data-empty])]:before:text-muted-foreground/60'
/** Keep that marker in step with the editor root's contents. */
export function markEditorEmptiness(editor: HTMLElement) {
if (editor.childNodes.length === 0) {
editor.dataset.empty = ''
} else {
delete editor.dataset.empty
}
}
/** @see referenceRe — the shared pattern every surface recognises a reference
* with. Module-level `/g` regexes carry `lastIndex`, so call sites reset it. */
export const REF_RE = referenceRe()
const ESC: Record<string, string> = { '&': '&amp;', '<': '&lt;', '>': '&gt;', '"': '&quot;', "'": '&#039;' }
@@ -58,42 +84,38 @@ export function refChipHtml(kind: string, rawValue: string, displayLabel?: strin
const label = displayLabel || refChipLabel(kind, id)
return `<span contenteditable="false" title="${escapeHtml(id)}" data-ref-text="${escapeHtml(text)}" data-ref-id="${escapeHtml(id)}" data-ref-kind="${escapeHtml(kind)}" class="${DIRECTIVE_CHIP_CLASS}">${directiveIconSvg(kind)}<span class="truncate">${escapeHtml(label)}</span></span>`
return `<span contenteditable="false" title="${escapeHtml(id)}" data-ref-text="${escapeHtml(text)}" data-ref-id="${escapeHtml(id)}" data-ref-kind="${escapeHtml(kind)}" ${refAttrsHtml(kind)}>${directiveIconSvg(kind)}${escapeHtml(label)}</span>`
}
export function refChipElement(kind: string, rawValue: string, displayLabel?: string) {
const id = unquoteRef(rawValue)
const text = `@${kind}:${quoteRefValue(id)}`
const chip = document.createElement('span')
const label = document.createElement('span')
chip.contentEditable = 'false'
chip.title = id
chip.dataset.refText = text
chip.dataset.refId = id
chip.dataset.refKind = kind
chip.className = DIRECTIVE_CHIP_CLASS
label.className = 'truncate'
label.textContent = displayLabel || refChipLabel(kind, id)
chip.append(directiveIconElement(kind), label)
chip.className = 'ref'
chip.dataset.ref = referenceKind(kind)
chip.append(directiveIconElement(kind), document.createTextNode(displayLabel || refChipLabel(kind, id)))
return chip
}
/** A non-editable pill for a picked slash command (`/skin nous`, `/tropes`).
/** A non-editable reference for a picked slash command (`/skin nous`, `/tropes`).
* `data-ref-text` carries the literal command so `composerPlainText` round-trips
* it back to the exact text that gets submitted. */
export function slashChipElement(command: string, kind: SlashChipKind, label?: string) {
const chip = document.createElement('span')
const text = document.createElement('span')
chip.contentEditable = 'false'
chip.dataset.refText = command
chip.dataset.slashKind = kind
chip.className = slashChipClass(kind)
text.className = 'truncate'
text.textContent = label || command
chip.append(slashIconElement(kind), text)
chip.className = 'ref'
chip.dataset.ref = kind
chip.append(slashIconElement(kind), document.createTextNode(label || command))
return chip
}
@@ -112,24 +134,63 @@ function appendTextWithBreaks(target: DocumentFragment | HTMLElement, text: stri
})
}
export function appendComposerContents(target: DocumentFragment | HTMLElement, text: string) {
let cursor = 0
/** Every span of `text` that renders as a chip, in source order. */
function chipSpans(text: string, options: SlashCommandScanOptions) {
REF_RE.lastIndex = 0
for (const match of text.matchAll(REF_RE)) {
const index = match.index ?? 0
appendTextWithBreaks(target, text.slice(cursor, index))
target.append(refChipElement(match[1] || 'file', match[2] || ''))
cursor = index + match[0].length
const refs = Array.from(text.matchAll(REF_RE)).map(match => {
const start = match.index ?? 0
return { end: start + match[0].length, node: () => refChipElement(match[1] || 'file', match[2] || ''), start }
})
const commands = slashCommandMatches(text, options).map(match => ({
end: match.end,
node: () => slashChipElement(match.command, match.kind),
start: match.start
}))
return [...refs, ...commands].sort((a, b) => a.start - b.start)
}
/** Build the chip/text DOM for `text`. Directives hydrate back to their pills —
* `@kind:value` refs and `/command` invocations both — so text that arrives
* whole (a paste, a restored draft, an undo step, a rebuilt line) carries the
* same chips the typed path would have committed. */
export function appendComposerContents(
target: DocumentFragment | HTMLElement,
text: string,
options: SlashCommandScanOptions = {}
) {
let cursor = 0
for (const span of chipSpans(text, options)) {
// A `@` ref wins an overlap: a command token can't contain an `@`, so the
// only way spans collide is a slash inside a quoted ref value
// (`` @url:`a /clean` ``), which belongs to that value.
if (span.start < cursor) {
continue
}
appendTextWithBreaks(target, text.slice(cursor, span.start))
target.append(span.node())
cursor = span.end
}
appendTextWithBreaks(target, text.slice(cursor))
}
export function renderComposerContents(target: HTMLElement, text: string) {
export function renderComposerContents(target: HTMLElement, text: string, options?: SlashCommandScanOptions) {
target.replaceChildren()
appendComposerContents(target, text)
// Defaults to live editing, where a token ending the text is still being
// typed (`/wor`) and must stay editable. Callers repainting inert text (a
// restored draft, a sent message opened for edit) pass `trailingCommitted`.
appendComposerContents(target, text, options)
// The other writer that reshapes the editor root: painting a restored draft
// in clears the marker, clearing back to '' sets it.
markEditorEmptiness(target)
}
/** Caret range when the selection lives inside `editor`; else null. */
@@ -144,20 +205,95 @@ function composerSelectionRange(editor: HTMLElement) {
return { range, selection }
}
/** Insert text at the caret (replacing any selection), with any `@kind:value`
* directives in it landing as chips. Pastes use this instead of
* `execCommand('insertText')` — Chromium's editing pipeline is ~O(n²) on large
* multiline blobs. */
export function insertComposerContentsAtCaret(editor: HTMLElement, text: string) {
/** Serialized text from the editor's start up to (`container`, `offset`).
*
* Chips are ATOMIC here: each contributes an object-replacement placeholder
* rather than leaking its label text, and a <br> contributes a newline. That
* makes a chip edge read as a token boundary, which is what both trigger
* detection and directive recognition need. */
export function serializeTextBefore(editor: HTMLElement, container: Node, offset: number): string {
const probe = document.createRange()
probe.selectNodeContents(editor)
probe.setEnd(container, offset)
const scratch = document.createElement('div')
scratch.append(probe.cloneContents())
for (const chip of scratch.querySelectorAll('[data-ref-text]')) {
chip.replaceWith('\uFFFC')
}
for (const br of scratch.querySelectorAll('br')) {
br.replaceWith('\n')
}
return scratch.textContent ?? ''
}
/** True when the insertion point starts a token — the editor's start, or after
* whitespace or a chip. `foo` + a pasted `/clean` is `foo/clean`, not a
* command; `foo ` + the same paste is. */
function atTokenBoundary(editor: HTMLElement, range: Range | null): boolean {
// No caret means the insert lands at the end, so the question is about the
// editor's last character either way.
const before = range
? serializeTextBefore(editor, range.startContainer, range.startOffset)
: serializeTextBefore(editor, editor, editor.childNodes.length)
const last = before.slice(-1)
return !last || /[\s\uFFFC]/.test(last)
}
/** Insert text at the caret (replacing any selection), with any directives in
* it landing as chips. Pastes use this instead of `execCommand('insertText')`
* — Chromium's editing pipeline is ~O(n²) on large multiline blobs.
*
* The text arrives whole rather than typed, so a `/command` ending it is
* complete rather than half-written and chips like the rest.
*
* `consumeBefore` characters immediately before the caret are swallowed by the
* insert. That's how a paste into an open `@url:` scope replaces the scope
* instead of stacking on it (`@url:@url:\`https://…\``). */
export function insertComposerContentsAtCaret(editor: HTMLElement, text: string, consumeBefore = 0) {
const scoped = consumeBefore > 0 ? rangeBeforeCaret(editor, consumeBefore) : null
if (scoped) {
scoped.deleteContents()
scoped.collapse(true)
const selection = window.getSelection()
selection?.removeAllRanges()
selection?.addRange(scoped)
}
const hit = composerSelectionRange(editor)
const fragment = document.createDocumentFragment()
appendComposerContents(fragment, text)
// Before measuring the boundary — a replaced selection puts the insertion
// point where the selection started, not where it ended.
if (hit) {
hit.range.deleteContents()
}
appendComposerContents(fragment, text, {
boundaryBefore: atTokenBoundary(editor, hit?.range ?? null),
trailingCommitted: true
})
// A slash pill ending the insert gets the trailing space the typed commit
// path appends, or the next full re-render reads it as a half-typed token
// and demotes it. `@` refs need no marker — REF_RE re-chips them either way.
if ((fragment.lastChild as HTMLElement | null)?.dataset?.slashKind) {
fragment.append(document.createTextNode(' '))
}
const tail = fragment.lastChild
if (hit) {
hit.range.deleteContents()
hit.range.insertNode(fragment)
} else {
editor.append(fragment)
@@ -173,27 +309,84 @@ export function insertComposerContentsAtCaret(editor: HTMLElement, text: string)
}
}
/** Swap the `length` characters immediately before a collapsed caret for
* `fragment`, leaving the caret after it. Returns whether it ran — a caret that
* isn't inside a text node holding the whole token is left alone. */
export function replaceBeforeCaret(editor: HTMLElement, length: number, fragment: DocumentFragment) {
/** Range covering exactly `length` serialized characters immediately before a
* collapsed caret, spanning Chromium's split text nodes. Null when the caret
* isn't a collapsed selection in `editor`, or when a chip/<br>/block boundary
* interrupts before `length` characters are covered — a trigger token is
* always contiguous text, so anything else means "don't touch the DOM here".
*
* This is what keeps chip insertion stable: Chromium fragments text nodes
* around contenteditable=false chips on every edit, so any commit path that
* demands the whole token inside ONE text node (the old check) degrades to a
* full re-render as soon as a chip exists anywhere in the line. */
export function rangeBeforeCaret(editor: HTMLElement, length: number): Range | null {
const hit = composerSelectionRange(editor)
if (!hit?.range.collapsed) {
return false
if (!hit?.range.collapsed || length <= 0) {
return null
}
const { startContainer, startOffset } = hit.range
let node: Node | null = hit.range.startContainer
let offset = hit.range.startOffset
if (startContainer.nodeType !== Node.TEXT_NODE || startOffset < length) {
return false
// An element-positioned caret (common right after programmatic caret moves)
// resolves to the end of the text node before it. A chip or <br> there means
// no text token precedes the caret — bail rather than guess.
if (node.nodeType !== Node.TEXT_NODE) {
node = node.childNodes[offset - 1] ?? null
if (node?.nodeType !== Node.TEXT_NODE) {
return null
}
offset = (node.textContent || '').length
}
let startNode = node as Text
let startOffset = offset
let remaining = length
while (remaining > 0) {
if (startOffset >= remaining) {
startOffset -= remaining
remaining = 0
break
}
remaining -= startOffset
const prev: Node | null = startNode.previousSibling
if (prev?.nodeType !== Node.TEXT_NODE) {
return null
}
startNode = prev as Text
startOffset = (prev.textContent || '').length
}
const range = document.createRange()
range.setStart(startNode, startOffset)
range.setEnd(hit.range.startContainer, hit.range.startOffset)
return range
}
/** Swap the `length` characters immediately before a collapsed caret for
* `fragment`, leaving the caret after it. Returns whether it ran. Spans split
* text nodes (see rangeBeforeCaret) — a token typed around existing chips
* still commits in place instead of falling back to a full re-render. */
export function replaceBeforeCaret(editor: HTMLElement, length: number, fragment: DocumentFragment) {
const range = rangeBeforeCaret(editor, length)
if (!range) {
return false
}
const tail = fragment.lastChild
range.setStart(startContainer, startOffset - length)
range.setEnd(startContainer, startOffset)
range.deleteContents()
range.insertNode(fragment)
@@ -202,8 +395,11 @@ export function replaceBeforeCaret(editor: HTMLElement, length: number, fragment
}
range.collapse(true)
hit.selection.removeAllRanges()
hit.selection.addRange(range)
const selection = window.getSelection()
selection?.removeAllRanges()
selection?.addRange(range)
return true
}
@@ -312,6 +508,15 @@ export function composerPlainText(node: Node): string {
return el.dataset.refText
}
// An editor holding nothing but the placeholder <br> is EMPTY. That <br> is
// scaffolding normalizeComposerEditorDom adds so the contenteditable keeps
// its height — not a line the user typed. Reading it as "\n" is how a
// just-cleared composer stayed non-empty: the newline got stashed as the
// session's draft and painted back on return.
if (el.dataset.slot === RICH_INPUT_SLOT && el.childNodes.length === 1 && el.firstChild?.nodeName === 'BR') {
return ''
}
if (el.tagName === 'BR') {
return '\n'
}
@@ -502,6 +707,9 @@ export function normalizeComposerEditorDom(editor: HTMLElement) {
// composer to appear as a tiny dot/pixel. Ensure there's always at least
// one <br> so the element maintains intrinsic height. The CSS min-height
// is a belt; the <br> is suspenders — together they prevent the shrink.
// That break is also why emptiness has to be marked, not inferred.
markEditorEmptiness(editor)
if (editor.childNodes.length === 0) {
editor.appendChild(document.createElement('br'))
}
@@ -0,0 +1,45 @@
import { describe, expect, it } from 'vitest'
import { slashCommandMatches } from './slash-refs'
const commands = (text: string, options?: Parameters<typeof slashCommandMatches>[1]) =>
slashCommandMatches(text, options).map(match => `${match.kind}:${match.command}`)
describe('slashCommandMatches', () => {
it('recognizes a leading command and a skill named mid-prose', () => {
expect(commands('/some-skill clean this with /other-skill please')).toEqual([
'skill:/some-skill',
'skill:/other-skill'
])
})
it('leaves a path alone — /usr/local/bin is not a command', () => {
expect(commands('see /usr/local/bin ')).toEqual([])
})
it('holds a trailing token as still-typed unless the text is inert', () => {
expect(commands('/some-skill')).toEqual([])
expect(commands('/some-skill', { trailingCommitted: true })).toEqual(['skill:/some-skill'])
})
it('leaves an arg-taking command as text — its tail may be prose', () => {
expect(commands('/goal ship the redesign')).toEqual([])
})
it('leaves a command with no desktop surface as text', () => {
expect(commands('/exit now')).toEqual([])
})
it('offers a built-in only as an invocation, never mid-message', () => {
// Mirrors what the popover offers: `/new` acts on the app, so it means
// nothing dropped into a sentence, while a skill reads as "handle this
// part with X".
expect(commands('/new ')).toEqual(['command:/new'])
expect(commands('start over with /new ')).toEqual([])
expect(commands('start over with /some-skill ')).toEqual(['skill:/some-skill'])
})
it('disqualifies a leading token when the text lands mid-word', () => {
expect(commands('/some-skill ', { boundaryBefore: false })).toEqual([])
})
})
@@ -0,0 +1,105 @@
/**
* Slash-command recognition for text the composer did not watch being typed —
* a paste, a restored draft, an undo step, a rebuilt line.
*
* The typed path chips a command as it's picked or accepted, so the composer
* agrees with what the sent message renders (`SLASH_SKILL_RE` in
* directive-text). Text that arrives whole never passed through that path, so
* it needs the same commands recognized in place — on exactly the terms the
* typed path would have used, or hydration invents pills the popover would
* never have committed.
*/
import type { SlashChipKind } from '@/components/assistant-ui/directive-text'
import {
desktopSlashCommandArgumentMode,
isDesktopSlashCommand,
resolveDesktopCommand
} from '@/lib/desktop-slash-commands'
// A command token starts a word and doesn't continue into a path: `/usr/local`
// is a path, not a `/usr` command. Same shape the sent message uses to decide
// what renders as a pill, so the composer and the transcript agree.
const SLASH_COMMAND_RE = /(?<=^|\s)\/([a-zA-Z][\w-]*)(?![\w-]*\/)/g
export interface SlashCommandMatch {
/** The command with its leading slash, e.g. `/clean`. */
command: string
end: number
kind: SlashChipKind
start: number
}
export interface SlashCommandScanOptions {
/**
* Whether the text is preceded by a token boundary. False when it's being
* inserted mid-word (a paste landing against existing characters), which
* disqualifies a token at index 0 — `foo/clean` is not a command. It also
* makes that token mid-message rather than an invocation.
*/
boundaryBefore?: boolean
/**
* Whether a token ending the text counts as committed. True for inert text
* (a paste, dropped content): nothing is being typed, so `/clean` at the end
* is the whole command. False while editing live, where a trailing `/wor` is
* a half-typed query the popover owns and must leave editable.
*/
trailingCommitted?: boolean
}
/**
* Only commands with NO argument stage chip: their committed pill is exactly
* the bare `/name`, so the boundary is unambiguous. Arg-taking commands
* (`/goal ship it`) stay text — their tail may be prose. Commands with no
* desktop surface at all (`/exit`, `/config`) stay text too.
*/
function chippableKind(command: string): SlashChipKind | null {
if (!isDesktopSlashCommand(command) || desktopSlashCommandArgumentMode(command) !== null) {
return null
}
return resolveDesktopCommand(command) ? 'command' : 'skill'
}
/** Every `/command` in `text` that should render as a pill, in source order. */
export function slashCommandMatches(text: string, options: SlashCommandScanOptions = {}): SlashCommandMatch[] {
const { boundaryBefore = true, trailingCommitted = false } = options
if (!text.includes('/')) {
return []
}
const matches: SlashCommandMatch[] = []
for (const match of text.matchAll(SLASH_COMMAND_RE)) {
const start = match.index ?? 0
const command = match[0]
const end = start + command.length
const after = text[end]
// A committed pill always carries its auto-inserted trailing space, which
// is what separates it from a token still being typed.
if (after === undefined ? !trailingCommitted : !/\s/.test(after)) {
continue
}
// Only the FIRST token can be an invocation, and only when the text lands
// on a token boundary — `foo` + a pasted `/clean` is `foo/clean`.
const invocation = start === 0
if (invocation && !boundaryBefore) {
continue
}
const kind = chippableKind(command)
// Later tokens are references dropped into prose, where the popover offers
// SKILLS alone — a built-in like `/new` acts on the app and means nothing
// mid-sentence. Hydration has to agree, or pasted text grows pills typing
// never would.
if (kind && (invocation || kind === 'skill')) {
matches.push({ command, end, kind, start })
}
}
return matches
}
@@ -15,7 +15,7 @@ import { Codicon } from '@/components/ui/codicon'
import { DiffCount } from '@/components/ui/diff-count'
import type { HermesGitBranch } from '@/global'
import { useI18n } from '@/i18n'
import { $repoStatus, $repoWorktrees } from '@/store/coding-status'
import { registerRepoStatusCwd, repoStatusForCwd, repoWorktreesForCwd } from '@/store/coding-status'
import { notifyError } from '@/store/notifications'
import { $newWorktreeRequest } from '@/store/projects'
@@ -62,14 +62,24 @@ export const CodingStatusRow = memo(function CodingStatusRow({
const { t } = useI18n()
const s = t.statusStack.coding
const p = t.sidebar.projects
const status = useStore($repoStatus)
const worktrees = useStore($repoWorktrees)
const resolvedRepoPath = repoPath?.trim() || undefined
// This surface's OWN worktree, always — never the primary's. The row used to
// fall back to the global `$repoStatus` for a blank repoPath, which painted
// the main pane's branch/± onto a tile whose cwd hadn't resolved yet. That
// fallback bought nothing (the primary's computed is keyed to `$currentCwd`,
// which is blank in exactly the same case) and cost a wrong-tree rail.
const status = useStore(repoStatusForCwd(resolvedRepoPath))
const worktrees = useStore(repoWorktreesForCwd(resolvedRepoPath))
// While mounted, keep this worktree in the coding-status refresh set so the
// turn-settle / tool-complete / focus edges re-probe it too (tiles otherwise
// only refreshed when the MAIN cwd probe happened to cover them).
useEffect(() => registerRepoStatusCwd(resolvedRepoPath), [resolvedRepoPath])
// Shared worktree dialog — replaces the old inline dialog. Opened by the
// dropdown menu's "branch off" items and the global ⌘⇧B hotkey.
const [worktreeOpen, setWorktreeOpen] = useState(false)
const [worktreeBase, setWorktreeBase] = useState<string | undefined>(undefined)
const resolvedRepoPath = repoPath?.trim() || undefined
const switchToBranch = async (branch: string) => {
if (!onSwitchBranch) {
@@ -1,12 +1,11 @@
import { useStore } from '@nanostores/react'
import { type ReactNode, useEffect, useLayoutEffect, useMemo, useRef } from 'react'
import { type ReactNode, useEffect, useMemo } from 'react'
import { useNavigate } from 'react-router-dom'
import { blurComposerInput } from '@/app/chat/composer/focus'
import { chatSurfaceRoot, clearSurfaceVar, setSurfaceVar, STATUS_STACK_VAR } from '@/app/chat/surface-vars'
import { AGENTS_ROUTE } from '@/app/routes'
import { BillingBanner } from '@/components/billing-banner'
import { composerDockCard, composerFloatingStrip } from '@/components/chat/composer-dock'
import { composerDockCard } from '@/components/chat/composer-dock'
import { StatusSection } from '@/components/chat/status-section'
import { Button } from '@/components/ui/button'
import { Codicon } from '@/components/ui/codicon'
@@ -15,7 +14,6 @@ import { type Translations, useI18n } from '@/i18n'
import { useSessionSlice } from '@/lib/use-session-slice'
import { cn } from '@/lib/utils'
import { $billingBlock } from '@/store/billing-block'
import { $composerActionsBySession } from '@/store/composer-actions'
import {
$statusItemsBySession,
type ComposerStatusItem,
@@ -30,7 +28,6 @@ import { $previewStatusBySession, dismissPreviewArtifact } from '@/store/preview
import { $threadScrolledUp } from '@/store/thread-scroll'
import { openSessionInNewWindow } from '@/store/windows'
import { ActionBadges } from './action-badges'
import { PreviewStatusRow } from './preview-row'
import { StatusItemRow } from './status-row'
@@ -95,7 +92,6 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro
// items actually changed.
const items = useSessionSlice($statusItemsBySession, sessionId)
const previews = useSessionSlice($previewStatusBySession, sessionId)
const actions = useSessionSlice($composerActionsBySession, sessionId)
const scrolledUp = useStore($threadScrolledUp)
const billing = useStore($billingBlock)
@@ -154,10 +150,6 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro
sections.push({ key: 'billing', node: <BillingBanner sessionId={sessionId} /> })
}
// Micro actions ride at the top of the stack — the one block you press
// rather than read. Rendered OUTSIDE the card (see `actionStrip`) so the
// pills float; a blocked account still gets the billing wall above them.
for (const group of groups) {
sections.push({
key: group.type,
@@ -219,50 +211,11 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro
// status card, above the billing wall, above everything. They're the only
// rows up here you press instead of read, so nothing may ever stack on top
// of them. Rendered outside the card (below) so the pills float.
const actionStrip = actions.length > 0 && sessionId ? <ActionBadges actions={actions} sessionId={sessionId} /> : null
const visible = sections.length > 0
const visible = sections.length > 0 || Boolean(actionStrip)
const stackRef = useRef<HTMLDivElement | null>(null)
// The stack is out of flow (overlays the thread), so the composer's measured
// height never sees it. Publish our own measured height — bucketed like the
// composer's, to avoid style invalidation churn — so the thread's
// last-message clearance can add it and the stack never hides messages.
// Scoped to THIS surface: tiles render their own stack (see surface-vars.ts).
useLayoutEffect(() => {
const el = stackRef.current
if (!visible || !el) {
return
}
// Resolve the owning surface NOW, while the node is attached. The cleanup
// below runs after the stack collapsed and React removed the div, so
// closest() from the detached node misses [data-chat-surface] and would
// clear the document root instead — leaving the stale height on the
// surface, which keeps inflating the thread's bottom clearance until the
// next publish.
const root = chatSurfaceRoot(el)
let last = -1
const sync = () => {
const bucket = Math.round(el.getBoundingClientRect().height / 8) * 8
if (bucket !== last) {
last = bucket
setSurfaceVar(el, STATUS_STACK_VAR, `${bucket}px`)
}
}
const observer = new ResizeObserver(sync)
observer.observe(el)
sync()
return () => {
observer.disconnect()
clearSurfaceVar(root, STATUS_STACK_VAR)
}
}, [visible])
// No height to publish: the stack is an in-flow child of the composer dock,
// so the dock's own measurement (--composer-measured-height) already covers
// it and the thread clears both with one number.
if (!visible) {
return null
@@ -270,56 +223,34 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro
return (
<div
// Sits in the overlay lane above the composer. The composer root has pt-2
// before the actual surface; translate by that amount so the stack returns
// to its original attachment point without intruding into the repo strip.
// pl matches the surface's own left edge: `inset-x-0` resolves against the
// root's PADDING box, while the surface and the underside strip sit in its
// CONTENT box, so without it the lane hangs 5px further left than both.
className="absolute inset-x-0 bottom-full z-3 flex max-h-[40vh] flex-col translate-y-2 pl-[0.3125rem]"
// In flow in the dock column, directly above the composer. The dock is
// bottom-anchored, so this grows upward over the thread without needing
// to be positioned — and it shares the dock's left edge for free.
className="flex max-h-[40vh] min-h-0 flex-col overflow-y-auto"
onPointerDownCapture={() => blurComposerInput()}
ref={stackRef}
>
{/* FIRST in the lane and OUTSIDE the scroller, so nothing can ever sit
above the pills — not the status card, not the billing wall — and a
long todo list can't scroll them out of view. Outside the card too:
they carry their own fill, so they must not paint on its background. */}
{actionStrip && (
{/* The card paints the shared --composer-fill (rest / scrolled / focused
all match the composer surface by construction); on scroll we only
ghost the CONTENT — element opacity on the card would kill the blur.
Rounded top, square bottom; the bottom border is TRANSPARENT — the
composer surface's visible top border (which sits at a higher z) is the
single shared seam, so the two read as one fused capsule. */}
{sections.length > 0 && (
<div
className={cn(
composerFloatingStrip,
'shrink-0 pb-1.5 transition-opacity duration-200 ease-out',
composerDockCard('top'),
// Inset (mx-2) so the stack reads slightly narrower than the composer
// surface below it — the original look.
'mx-2 overflow-hidden rounded-b-none border-b border-b-transparent pt-0.5',
'transition-opacity duration-200 ease-out',
scrolledUp ? 'opacity-30 group-hover/composer:opacity-100' : 'opacity-100'
)}
>
{actionStrip}
{sections.map(section => (
<div key={section.key}>{section.node}</div>
))}
</div>
)}
{/* Everything else scrolls under them. */}
<div className="min-h-0 overflow-y-auto">
{/* The card paints the shared --composer-fill (rest / scrolled / focused
all match the composer surface by construction); on scroll we only
ghost the CONTENT — element opacity on the card would kill the blur.
Rounded top, square bottom; the bottom border is TRANSPARENT — the
composer surface's visible top border (which sits at a higher z) is the
single shared seam, so the two read as one fused capsule. */}
{sections.length > 0 && (
<div
className={cn(
composerDockCard('top'),
// Inset (mx-2) so the stack reads slightly narrower than the composer
// surface below it — the original look.
'mx-2 overflow-hidden rounded-b-none border-b border-b-transparent pt-0.5',
'transition-opacity duration-200 ease-out',
scrolledUp ? 'opacity-30 group-hover/composer:opacity-100' : 'opacity-100'
)}
>
{sections.map(section => (
<div key={section.key}>{section.node}</div>
))}
</div>
)}
</div>
</div>
)
}
@@ -1,94 +0,0 @@
import { act, cleanup, render } from '@testing-library/react'
import { MemoryRouter } from 'react-router-dom'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { STATUS_STACK_VAR } from '@/app/chat/surface-vars'
import { I18nProvider } from '@/i18n'
import { $goalsBySession, type SessionGoal } from '@/store/goals'
import { ComposerStatusStack } from './index'
// The stack measures itself into a surface var — jsdom has no ResizeObserver.
class ResizeObserverStub {
observe() {}
unobserve() {}
disconnect() {}
}
vi.stubGlobal('ResizeObserver', ResizeObserverStub)
const SID = 'sess-height-1'
const goal = (): SessionGoal => ({ status: 'active', title: 'ship the feature', updatedAt: Date.now() })
/**
* Regression: when the stack collapses (its last item finishes), React removes
* the stack div BEFORE the layout-effect cleanup runs. Resolving the surface
* from the ref at cleanup time then walks a DETACHED node, misses
* [data-chat-surface], and clears the document root instead — the stale height
* stays on the surface and keeps inflating the thread's bottom clearance
* (`--thread-last-message-clearance`) until the next publish. The effect must
* capture its surface root while the node is still attached.
*/
describe('ComposerStatusStack surface-var lifecycle', () => {
beforeEach(() => {
$goalsBySession.set({})
})
afterEach(() => {
cleanup()
$goalsBySession.set({})
document.documentElement.style.removeProperty(STATUS_STACK_VAR)
})
function renderOnSurface() {
const surface = document.createElement('div')
surface.setAttribute('data-chat-surface', '')
document.body.append(surface)
const view = render(
<MemoryRouter>
<I18nProvider configClient={null} initialLocale="en">
<ComposerStatusStack queue={null} sessionId={SID} />
</I18nProvider>
</MemoryRouter>,
{ container: surface }
)
return { surface, view }
}
it('publishes its measured height onto the owning surface while visible', () => {
$goalsBySession.set({ [SID]: goal() })
const { surface } = renderOnSurface()
// jsdom measures 0 — the value is irrelevant, the target element is not.
expect(surface.style.getPropertyValue(STATUS_STACK_VAR)).toBe('0px')
expect(document.documentElement.style.getPropertyValue(STATUS_STACK_VAR)).toBe('')
})
it('clears the surface var when the stack collapses to nothing', () => {
$goalsBySession.set({ [SID]: goal() })
const { surface } = renderOnSurface()
expect(surface.style.getPropertyValue(STATUS_STACK_VAR)).toBe('0px')
// Last status item goes away → the component renders null and React
// detaches the stack div before the cleanup runs.
act(() => $goalsBySession.set({}))
expect(surface.style.getPropertyValue(STATUS_STACK_VAR)).toBe('')
})
it('clears the surface var on unmount', () => {
$goalsBySession.set({ [SID]: goal() })
const { surface, view } = renderOnSurface()
expect(surface.style.getPropertyValue(STATUS_STACK_VAR)).toBe('0px')
view.unmount()
expect(surface.style.getPropertyValue(STATUS_STACK_VAR)).toBe('')
})
})
@@ -4,19 +4,19 @@ import { blobDedupeKey, detectTrigger, extractClipboardImageBlobs } from './text
describe('detectTrigger', () => {
it('detects a bare slash trigger with an empty query', () => {
expect(detectTrigger('/')).toEqual({ kind: '/', query: '', tokenLength: 1 })
expect(detectTrigger('/')).toEqual({ kind: '/', query: '', tokenLength: 1, value: '' })
})
it('detects a slash command query', () => {
expect(detectTrigger('/skill')).toEqual({ kind: '/', query: 'skill', tokenLength: 6 })
expect(detectTrigger('/skill')).toEqual({ kind: '/', query: 'skill', tokenLength: 6, value: 'skill' })
})
it('detects a bare at-mention trigger with an empty query', () => {
expect(detectTrigger('@')).toEqual({ kind: '@', query: '', tokenLength: 1 })
expect(detectTrigger('@')).toEqual({ kind: '@', query: '', tokenLength: 1, value: '' })
})
it('detects an at-mention query', () => {
expect(detectTrigger('@file')).toEqual({ kind: '@', query: 'file', tokenLength: 5 })
expect(detectTrigger('@file')).toEqual({ kind: '@', query: 'file', tokenLength: 5, value: 'file' })
})
it('returns null for plain text', () => {
@@ -27,17 +27,20 @@ describe('detectTrigger', () => {
expect(detectTrigger('/personality ')).toEqual({
kind: '/',
query: 'personality ',
tokenLength: 13
tokenLength: 13,
value: 'personality '
})
expect(detectTrigger('/personality alic')).toEqual({
kind: '/',
query: 'personality alic',
tokenLength: 17
tokenLength: 17,
value: 'personality alic'
})
expect(detectTrigger('/tools enable foo')).toEqual({
kind: '/',
query: 'tools enable foo',
tokenLength: 17
tokenLength: 17,
value: 'tools enable foo'
})
})
@@ -54,24 +57,75 @@ describe('detectTrigger', () => {
it('keeps the at-mention live while walking into subfolders', () => {
// A `/` inside the query is path navigation, not the end of the token —
// the popover has to stay open so the next directory level can load.
expect(detectTrigger('@./')).toEqual({ kind: '@', query: './', tokenLength: 3 })
expect(detectTrigger('@./src')).toEqual({ kind: '@', query: './src', tokenLength: 6 })
expect(detectTrigger('@~/Desktop/')).toEqual({ kind: '@', query: '~/Desktop/', tokenLength: 11 })
expect(detectTrigger('@/usr/local')).toEqual({ kind: '@', query: '/usr/local', tokenLength: 11 })
expect(detectTrigger('@./')).toEqual({ kind: '@', query: './', tokenLength: 3, value: './' })
expect(detectTrigger('@./src')).toEqual({ kind: '@', query: './src', tokenLength: 6, value: './src' })
expect(detectTrigger('@~/Desktop/')).toEqual({
kind: '@',
query: '~/Desktop/',
tokenLength: 11,
value: '~/Desktop/'
})
expect(detectTrigger('@/usr/local')).toEqual({
kind: '@',
query: '/usr/local',
tokenLength: 11,
value: '/usr/local'
})
expect(detectTrigger('@apps/desktop/src')).toEqual({
kind: '@',
query: 'apps/desktop/src',
tokenLength: 17
tokenLength: 17,
value: 'apps/desktop/src'
})
})
it('keeps the at-mention live for a typed ref kind with a path', () => {
it('treats a chip edge as a token boundary, like whitespace', () => {
// U+FFFC is textBeforeCaret's placeholder for a committed pill. Upstream
// assistant-ui's Lexical DirectivePlugin gets the same semantics from node
// boundaries: typing a trigger right after a chip (no space) still opens
// the popover, and a chip inside a token ends it.
expect(detectTrigger('\uFFFC@Desk')).toEqual({ kind: '@', query: 'Desk', tokenLength: 5, value: 'Desk' })
// Not position 0, so it's an inline reference — not a command invocation.
expect(detectTrigger('\uFFFC/cle')).toEqual({
inline: true,
kind: '/',
query: 'cle',
tokenLength: 4,
value: 'cle'
})
// The placeholder itself never leaks into a query.
expect(detectTrigger('@a\uFFFCb')).toBeNull()
})
it('splits a typed ref kind off as the browse scope', () => {
// `@folder:apps/` is ONE token with TWO parts. The kind is the mode the
// user is browsing in, so it's held as `scope` rather than left in `value`
// for every consumer to re-parse (or, worse, to preserve by hand).
expect(detectTrigger('@file:src/main.tsx')).toEqual({
kind: '@',
query: 'file:src/main.tsx',
tokenLength: 18
scope: 'file',
tokenLength: 18,
value: 'src/main.tsx'
})
expect(detectTrigger('@folder:apps/')).toEqual({ kind: '@', query: 'folder:apps/', tokenLength: 13 })
expect(detectTrigger('@folder:apps/')).toEqual({
kind: '@',
query: 'folder:apps/',
scope: 'folder',
tokenLength: 13,
value: 'apps/'
})
// A scope with nothing typed after it is the empty-browse state the
// popover renders a header for.
expect(detectTrigger('@url:')).toEqual({ kind: '@', query: 'url:', scope: 'url', tokenLength: 5, value: '' })
})
it('only treats a KNOWN kind as a scope', () => {
// `@teknium1:` is a handle with a colon, not a directive — inventing a
// scope for it would make Backspace eat the whole word.
expect(detectTrigger('@teknium1:')?.scope).toBeUndefined()
expect(detectTrigger('@teknium1:')?.value).toBe('teknium1:')
expect(detectTrigger('@localhost:8080')?.scope).toBeUndefined()
})
it('still ends the at-mention token at whitespace', () => {
@@ -80,15 +134,28 @@ describe('detectTrigger', () => {
expect(detectTrigger('look at @apps/desktop')).toEqual({
kind: '@',
query: 'apps/desktop',
tokenLength: 13
tokenLength: 13,
value: 'apps/desktop'
})
})
it('treats a mid-message slash as an inline reference', () => {
// Skills have to be reachable anywhere in a prompt, not just at position 0.
expect(detectTrigger('hello /')).toEqual({ kind: '/', inline: true, query: '', tokenLength: 1 })
expect(detectTrigger('hello /clean')).toEqual({ kind: '/', inline: true, query: 'clean', tokenLength: 6 })
expect(detectTrigger('text\n/skill')).toEqual({ kind: '/', inline: true, query: 'skill', tokenLength: 6 })
expect(detectTrigger('hello /')).toEqual({ kind: '/', inline: true, query: '', tokenLength: 1, value: '' })
expect(detectTrigger('hello /clean')).toEqual({
kind: '/',
inline: true,
query: 'clean',
tokenLength: 6,
value: 'clean'
})
expect(detectTrigger('text\n/skill')).toEqual({
kind: '/',
inline: true,
query: 'skill',
tokenLength: 6,
value: 'skill'
})
})
it('does not carry arg completion into an inline slash reference', () => {
@@ -1,14 +1,35 @@
import { DATA_IMAGE_URL_RE, dataUrlToBlob } from '@/lib/embedded-images'
import { $reactionsEnabled } from '@/store/reactions-enabled'
import { serializeTextBefore } from './rich-editor'
export interface TriggerState {
/** True for a `/` typed mid-message — an inline skill/command reference in
* prose rather than a command invocation. Arg completion doesn't apply. */
inline?: boolean
kind: '@' | '/'
kind: '@' | '/' | ':'
query: string
/** The `@kind:` prefix the user scoped the browse to, when there is one. */
scope?: DirectiveScope
tokenLength: number
/** `query` minus the `scope:` prefix — the value actually being typed. */
value: string
}
/** Directive kinds the `@` popover can scope a browse to. Mirrors the starter
* rows in use-at-completions and the gateway's `complete.path` prefixes. */
export const DIRECTIVE_SCOPES = ['file', 'folder', 'url', 'image', 'tool', 'git'] as const
export type DirectiveScope = (typeof DIRECTIVE_SCOPES)[number]
// Picking "attach a folder" types `@folder:` into the editor, and everything
// after it is the value being browsed. Parsing that prefix off the query is
// what lets the rest of the composer treat it as the BROWSE MODE it is rather
// than characters the user has to maintain by hand — Tab-descending has to
// carry it down, Backspace has to drop it whole, and a chip landing on it has
// to consume it.
const AT_SCOPE_RE = new RegExp(`^(${DIRECTIVE_SCOPES.join('|')}):(.*)$`)
// `@` triggers stop at the first whitespace — `@file:path` and `@diff` are
// single tokens, and a path is part of that token: `@./src/`, `@~/Desktop/`,
// and `@file:src/foo` all have to keep the popover live while the user walks
@@ -36,9 +57,18 @@ export interface TriggerState {
// The inline shape is what makes skills reachable anywhere in a prompt. Both
// shapes need the trailing `$`: detection runs against the text BEFORE the
// caret, so the match must end where the user is typing.
const AT_TRIGGER_RE = /(?:^|[\s])(@)([^\s@]*)$/
//
// U+FFFC is the placeholder textBeforeCaret emits for a committed chip. A chip
// edge is a token boundary just like whitespace (upstream assistant-ui's
// Lexical DirectivePlugin gets the same semantics from node boundaries), so
// `@` or `/` typed immediately after a pill still opens the popover.
const AT_TRIGGER_RE = /(?:^|[\s\uFFFC])(@)([^\s@\uFFFC]*)$/
const SLASH_COMMAND_TRIGGER_RE = /^(\/)((?:[a-zA-Z][\w-]*(?:\s+\S*)*)?)$/
const SLASH_INLINE_TRIGGER_RE = /[\s](\/)([a-zA-Z][\w-]*)?$/
const SLASH_INLINE_TRIGGER_RE = /[\s\uFFFC](\/)([a-zA-Z][\w-]*)?$/
// `:joy` → emoji completions, Slack-style. Boundary-anchored so a mid-word
// colon (`localhost:8080`, `note:`) never fires; two chars minimum so a bare
// `:` or `:D` smiley doesn't open a popover the user didn't ask for.
const EMOJI_TRIGGER_RE = /(?:^|[\s\uFFFC])(:)([a-zA-Z0-9_+-]{2,})$/
/** Stable key for paste dedupe — `items` and `files` often mirror the same image as different objects. */
export function blobDedupeKey(blob: Blob): string {
@@ -112,7 +142,17 @@ export function extractClipboardImageBlobs(clipboard: DataTransfer): Blob[] {
return blobs
}
/** Caret-anchored text before the cursor, or null if the selection isn't a collapsed caret inside `editor`. */
/** Caret-anchored text before the cursor, or null if the selection isn't a
* collapsed caret inside `editor`.
*
* Chips are ATOMIC to trigger detection: a committed pill must not leak its
* label text into the string the trigger regexes see. A `/work` pill whose
* label serialized into this text made the `^`-anchored command regex treat
* everything after it as that command's argument — which silenced the `@`
* popover for the rest of the message (`/work @Desk` → no trigger → the
* typed path never chips and submits as plain text). Each chip contributes
* an object-replacement placeholder instead, and <br> contributes a newline
* so a trigger at the start of a wrapped line still detects. */
export function textBeforeCaret(editor: HTMLDivElement): string | null {
const sel = window.getSelection()
const range = sel?.rangeCount ? sel.getRangeAt(0) : null
@@ -121,11 +161,16 @@ export function textBeforeCaret(editor: HTMLDivElement): string | null {
return null
}
const before = range.cloneRange()
before.selectNodeContents(editor)
before.setEnd(range.startContainer, range.startOffset)
return serializeTextBefore(editor, range.startContainer, range.startOffset)
}
return before.toString()
/** How many characters of directive scope the caret is sitting inside (`@url:`
* with nothing typed after it), or 0. A paste lands INTO that scope: the scope
* text is consumed rather than left in front of the chip as leftover syntax. */
export function openDirectiveScope(editor: HTMLDivElement): number {
const trigger = detectTrigger(textBeforeCaret(editor) ?? '')
return trigger?.kind === '@' && trigger.scope && !trigger.value ? trigger.tokenLength : 0
}
export function detectTrigger(textBefore: string): TriggerState | null {
@@ -138,19 +183,37 @@ export function detectTrigger(textBefore: string): TriggerState | null {
if (inline) {
const query = inline[2] ?? ''
return { inline: true, kind: '/', query, tokenLength: 1 + query.length }
return { inline: true, kind: '/', query, tokenLength: 1 + query.length, value: query }
}
const command = SLASH_COMMAND_TRIGGER_RE.exec(textBefore)
if (command) {
return { kind: '/', query: command[2], tokenLength: 1 + command[2].length }
return { kind: '/', query: command[2], tokenLength: 1 + command[2].length, value: command[2] }
}
const at = AT_TRIGGER_RE.exec(textBefore)
if (at) {
return { kind: '@', query: at[2], tokenLength: 1 + at[2].length }
const query = at[2]
const scoped = AT_SCOPE_RE.exec(query)
return {
kind: '@',
query,
...(scoped ? { scope: scoped[1] as DirectiveScope } : {}),
tokenLength: 1 + query.length,
value: scoped ? (scoped[2] ?? '') : query
}
}
// After `@` so a directive starter's colon (`@file:`) stays an `@` query.
// Rides the reactions opt-in (Settings → Appearance) — both are one
// "emoji features" surface, off by default.
const emoji = $reactionsEnabled.get() ? EMOJI_TRIGGER_RE.exec(textBefore) : null
if (emoji) {
return { kind: ':', query: emoji[2], tokenLength: 1 + emoji[2].length, value: emoji[2] }
}
return null
@@ -0,0 +1,151 @@
import { render, screen } from '@testing-library/react'
import { describe, expect, it, vi } from 'vitest'
import { ComposerTriggerPopover } from './trigger-popover'
vi.mock('@/i18n', () => ({
useI18n: () => ({
t: {
composer: {
lookupLoading: 'Loading…',
lookupNoMatches: 'No matches',
lookupTry: 'Try',
lookupOr: 'or'
}
}
})
}))
function atItem(type: string, display: string, rawText: string, meta = '') {
return {
id: `${rawText}|0`,
type,
label: display,
metadata: { icon: type, display, meta, rawText, insertId: display }
}
}
function slashItem(command: string, group: string, meta = '') {
return {
id: `${command}|0`,
type: 'slash',
label: command.slice(1),
metadata: { command, display: command, meta, group, action: '', rawText: command }
}
}
const noop = () => {}
/** The rendered shape of one row: does it have an icon, and how is it laid out?
* Icons are codicons — an `<i class="codicon codicon-<name>">`, not an SVG. */
function rowShape(root: HTMLElement) {
const row = root.querySelector('button') as HTMLElement
const icon = row.querySelector('i.codicon')
return {
hasIcon: Boolean(icon),
iconName: icon?.className.match(/codicon-([\w-]+)/)?.[1],
classes: row.className
}
}
describe('@ and / are one menu', () => {
it('a slash row has an icon, just like an @ row', () => {
const at = render(
<ComposerTriggerPopover
activeIndex={0}
items={[atItem('folder', 'apps/desktop/', '@folder:apps/desktop/', 'dir')]}
kind="@"
loading={false}
onHover={noop}
onPick={noop}
/>
)
const atShape = rowShape(at.container)
at.unmount()
const slash = render(
<ComposerTriggerPopover
activeIndex={0}
items={[slashItem('/work', 'Skills', 'Start in a worktree')]}
kind="/"
loading={false}
onHover={noop}
onPick={noop}
/>
)
const slashShape = rowShape(slash.container)
// The whole point: `/` used to render a stacked, icon-less row.
expect(slashShape.hasIcon).toBe(true)
expect(atShape.hasIcon).toBe(true)
expect(slashShape.classes).toBe(atShape.classes)
// And the glyph reflects the kind, not one generic bullet.
expect(slashShape.iconName).toBe('zap')
expect(atShape.iconName).toBe('folder')
})
it('renders the name and description for both kinds', () => {
const { rerender } = render(
<ComposerTriggerPopover
activeIndex={0}
items={[slashItem('/work', 'Skills', 'Start in a worktree')]}
kind="/"
loading={false}
onHover={noop}
onPick={noop}
/>
)
expect(screen.getByText('/work')).toBeTruthy()
expect(screen.getByText('Start in a worktree')).toBeTruthy()
rerender(
<ComposerTriggerPopover
activeIndex={0}
items={[atItem('file', 'src/main.tsx', '@file:src/main.tsx', 'src')]}
kind="@"
loading={false}
onHover={noop}
onPick={noop}
/>
)
expect(screen.getByText('src/main.tsx')).toBeTruthy()
expect(screen.getByText('src')).toBeTruthy()
})
it('an emoji row stays icon-less — the emoji IS the icon', () => {
const { container } = render(
<ComposerTriggerPopover
activeIndex={0}
items={[{ id: ':joy:|0', type: 'emoji', label: '😂 :joy:', metadata: { display: '😂 :joy:' } }]}
kind=":"
loading={false}
onHover={noop}
onPick={noop}
/>
)
expect(rowShape(container).hasIcon).toBe(false)
})
it('labels the active browse scope from the shared vocabulary', () => {
render(
<ComposerTriggerPopover
activeIndex={0}
items={[atItem('folder', 'apps/', '@folder:apps/', 'dir')]}
kind="@"
loading={false}
onHover={noop}
onPick={noop}
scope="folder"
/>
)
expect(screen.getByText('Folders')).toBeTruthy()
})
})
@@ -1,39 +1,14 @@
import type { Unstable_TriggerItem } from '@assistant-ui/core'
import { Fragment } from 'react'
import { referenceKind, referenceStyle } from '@/components/assistant-ui/reference-kinds'
import { Codicon } from '@/components/ui/codicon'
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
import { useI18n } from '@/i18n'
import { cn } from '@/lib/utils'
import { COMPLETION_DRAWER_BELOW_CLASS, COMPLETION_DRAWER_CLASS, CompletionDrawerEmpty } from './completion-drawer'
const AT_ICON_BY_TYPE: Record<string, string> = {
diff: 'diff',
file: 'book',
folder: 'folder',
git: 'git-branch',
image: 'file-media',
simple: 'symbol-misc',
staged: 'diff-added',
tool: 'tools',
url: 'globe'
}
function atIcon(item: Unstable_TriggerItem) {
const meta = item.metadata as { rawText?: string } | undefined
const raw = meta?.rawText || item.label
if (raw.startsWith('@diff')) {
return AT_ICON_BY_TYPE.diff
}
if (raw.startsWith('@staged')) {
return AT_ICON_BY_TYPE.staged
}
return AT_ICON_BY_TYPE[item.type] || AT_ICON_BY_TYPE.simple
}
import type { DirectiveScope } from './text-utils'
interface RowMeta {
display?: string
@@ -41,22 +16,67 @@ interface RowMeta {
meta?: string
}
const ROW_BASE_CLASS = [
'relative flex w-full cursor-default select-none rounded-md px-2 py-1 text-left',
/** The kind a row represents, for its icon. `@` rows carry it as the item type;
* `/` rows carry it as the completion group (Skills / Themes / Commands). */
function rowKind(item: Unstable_TriggerItem, isSlash: boolean): string {
const meta = item.metadata as (RowMeta & { rawText?: string }) | undefined
if (isSlash) {
const group = meta?.group?.trim()
return group === 'Skills' ? 'skill' : group === 'Themes' ? 'theme' : 'command'
}
// The gateway's simple refs (`@diff`, `@staged`) share one item type, so the
// glyph comes from the directive itself.
const raw = meta?.rawText || item.label
if (raw.startsWith('@diff')) {
return 'diff'
}
if (raw.startsWith('@staged')) {
return 'staged'
}
return item.type
}
const ROW_CLASS = [
'relative flex w-full cursor-default select-none items-center gap-2 rounded-md px-2 py-1 text-left',
'outline-hidden transition-colors hover:bg-(--ui-bg-tertiary)',
'data-[highlighted]:bg-(--ui-bg-tertiary) data-[highlighted]:text-foreground'
].join(' ')
const GROUP_HEADER_CLASS =
'select-none px-2 pb-0.5 text-[0.625rem] font-semibold uppercase tracking-wider text-(--ui-text-tertiary)'
interface ComposerTriggerPopoverProps {
activeIndex: number
items: readonly Unstable_TriggerItem[]
kind: '@' | '/'
kind: '@' | '/' | ':'
loading: boolean
onHover: (index: number) => void
onPick: (item: Unstable_TriggerItem) => void
placement?: 'bottom' | 'top'
/** The `@kind:` browse the list is filtered to, when there is one. Rendered
* as a header so the scope reads as the mode it is — the raw `@folder:` in
* the editor otherwise looks like syntax the user has to finish by hand. */
scope?: DirectiveScope
}
/**
* The composer's completion list, for every trigger.
*
* `@` and `/` render through the SAME row: icon, name, description. They used
* to be two layouts in one file — `@` horizontal with an icon, `/` stacked with
* none — which is why picking a file and picking a skill felt like features
* from different apps. Icons and accents come from the shared reference
* vocabulary, so a row looks like the chip it will become.
*
* `:` emoji is the one exception: the emoji IS the icon, so it renders as a
* single display string (Slack's exact shape).
*/
export function ComposerTriggerPopover({
activeIndex,
items,
@@ -64,11 +84,13 @@ export function ComposerTriggerPopover({
loading,
onHover,
onPick,
placement = 'top'
placement = 'top',
scope
}: ComposerTriggerPopoverProps) {
const { t } = useI18n()
const copy = t.composer
const isSlash = kind === '/'
const isEmoji = kind === ':'
let lastGroup: string | undefined
@@ -80,6 +102,7 @@ export function ComposerTriggerPopover({
onMouseDown={event => event.preventDefault()}
role="listbox"
>
{scope && <div className={cn(GROUP_HEADER_CLASS, 'pt-0.5')}>{referenceStyle(scope).label}</div>}
{items.length === 0 ? (
loading ? (
<div className="flex items-center gap-2 px-2 py-1.5 text-(--ui-text-tertiary)">
@@ -93,6 +116,10 @@ export function ComposerTriggerPopover({
{copy.lookupTry} <span className="font-mono text-foreground/80">@file:</span> {copy.lookupOr}{' '}
<span className="font-mono text-foreground/80">@folder:</span>.
</>
) : isEmoji ? (
<>
{copy.lookupTry} <span className="font-mono text-foreground/80">:joy:</span>.
</>
) : (
<>
{copy.lookupTry} <span className="font-mono text-foreground/80">/help</span>.
@@ -110,58 +137,28 @@ export function ComposerTriggerPopover({
const isFirstHeader = lastGroup === undefined
lastGroup = group || lastGroup
const active = index === activeIndex
const refKind = referenceKind(rowKind(item, isSlash))
return (
<Fragment key={item.id}>
{showHeader && (
<div
className={cn(
'select-none px-2 pb-0.5 text-[0.625rem] font-semibold uppercase tracking-wider text-(--ui-text-tertiary)',
isFirstHeader ? 'pt-0.5' : 'pt-2'
)}
>
{group}
</div>
)}
{showHeader && <div className={cn(GROUP_HEADER_CLASS, isFirstHeader ? 'pt-0.5' : 'pt-2')}>{group}</div>}
<button
className={cn(ROW_BASE_CLASS, isSlash ? 'flex-col gap-0' : 'items-center gap-2')}
className={ROW_CLASS}
data-highlighted={active ? '' : undefined}
onClick={() => onPick(item)}
onMouseEnter={() => onHover(index)}
type="button"
>
{isSlash ? (
<>
{/* Active row (keyboard nav or hover) un-truncates inline so
long command names / descriptions stay readable without a
floating tooltip. */}
<span
className={cn(
'font-medium leading-snug text-foreground',
active ? 'whitespace-normal break-words' : 'truncate'
)}
>
{display}
</span>
{description && (
<span
className={cn(
'leading-snug text-(--ui-text-tertiary)',
active ? 'whitespace-normal break-words' : 'truncate'
)}
>
{description}
</span>
)}
</>
{isEmoji ? (
// The emoji is its own icon — a glyph column beside it reads
// as decoration.
<span className="min-w-0 shrink truncate leading-5 text-foreground">{display}</span>
) : (
<>
<span className="grid size-4 shrink-0 place-items-center text-(--ui-text-tertiary)">
<Codicon name={atIcon(item)} size="0.875rem" />
</span>
<span className="min-w-0 shrink truncate font-mono font-medium leading-5 text-foreground">
{display}
<span className="grid size-4 shrink-0 place-items-center text-(--ref-color)" data-ref={refKind}>
<Codicon name={referenceStyle(refKind).codicon} size="0.875rem" />
</span>
<span className="min-w-0 shrink truncate font-medium leading-5 text-foreground">{display}</span>
{description && (
<span className="min-w-0 flex-1 truncate leading-5 text-(--ui-text-tertiary)">{description}</span>
)}
+10 -4
View File
@@ -39,7 +39,8 @@ import {
$sessions,
resolveComposerSessionKey,
sessionMatchesStoredId,
sessionPinId
sessionPinId,
shouldMigrateComposerScope
} from '@/store/session'
import { isSecondaryWindow, isWatchWindow } from '@/store/windows'
import type { ModelOptionsResponse } from '@/types/hermes'
@@ -326,14 +327,19 @@ export function ChatView({
// When the tip row arrives after compression, migrate any tip-keyed stash onto
// the durable lineage key before the composer remounts onto that key.
//
// ONLY same-conversation rekeys (tip → root). The route-driven queueSessionKey
// can flip to Session B a frame before the store selection leaves Session A;
// migrating on bare inequality would re-home A's queued prompts onto B and
// auto-drain them into the wrong chat.
useEffect(() => {
if (!selectedSessionId || !queueSessionKey || selectedSessionId === queueSessionKey) {
if (!shouldMigrateComposerScope(selectedSessionId, queueSessionKey, sessions)) {
return
}
migrateSessionDraft(selectedSessionId, queueSessionKey)
migrateQueuedPrompts(selectedSessionId, queueSessionKey)
}, [queueSessionKey, selectedSessionId])
}, [queueSessionKey, selectedSessionId, sessions])
// Transcript-side stops (the streaming message's hover Stop, the runtime's
// cancel) are explicit halts, same as the composer's Stop button: park any
@@ -542,7 +548,7 @@ export function ChatView({
config={COMPOSER_HEART_CONFIG}
style={{
top: 0,
bottom: 'calc(var(--composer-measured-height) + var(--status-stack-measured-height) + 0.25rem)'
bottom: 'calc(var(--composer-measured-height) + 0.25rem)'
}}
/>
)}
@@ -11,8 +11,8 @@ import { $threadJumpButtonVisible, requestScrollToBottom } from '@/store/thread-
/**
* Floating "jump to bottom" control. Sits centered just above the composer,
* clearing the out-of-flow status stack via the same measured-height CSS vars
* the thread's bottom clearance uses (`--composer-measured-height` +
* `--status-stack-measured-height`), so it never overlaps the queue / subagent
* the thread's bottom clearance uses (`--composer-measured-height`, which
* covers the whole dock), so it never overlaps the queue / subagent
* / background cards. Visible only while the user has scrolled meaningfully
* away from the bottom; clicking re-arms sticky-bottom and pins the viewport.
*
@@ -62,7 +62,7 @@ export function ScrollToBottomButton() {
requestScrollToBottom()
}}
style={{
bottom: 'calc(var(--composer-measured-height) + var(--status-stack-measured-height) + 0.625rem)'
bottom: 'calc(var(--composer-measured-height) + 0.625rem)'
}}
tabIndex={visible ? 0 : -1}
type="button"
+11 -5
View File
@@ -27,7 +27,7 @@ import { ModelMenuPanel } from '@/app/shell/model-menu-panel'
import { formatRefValue } from '@/components/assistant-ui/directive-text'
import { CenteredThreadSpinner } from '@/components/assistant-ui/thread/status'
import { findGroupOfPane } from '@/components/pane-shell/tree/model'
import { $layoutTree, moveTreePane, setTreeGroupHeaderHidden } from '@/components/pane-shell/tree/store'
import { $layoutTree, closeTreePane, moveTreePane, setTreeGroupHeaderHidden } from '@/components/pane-shell/tree/store'
import { Button } from '@/components/ui/button'
import { ConfirmDialog } from '@/components/ui/confirm-dialog'
import { transcribeAudio } from '@/hermes'
@@ -504,9 +504,10 @@ export function SessionTabMenu({
}
/** The MAIN tab's menu: the same session verbs targeting the primary's loaded
* session, plus the bar's off switch (the bar sticky-shows once a tab is
* ever gained; this is the explicit way back). A fresh draft has no session —
* no menu. */
* session, plus Close (the tab empties to a fresh draft — the workspace pane
* itself never leaves the tree) and the bar's off switch (the bar sticky-shows
* once a tab is ever gained; this is the explicit way back). A fresh draft has
* no session — no menu. */
export function WorkspaceTabMenu({ children }: { children: React.ReactElement }) {
const selected = useStore($selectedStoredSessionId)
@@ -524,7 +525,12 @@ export function WorkspaceTabMenu({ children }: { children: React.ReactElement })
}
return (
<SessionTabMenu onHideTabBar={hideTabBar} storedSessionId={selected} tabPaneId="workspace">
<SessionTabMenu
onClose={() => closeTreePane('workspace')}
onHideTabBar={hideTabBar}
storedSessionId={selected}
tabPaneId="workspace"
>
{children}
</SessionTabMenu>
)
+1 -1
View File
@@ -10,7 +10,7 @@ import { cn } from '@/lib/utils'
/** The muted slot beside a section label (loading glyph, status hint). */
export function SidebarSectionMeta({ children }: { children: React.ReactNode }) {
return <span className="text-[0.6875rem] font-medium text-(--ui-text-quaternary)">{children}</span>
return <span className="shrink-0 text-[0.6875rem] font-medium text-(--ui-text-quaternary)">{children}</span>
}
// ── Row geometry (session row is canonical — everything composes these) ─────
@@ -135,7 +135,7 @@ export function SidebarCronJobsSection({
<SidebarGroup className="shrink-0 p-0 pb-1">
<div className="group/section flex shrink-0 items-center justify-between pb-1 pt-1.5">
<button
className="group/section-label flex w-fit items-center gap-1 bg-transparent text-left leading-none"
className="group/section-label flex w-fit min-w-0 items-center gap-1 bg-transparent text-left leading-none"
onClick={onToggle}
type="button"
>
+13 -16
View File
@@ -46,6 +46,7 @@ import {
$sidebarSessionOrderManual,
$sidebarWorkspaceOrderIds,
$sidebarWorkspaceParentOrderIds,
filterVisibleProjects,
pinSession,
SESSION_SEARCH_FOCUS_EVENT,
setPinnedSessionOrder,
@@ -609,23 +610,19 @@ export function ChatSidebar({
return []
}
const dismissed = new Set(dismissedAutoProjects)
const sorted = sortProjectsForOverview(
projectTree
.filter(node => !(node.isAuto && dismissed.has(node.id)))
.map(project =>
excludeProjectSessions(
{
...project,
// Home is synthetic, so its name is ours to translate — every other
// label is a repo basename or a name the user typed.
label: project.isNoProject ? s.projects.home : project.label,
repos: orderRepos(project.repos)
},
isPinnedSession
)
),
filterVisibleProjects(projectTree, dismissedAutoProjects).map(project =>
excludeProjectSessions(
{
...project,
// Home is synthetic, so its name is ours to translate — every other
// label is a repo basename or a name the user typed.
label: project.isNoProject ? s.projects.home : project.label,
repos: orderRepos(project.repos)
},
isPinnedSession
)
),
activeProjectId
)
@@ -17,7 +17,7 @@ vi.mock('@/i18n', () => ({
projects: {
enter: (label: string) => `Enter ${label}`,
reorder: (label: string) => `Reorder ${label}`,
toggle: (label: string) => `Toggle ${label} sessions`
toggle: (label: string, open: boolean) => `${open ? 'Show' : 'Hide'} ${label} sessions`
}
}
}
@@ -60,14 +60,15 @@ describe('ProjectOverviewRow', () => {
/>
)
const button = screen.getByRole('button', { name: 'Toggle Test D sessions' })
// Collapsed by default, so the disclosure offers to show the sessions.
const button = screen.getByRole('button', { name: 'Show Test D sessions' })
expect(tipTrigger(button)).toBeTruthy()
})
it('does not render the disclosure toggle when there is nothing to preview', () => {
render(<ProjectOverviewRow project={project} />)
expect(screen.queryByRole('button', { name: 'Toggle Test D sessions' })).toBeNull()
expect(screen.queryByRole('button', { name: 'Show Test D sessions' })).toBeNull()
})
it('drops the "new session" add button on Home, which has no folder to start in', () => {

Some files were not shown because too many files have changed in this diff Show More