Merge updated tool metrics into skill metrics

Signed-off-by: Alex Fournier <afournier@nvidia.com>

# Conflicts:
#	tests/tools/test_skills_hub.py
This commit is contained in:
Alex Fournier
2026-08-02 20:12:37 -07:00
881 changed files with 88099 additions and 10987 deletions
@@ -15,6 +15,9 @@ outputs:
python:
description: Run Python tests / ruff / ty / windows-footguns.
value: ${{ steps.classify.outputs.python }}
python_prod:
description: Python changes outside tests/ — gates product jobs (Desktop E2E, Docker).
value: ${{ steps.classify.outputs.python_prod }}
frontend:
description: Run the TypeScript testing matrix + desktop build.
value: ${{ steps.classify.outputs.frontend }}
+18 -3
View File
@@ -42,6 +42,7 @@ jobs:
timeout-minutes: 10
outputs:
python: ${{ steps.classify.outputs.python }}
python_prod: ${{ steps.classify.outputs.python_prod }}
frontend: ${{ steps.classify.outputs.frontend }}
site: ${{ steps.classify.outputs.site }}
scan: ${{ steps.classify.outputs.scan }}
@@ -89,7 +90,19 @@ jobs:
e2e-desktop:
name: Desktop E2E
needs: detect
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true'
# python_prod (not python): the Playwright suite exercises the built app
# + `hermes serve` backend, which never import anything under tests/.
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
# single job in the workflow — while still running the full pytest lanes.
#
# ⛔ TEMPORARILY DISABLED (Aug 2, 2026, Teknium) — the suite is red on
# every PR and on main itself since the Aug 1 night engines/npm churn
# (#76499 → #76562 → #76575): the mock-backend Electron window never
# gets a title, so boot/chat/setup/interim specs all fail identically
# regardless of the PR's diff (verified on #76573 and the docs-only
# #76582). Tracking issue: #76627 (assigned: Ari). To re-enable,
# delete the `false &&` below — nothing else changed.
if: ${{ false && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true') }}
uses: ./.github/workflows/e2e-desktop.yml
docs-site:
@@ -137,8 +150,10 @@ jobs:
needs: detect
# Trusted main pushes run docker.yml directly so its container-publish
# environment secrets never cross this reusable-workflow call. PR runs
# remain build/test-only and secret-free.
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true')
# remain build/test-only and secret-free. Gated on python_prod (not
# python): the image copies installed code, never tests/ — tests-only
# PRs skip the build.
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true')
uses: ./.github/workflows/docker.yml
supply-chain:
+3 -1
View File
@@ -67,6 +67,8 @@ jobs:
echo -e "$MISSING"
echo ""
echo "Add a mapping file (do NOT edit AUTHOR_MAP in release.py):"
echo " python3 scripts/audit_pr_attribution.py --fix # auto-resolve + create files"
echo "or manually:"
echo -e "$MISSING" | while read -r line; do
email=$(echo "$line" | sed 's/^ *//' | cut -d' ' -f1)
[ -z "$email" ] && continue
@@ -78,7 +80,7 @@ jobs:
# Emit review_status for unmapped emails
DETAIL=$(echo -e "$MISSING" | sed '/^$/d; s/^ //')
HOW_TO_FIX=$'Add mappings to scripts/release.py AUTHOR_MAP:\n```\n"<email>": "<github-username>",\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
HOW_TO_FIX=$'Run from the PR branch:\n```\npython3 scripts/audit_pr_attribution.py --fix\ngit add contributors && git commit -m "chore: map contributor emails" && git push\n```\nOr map one email manually (do NOT edit AUTHOR_MAP in release.py):\n```\npython3 scripts/add_contributor.py <email> <github-username>\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
REVIEW_STATUS=$(jq -nc \
--arg detail "$DETAIL" \
--arg how_to_fix "$HOW_TO_FIX" \
+5 -1
View File
@@ -65,10 +65,14 @@ jobs:
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
cache-dependency-path: website/package-lock.json
- name: grab npm 12
run: |
npm i -g npm@12
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version: '3.11'
+5 -1
View File
@@ -15,10 +15,14 @@ jobs:
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
cache-dependency-path: website/package-lock.json
- name: grab npm 12
run: |
npm i -g npm@12
- name: Install website dependencies
uses: ./.github/actions/retry
with:
+11 -6
View File
@@ -39,8 +39,13 @@ jobs:
# ── Node ───────────────────────────────────────────────────────────
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
npm i -g npm@12
# Full npm ci (not --ignore-scripts): electron's postinstall
# downloads the binary we launch, and node-pty's native build is
# needed for the terminal pane.
@@ -56,7 +61,7 @@ jobs:
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
version: '0.9.28'
enable-cache: true
cache-dependency-glob: |
pyproject.toml
@@ -106,11 +111,11 @@ jobs:
npx playwright test --reporter=list
fi
env:
CI: "true"
CI: 'true'
# Ensure no real API keys leak into the test env.
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
NOUS_API_KEY: ""
OPENROUTER_API_KEY: ''
OPENAI_API_KEY: ''
NOUS_API_KEY: ''
# ── Save updated baselines to cache (main only) ───────────────────
- name: Save updated baselines to cache
+5 -1
View File
@@ -67,9 +67,13 @@ jobs:
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
npm i -g npm@12
# --ignore-scripts: eslint only needs TS sources + eslint packages.
- uses: ./.github/actions/retry
with:
+12 -2
View File
@@ -15,8 +15,13 @@ jobs:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
npm i -g npm@12
- uses: ./.github/actions/retry
with:
command: npm ci --ignore-scripts
@@ -61,8 +66,13 @@ jobs:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
npm i -g npm@12
- uses: ./.github/actions/retry
with:
command: npm ci
+3 -1
View File
@@ -43,11 +43,13 @@ jobs:
uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
with:
# Scan explicit lockfiles rather than recursing, so we only look at
# the three sources of truth and skip vendored / test / worktree dirs.
# the five sources of truth and skip vendored / test / worktree dirs.
scan-args: |-
--lockfile=uv.lock
--lockfile=package-lock.json
--lockfile=website/package-lock.json
--lockfile=plugins/platforms/photon/sidecar/package-lock.json
--lockfile=scripts/whatsapp-bridge/package-lock.json
# The upstream reusable workflow uploads this exact file under its
# fixed artifact name, which the wrapper downloads below.
results-file-name: osv-results.sarif
+51
View File
@@ -0,0 +1,51 @@
# needed to prevent bad npm that has min-release-age but not exclude
engine-strict=true
min-release-age=14
# allow assistant-ui packages & a couple specific deps since they update a LOT.
# remove this when we stabilize (or we haven't updated in 2 wks)
min-release-age-exclude[]=@assistant-ui/*
min-release-age-exclude[]=assistant-cloud
min-release-age-exclude[]=assistant-stream
min-release-age-exclude[]=@radix-ui/*
min-release-age-exclude[]=radix-ui
min-release-age-exclude[]=safe-content-frame
# react-router 8.3.0 includes fixes for vulns. remove this when 8.3.0 is > 2wks old.
min-release-age-exclude[]=react-router
# eslint 10.8.0 includes fixes for vulns. remove this when 10.8.0 is > 2wks old.
min-release-age-exclude[]=eslint
min-release-age-exclude[]=@eslint/*
# tar 7.5.21 includes fixes for vulns. remove this when 7.5.21 is > 2wks old
min-release-age-exclude[]=tar
# concurrently 10.0.4 includes fixes for vulns. remove this when 10.0.4 is > 2wks old
min-release-age-exclude[]=concurrently
# fast-uri 3.1.4 includes fixes for vulns. remove this when 3.1.4 is > 2wks old
min-release-age-exclude[]=fast-uri
# minimatch 10.2.6 includes fixes for vulns. remove this when 10.2.6 is > 2wks old
min-release-age-exclude[]=minimatch
# brace-expansion 5.0.8 includes fixes for vulns. remove this when 5.0.8 is > 2wks old
min-release-age-exclude[]=brace-expansion
# vite 8.2.0 is the first release depending on rolldown >= 1.2.1, which fixes
# a rolldown panic that breaks `npm run build` in apps/desktop
# (rolldown/rolldown#10337 — a regression in 1.1.5, the version vite 8.1.5
# pins as ~1.1.5). @oxc-project/types is here because rolldown 1.2.1 pins it
# as `=0.142.0` — an exact pin, so no older release satisfies it and the age
# gate would fail the whole install with ETARGET.
# remove these once vite 8.2.0 is > 2wks old.
min-release-age-exclude[]=vite
min-release-age-exclude[]=rolldown
min-release-age-exclude[]=@rolldown/*
min-release-age-exclude[]=@oxc-project/types
# ink needs
min-release-age-exclude[]=lightningcss
min-release-age-exclude[]=postcss
+1
View File
@@ -0,0 +1 @@
26
+1
View File
@@ -0,0 +1 @@
3.11
+45 -30
View File
@@ -41,14 +41,14 @@ RUN apt-get -o Acquire::Retries=3 update && \
make install
FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source
# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x
# which reached EOL in April 2026 — we copy node + npm + corepack from the
# upstream node:22 image instead so we can stay on a supported LTS without
# waiting for Debian 14 (forky, ~mid-2027). Bookworm-based slim image used
# so the produced binary links against glibc 2.36, which runs cleanly on
# our Debian 13 (trixie, glibc 2.41) runtime. Bumping to a new Node major
# is a one-line ARG change; see #4977.
FROM node:22-bookworm-slim@sha256:7af03b14a13c8cdd38e45058fd957bf00a72bbe17feac43b1c15a689c029c732 AS node_source
# Node 26 source stage. Debian trixie's bundled nodejs is pinned to 20.x
# which reached EOL in April 2026 — we copy node + npm from the upstream
# node:26 image instead (Hermes pins its toolchain to Node 26 everywhere).
# Bookworm-based slim image used so the produced binary links
# against glibc 2.36, which runs cleanly on our Debian 13 (trixie, glibc
# 2.41) runtime. Bumping to a new Node major is a one-line ARG change; see
# #4977.
FROM node:26-bookworm-slim@sha256:9e6f9357d371591e32ab6f2d8a26d63bdd0d17c29eee3f4f3e7e454d9634bf73 AS node_source
FROM debian:13.4
# Disable Python stdout buffering to ensure logs are printed immediately.
@@ -70,7 +70,7 @@ ENV PLAYWRIGHT_BROWSERS_PATH=/opt/hermes/.playwright
# hermes process, the dashboard, and per-profile gateways.
RUN apt-get -o Acquire::Retries=3 update && \
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev procps git openssh-client docker-cli xz-utils && \
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev libatomic1 procps git openssh-client docker-cli xz-utils && \
rm -rf /var/lib/apt/lists/*
# Prefer the fixed SQLite over Debian's vulnerable libsqlite3.so.0. Keep the
@@ -151,17 +151,20 @@ RUN useradd -u 10000 -m -d /opt/data hermes
COPY --chmod=0755 --from=uv_source /usr/local/bin/uv /usr/local/bin/uvx /usr/local/bin/
# Node 22 LTS: copy the node binary plus the bundled npm + corepack JS
# installs from the upstream image. npm and npx are recreated as symlinks
# because they're symlinks in the source image (and need to live on PATH).
# Node 26: copy the node binary plus the bundled npm JS install from the
# upstream image. npm and npx are recreated as symlinks because they're
# symlinks in the source image (and need to live on PATH).
#
# No corepack: Node unbundled it upstream, so node:26 ships only npm in
# /usr/local/lib/node_modules. Nothing here needs it — no package.json
# declares a `packageManager`, and no build step shells out to yarn or pnpm.
#
# See node_source stage at the top of the file for the version-bump
# rationale (#4977).
COPY --chmod=0755 --from=node_source /usr/local/bin/node /usr/local/bin/
COPY --from=node_source /usr/local/lib/node_modules/npm /usr/local/lib/node_modules/npm
COPY --from=node_source /usr/local/lib/node_modules/corepack /usr/local/lib/node_modules/corepack
RUN ln -sf /usr/local/lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm && \
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx && \
ln -sf /usr/local/lib/node_modules/corepack/dist/corepack.js /usr/local/bin/corepack
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx
WORKDIR /opt/hermes
@@ -400,6 +403,8 @@ ENV HERMES_LAZY_INSTALL_TARGET=/opt/data/lazy-packages
# Recursion is impossible because the shim exec's the venv binary by
# absolute path (/opt/hermes/.venv/bin/hermes). See the shim source for
# the opt-out env var (HERMES_DOCKER_EXEC_AS_ROOT=1).
COPY --chmod=0755 docker/hermes-exec-shim.sh /opt/hermes/bin/hermes
COPY --chmod=0755 docker/entrypoint-dispatch.sh /opt/hermes/docker/entrypoint-dispatch.sh
# Pre-s6 entrypoint.sh did `source .venv/bin/activate` which exported
# the venv bin onto PATH; Architecture B's main-wrapper.sh does the
@@ -416,27 +421,37 @@ ENV PATH="/opt/hermes/bin:/opt/hermes/.venv/bin:/opt/data/.local/bin:${PATH}"
RUN mkdir -p /opt/data
VOLUME [ "/opt/data" ]
# s6-overlay's /init is PID 1. It sets up the supervision tree, runs
# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services
# declared in /etc/s6-overlay/s6-rc.d/, then exec's its remaining
# argv as the container's "main program" with stdin/stdout/stderr
# inherited (this is what makes interactive --tui work). When the
# main program exits, /init begins stage 3 shutdown and the container
# exits with the program's exit code. Replaces tini — see Phase 2 of
# docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md.
# The image ENTRYPOINT is a tiny dispatcher rather than `/init` directly.
# When the image really owns PID 1 (normal Docker / Podman), the dispatcher
# execs `/init` and preserves the full s6 supervision tree. When a platform
# wraps the image entrypoint under its own PID-1 init (Fly Machines,
# `docker run --init`, some schedulers), `/init` would abort with
# `can only run as pid 1`; in that case the dispatcher falls back to
# `stage2-hook.sh` + `main-wrapper.sh` directly so foreground commands still
# work. See #38349.
#
# On the PID-1 path, s6-overlay's /init sets up the supervision tree, runs
# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services declared in
# /etc/s6-overlay/s6-rc.d/, then exec's its remaining argv as the container's
# "main program" with stdin/stdout/stderr inherited (this is what makes
# interactive --tui work). When the main program exits, /init begins stage 3
# shutdown and the container exits with the program's exit code. Replaces
# tini — see Phase 2 of docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md.
#
# We use the ENTRYPOINT+CMD split rather than CMD alone so the
# wrapper is prepended to user-supplied args automatically:
#
# docker run <image> → /init main-wrapper.sh (CMD default)
# docker run <image> chat -q "hi" → /init main-wrapper.sh chat -q hi
# docker run <image> sleep infinity → /init main-wrapper.sh sleep infinity
# docker run <image> --tui → /init main-wrapper.sh --tui
# docker run <image> → entrypoint-dispatch.sh (CMD default)
# docker run <image> chat -q "hi" → entrypoint-dispatch.sh chat -q hi
# docker run <image> sleep infinity → entrypoint-dispatch.sh sleep infinity
# docker run <image> --tui → entrypoint-dispatch.sh --tui
#
# main-wrapper.sh handles arg routing (bare-exec vs. hermes
# subcommand vs. no-args), drops to the hermes user via s6-setuidgid,
# and exec's the final program so its exit code becomes the container
# exit code. Without the wrapper-as-ENTRYPOINT, leading-dash args
# like `--version` would be intercepted by /init's POSIX shell.
ENTRYPOINT [ "/init", "/opt/hermes/docker/main-wrapper.sh" ]
# exit code. The dispatcher preserves that contract across both the
# supervised PID-1 path and the non-PID-1 fallback path. Without the
# wrapper-as-ENTRYPOINT, leading-dash args like `--version` would be
# intercepted by /init's POSIX shell.
ENTRYPOINT [ "/opt/hermes/docker/entrypoint-dispatch.sh" ]
CMD [ ]
+13 -7
View File
@@ -247,16 +247,22 @@ def main(argv: list[str] | None = None) -> None:
import acp
from .server import HermesACPAgent
# MCP tool discovery from config.yaml — run before asyncio.run() so
# it's safe to use blocking waits. (ACP also registers per-session
# MCP servers dynamically via asyncio.to_thread inside the event
# loop; that path is unaffected.) Moved from model_tools.py module
# scope to avoid freezing the gateway's loop on lazy import (#16856).
# MCP tool discovery from config.yaml — fire-and-forget in a
# background daemon thread so the ACP server becomes responsive
# immediately while MCP servers connect. Previously this blocked
# asyncio.run() for 2-5 s. (ACP also registers per-session MCP
# servers dynamically via asyncio.to_thread inside the event loop;
# that path is unaffected.) Moved from model_tools.py module scope
# to avoid freezing the gateway's loop on lazy import (#16856).
# Metadata-only hosts can opt out of unrelated global MCP startup.
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
try:
from tools.mcp_tool import discover_mcp_tools
discover_mcp_tools()
from hermes_cli.mcp_startup import start_background_mcp_discovery
start_background_mcp_discovery(
logger=logger,
thread_name="acp-mcp-discovery",
)
except Exception:
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
+106 -3
View File
@@ -78,6 +78,7 @@ from agent.context_compressor import (
COMPRESSED_SUMMARY_METADATA_KEY,
ContextCompressor,
)
from agent.interrupt_compat import request_hard_interrupt
from tools.approval import (
reset_hermes_interactive_context,
set_hermes_interactive_context,
@@ -1037,6 +1038,102 @@ class HermesACPAgent(acp.Agent):
exc_info=True,
)
def _schedule_mcp_late_refresh(self, state: SessionState) -> None:
"""Refresh the agent's tool snapshot when background MCP discovery lands late.
ACP entry.py starts MCP tool discovery in a background daemon thread so a
slow/dead configured server can't block ``asyncio.run()``. ``_make_agent``
briefly joins that thread (``wait_for_mcp_discovery``, bounded ~1.5s) so
already-spawning fast servers land in the snapshot — but a server slower
than the bound lands *after* the agent is built, leaving its tools absent
for the whole session.
This schedules an off-critical-path daemon that waits for discovery to
finish (bounded 30s), then rebuilds the snapshot via the shared
``refresh_agent_mcp_tools`` helper — the same rebuild ``/reload-mcp``
performs, but automatic. Mirrors the TUI late-refresh (PR #48403).
Cache safety: the rebuild only runs while the session is still
pre-first-turn (no API call made yet → nothing cached to invalidate).
Once the user has sent a message we leave the snapshot frozen rather
than break the cached prompt prefix mid-conversation; servers that land
later are picked up cache-safely by the between-turns prologue refresh
(``agent/turn_context.py``) at the next turn boundary. The marginal
value of this pre-first-turn daemon is therefore freshness in the
window [session created → first message] — e.g. the "Available tools"
listing a client may request before the first prompt.
No-op when discovery already finished, when the join times out, when the
registry was unchanged, or when the session was closed while waiting.
"""
try:
from hermes_cli.mcp_startup import mcp_discovery_in_flight
except Exception:
return
if not mcp_discovery_in_flight():
return
import threading
agent = state.agent
session_id = state.session_id
def _wait_then_refresh() -> None:
try:
from hermes_cli.mcp_startup import join_mcp_discovery
if not join_mcp_discovery(timeout=30.0):
return
# Session may have been closed while we waited. In-memory-only
# lookup on purpose: ``get_session()`` falls through to a DB
# restore that builds a whole new AIAgent as a side effect just
# to decide "no-op" here (the TUI equivalent also checks its
# in-memory dict only).
with self.session_manager._lock:
current = self.session_manager._sessions.get(session_id)
if current is None or current.agent is not agent:
return
# Cache safety: never rebuild the tool list once the conversation
# has started — that would invalidate the cached prompt prefix.
# Serialized with turn start: ``prompt()`` flips ``is_running``
# under ``runtime_lock`` before dispatching, so holding it here
# (and bailing when a turn is already running) closes the window
# where the guard passes but the first prompt starts before the
# refresh publishes — which would swap ``tools=`` mid-turn and
# break the just-created cache prefix.
with current.runtime_lock:
if current.is_running:
return
if (
int(getattr(agent, "_user_turn_count", 0) or 0) > 0
or int(getattr(agent, "_api_call_count", 0) or 0) > 0
):
return
from tools.mcp_tool import refresh_agent_mcp_tools
added = refresh_agent_mcp_tools(agent, quiet_mode=True)
if added:
logger.info(
"Session %s: late MCP refresh added %d tools: %s",
session_id,
len(added),
", ".join(sorted(added)),
)
except Exception:
logger.debug(
"Session %s: late MCP refresh failed",
session_id,
exc_info=True,
)
threading.Thread(
target=_wait_then_refresh,
name=f"acp-mcp-late-refresh-{session_id}",
daemon=True,
).start()
# ---- ACP lifecycle ------------------------------------------------------
async def initialize(
@@ -1343,6 +1440,7 @@ class HermesACPAgent(acp.Agent):
) -> NewSessionResponse:
state = self.session_manager.create_session(cwd=cwd)
await self._register_session_mcp_servers(state, mcp_servers)
self._schedule_mcp_late_refresh(state)
logger.info("New session %s (cwd=%s)", state.session_id, cwd)
self._schedule_available_commands_update(state.session_id)
self._schedule_usage_update(state)
@@ -1367,6 +1465,7 @@ class HermesACPAgent(acp.Agent):
logger.warning("load_session: session %s not found", session_id)
return None
await self._register_session_mcp_servers(state, mcp_servers)
self._schedule_mcp_late_refresh(state)
logger.info("Loaded session %s", session_id)
# Per ACP spec, `session/load` must stream the prior conversation back
# to the client via `session/update` notifications BEFORE responding,
@@ -1414,6 +1513,7 @@ class HermesACPAgent(acp.Agent):
logger.warning("resume_session: session %s not found, creating new", session_id)
state = self.session_manager.create_session(cwd=cwd)
await self._register_session_mcp_servers(state, mcp_servers)
self._schedule_mcp_late_refresh(state)
logger.info("Resumed session %s", state.session_id)
# See `load_session` above for the spec rationale — replay must
# complete before the response so clients receive the full transcript
@@ -1448,8 +1548,8 @@ class HermesACPAgent(acp.Agent):
# redirectable work.
state.cancel_event.set()
try:
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
state.agent.interrupt()
if getattr(state, "agent", None):
request_hard_interrupt(state.agent)
except Exception:
logger.debug(
"Failed to interrupt ACP session %s",
@@ -1768,8 +1868,11 @@ class HermesACPAgent(acp.Agent):
# while the tools are rooted at the client's project, so the
# model emits absolute paths under ~/.hermes/workspace and the
# edit silently lands outside the editor's workspace.
# cron_session="" explicitly marks this as a non-cron context,
# masking any leaked process-global HERMES_CRON_SESSION (#37968).
session_tokens = set_session_vars(
session_key=session_id, cwd=state.cwd,
session_key=session_id, session_id=session_id, cwd=state.cwd,
cron_session="",
)
except Exception:
session_tokens = None
+24
View File
@@ -648,6 +648,30 @@ class SessionManager:
logger.debug("ACP session falling back to default provider resolution", exc_info=True)
_register_task_cwd(session_id, cwd)
# Bounded wait for background MCP discovery so already-spawning fast
# servers land in the agent's tool snapshot. ACP entry.py fires
# discovery in a background daemon thread (start_background_mcp_discovery);
# the agent snapshots tools once at build (run_agent/agent_init) and
# never re-reads the registry, so without this join a reachable-but-
# slow configured server would be invisible for the whole session.
# ``ensure_mcp_discovery_before_agent_build`` also (re)starts discovery
# when the entry.py spawn never ran or exited with zero connected
# servers (the retry-after-zero-connected allowance), making this
# construction site self-sufficient. Bounded by
# ``mcp_discovery_timeout`` (config.yaml, default ~1.5s) so a dead
# server can't block — servers that miss the bound are picked up by
# the automatic late-refresh (see HermesACPAgent._schedule_mcp_late_refresh).
try:
from hermes_cli.mcp_startup import ensure_mcp_discovery_before_agent_build
ensure_mcp_discovery_before_agent_build(
logger=logger,
thread_name="acp-mcp-discovery",
)
except Exception:
logger.debug("ACP: bounded MCP discovery wait failed", exc_info=True)
agent = AIAgent(**kwargs)
# Codex app-server sessions are spawned lazily on the first turn. Stamp
# the ACP workspace onto the agent so the Codex runtime starts from the
+50 -11
View File
@@ -33,6 +33,7 @@ from urllib.parse import parse_qs, urlparse, urlunparse
from agent.context_compressor import ContextCompressor
from agent.iteration_budget import IterationBudget
from agent.memory_manager import StreamingContextScrubber
from agent.session_activity import ActivityProvenance
from agent.model_metadata import (
MINIMUM_CONTEXT_LENGTH,
fetch_model_metadata,
@@ -765,6 +766,9 @@ def init_agent(
# Interrupt mechanism for breaking out of tool loops
agent._interrupt_requested = False
agent._interrupt_message = None # Optional message that triggered interrupt
# Explicit hard cancellation is separate from redirect/message state. A
# thread-safe Event makes the cause atomic for auxiliary stream pollers.
agent._hard_interrupt_requested = threading.Event()
agent._execution_thread_id: int | None = None # Set at run_conversation() start
agent._interrupt_thread_signal_pending = False
agent._client_lock = threading.RLock()
@@ -834,18 +838,34 @@ def init_agent(
agent._use_prompt_caching, agent._use_native_cache_layout = (
agent._anthropic_prompt_cache_policy()
)
agent._cache_disabled = False
# Anthropic supports "5m" (default) and "1h" cache TTL tiers. Read from
# config.yaml under prompt_caching.cache_ttl; unknown values keep "5m".
# 1h tier costs 2x on write vs 1.25x for 5m, but amortizes across long
# sessions with >5-minute pauses between turns (#14971).
#
# Setting cache_ttl to a falsy value (false / null / "off" / "disabled" /
# "no" / "none") disables prompt caching entirely. This is useful for
# OAuth subscription users where cache writes bill against "extra usage"
# or for third-party proxies that inject their own cache_control markers
# (#13477). The disable propagates through anthropic_prompt_cache_policy()
# and restore_primary_runtime() so it survives /model switches and
# fallback re-derivation (#33555).
agent._cache_ttl = "5m"
try:
from hermes_cli.config import load_config_readonly as _load_pc_cfg
from agent.agent_runtime_helpers import cache_ttl_means_disabled
_pc_cfg = _load_pc_cfg().get("prompt_caching", {}) or {}
_ttl = _pc_cfg.get("cache_ttl", "5m")
if _ttl in {"5m", "1h"}:
agent._cache_ttl = _ttl
elif cache_ttl_means_disabled(_ttl):
agent._use_prompt_caching = False
agent._use_native_cache_layout = False
agent._cache_ttl = None
agent._cache_disabled = True
except Exception:
pass
@@ -864,6 +884,11 @@ def init_agent(
# notifications to show progress.
agent._last_activity_ts: float = time.time()
agent._last_activity_desc: str = "initializing"
# Default / unmigrated paths and _touch_activity stamp unknown; named
# provenances are stamped by compression writers (heartbeat / timeout / cooldown).
agent._last_activity_provenance = ActivityProvenance.UNKNOWN
# Rate-limit durable SessionDB activity stamps from _touch_activity (#72016).
agent._session_activity_last_persist_mono: float = 0.0
agent._current_tool: str | None = None
agent._api_call_count: int = 0
# Opt-out flag for the between-turns MCP tool refresh (build_turn_context).
@@ -1221,16 +1246,20 @@ def init_agent(
_fb_entries = [fallback_model]
_fb_resolved = False
for _fb in _fb_entries:
_fb_explicit_key = (_fb.get("api_key") or "").strip() or None
if not _fb_explicit_key:
_fb_key_env = (_fb.get("key_env") or _fb.get("api_key_env") or "").strip()
if _fb_key_env:
_fb_explicit_key = os.getenv(_fb_key_env, "").strip() or None
_fb_client, _fb_model = resolve_provider_client(
_fb["provider"], model=_fb["model"], raw_codex=True,
explicit_base_url=_fb.get("base_url"),
explicit_api_key=_fb_explicit_key,
)
try:
from hermes_cli.fallback_config import resolve_entry_api_key
_fb_explicit_key = resolve_entry_api_key(_fb)
_fb_client, _fb_model = resolve_provider_client(
_fb["provider"], model=_fb["model"], raw_codex=True,
explicit_base_url=_fb.get("base_url"),
explicit_api_key=_fb_explicit_key,
)
except Exception as _fb_exc:
logger.debug(
"Init-time fallback entry %s failed: %s",
_fb.get("provider"), _fb_exc,
)
continue
if _fb_client is not None:
agent.provider = _fb["provider"]
agent.model = _fb_model or _fb["model"]
@@ -1465,7 +1494,17 @@ def init_agent(
set_current_session_id(agent.session_id)
except Exception:
os.environ["HERMES_SESSION_ID"] = agent.session_id
# Preserve the root-agent legacy fallback, but never let delegated
# construction publish a child ID process-wide even if the ContextVar
# bridge itself failed to import.
try:
from agent.delegation_context import is_delegated_child_context
delegated_child = is_delegated_child_context()
except Exception:
delegated_child = False
if not delegated_child:
os.environ["HERMES_SESSION_ID"] = agent.session_id
# Session logs go into ~/.hermes/sessions/ alongside gateway sessions
hermes_home = get_hermes_home()
+204 -15
View File
@@ -151,7 +151,7 @@ def convert_to_trajectory_format(agent, messages: List[Dict[str, Any]], user_que
except json.JSONDecodeError:
# This shouldn't happen since we validate and retry during conversation,
# but if it does, log warning and use empty dict
logger.warning(f"Unexpected invalid JSON in trajectory conversion: {tool_call['function']['arguments'][:100]}")
logger.warning("Unexpected invalid JSON in trajectory conversion: %s", tool_call['function']['arguments'][:100])
arguments = {}
tool_call_json = {
@@ -1210,7 +1210,7 @@ def recover_with_credential_pool(
refreshed_id,
)
return False, has_retried_429
_ra().logger.info(f"Credential auth failure — refreshed pool entry {getattr(refreshed, 'id', '?')}")
_ra().logger.info("Credential auth failure — refreshed pool entry %s", getattr(refreshed, 'id', '?'))
agent._swap_credential(refreshed)
return True, has_retried_429
# Refresh failed — rotate to next credential instead of giving up.
@@ -1475,6 +1475,12 @@ def restore_primary_runtime(agent) -> bool:
"use_native_cache_layout",
agent.api_mode == "anthropic_messages" and agent.provider == "anthropic",
)
# If the operator has disabled caching via config (cache_ttl is
# falsy → _cache_disabled flag is set), the disable must survive
# runtime snapshot restoration (#33555).
if getattr(agent, "_cache_disabled", False):
agent._use_prompt_caching = False
agent._use_native_cache_layout = False
# ── Rebuild client for the primary provider ──
if agent.provider == "moa":
@@ -1829,11 +1835,148 @@ def dump_api_request_debug(
return dump_file
except Exception as dump_error:
if agent.verbose_logging:
logger.warning(f"Failed to dump API request debug payload: {dump_error}")
logger.warning("Failed to dump API request debug payload: %s", dump_error)
return None
def _direct_native_anthropic_tool_cache_capability(
agent,
*,
provider: Optional[str] = None,
base_url: Optional[str] = None,
api_mode: Optional[str] = None,
model: Optional[str] = None,
) -> bool:
"""Return whether this resolved destination accepts native tool markers."""
eff_base_url = base_url if base_url is not None else (agent.base_url or "")
eff_api_mode = api_mode if api_mode is not None else (agent.api_mode or "")
return (
eff_api_mode == "anthropic_messages"
and base_url_hostname(eff_base_url) == "api.anthropic.com"
)
def cache_ttl_means_disabled(ttl: Any) -> bool:
"""Return True when a ``prompt_caching.cache_ttl`` value means caching off.
Single source of truth for the disable-synonym detection shared by
``agent_init`` (live-agent ``_cache_disabled`` flag) and the stub policy
paths below. Keeping one predicate prevents the two sites from drifting
(a synonym added in only one place would recreate #76085).
Unknown values (e.g. ``"2h"``, integers) are NOT a disable — callers keep
caching enabled with the default TTL, matching ``agent_init``.
"""
if ttl in ("5m", "1h"):
return False
if ttl is False or ttl is None:
return True
return str(ttl).lower() in ("off", "false", "disabled", "no", "none")
def prompt_caching_disabled_from_config() -> bool:
"""Return True when ``prompt_caching.cache_ttl`` is configured as off.
Same disable detection as ``agent_init`` (via ``cache_ttl_means_disabled``)
so stub-based policy paths (MoA slot decoration, auxiliary fallback
replan) honor the same config contract without holding a live
``AIAgent`` (#76085 / #33555).
"""
try:
from hermes_cli.config import load_config_readonly
pc_cfg = load_config_readonly().get("prompt_caching", {}) or {}
ttl = pc_cfg.get("cache_ttl", "5m")
except Exception:
return False
return cache_ttl_means_disabled(ttl)
def blank_cache_policy_stub(cache_disabled: Optional[bool] = None):
"""Build the destination-identity-blank stub for ``anthropic_prompt_cache_policy``.
Single sanctioned constructor for that stub. Callers that resolve cache
policy against a destination identified out-of-band (not a live
``AIAgent``) must go through here so ``_cache_disabled`` is never left
off a hand-rolled ``SimpleNamespace`` (#76085).
When ``cache_disabled`` is omitted, falls back to the global config so
stub paths without an agent snapshot still honor an operator disable.
"""
from types import SimpleNamespace
if cache_disabled is None:
cache_disabled = prompt_caching_disabled_from_config()
return SimpleNamespace(
provider="",
base_url="",
api_mode="",
model="",
_cache_disabled=bool(cache_disabled),
)
def plan_cache_sections_for_destination(
messages: list,
tools: Optional[list],
*,
provider: str,
base_url: str,
api_mode: str,
model: str,
cache_disabled: Optional[bool] = None,
) -> Tuple[list, list]:
"""Plan request-local cache sections for one resolved destination.
Shared core of the synchronous acting-aggregator (MoA) and auxiliary
fallback senders: resolve the cache policy for the destination's real
provider/base_url/api_mode/model, then either return stripped canonical
copies (non-caching route) or a :func:`build_prompt_cache_plan` layout
(caching route, with the direct-native tool marker when the destination
is api.anthropic.com on the Messages wire).
Never mutates ``messages`` or ``tools`` — both return values are
request-local copies.
``cache_disabled`` threads the operator's ``prompt_caching.cache_ttl``
disable into the blank policy stub. When omitted, the live config is
consulted so MoA/auxiliary paths cannot re-enable markers after the
user turned caching off (#76085).
"""
from agent.prompt_caching import (
build_prompt_cache_plan,
strip_anthropic_cache_control,
strip_anthropic_tool_cache_control,
)
stub = blank_cache_policy_stub(cache_disabled)
should_cache, native_layout = anthropic_prompt_cache_policy(
stub,
provider=provider,
base_url=base_url,
api_mode=api_mode,
model=model,
)
if not should_cache:
canonical_messages = copy.deepcopy(messages or [])
strip_anthropic_cache_control(canonical_messages)
return canonical_messages, strip_anthropic_tool_cache_control(tools)
plan = build_prompt_cache_plan(
messages,
tools,
native_anthropic=native_layout,
direct_native_tool_cache=_direct_native_anthropic_tool_cache_capability(
stub,
provider=provider,
base_url=base_url,
api_mode=api_mode,
model=model,
),
)
return plan.messages, plan.tools
def anthropic_prompt_cache_policy(
agent,
*,
@@ -1860,13 +2003,25 @@ def anthropic_prompt_cache_policy(
gateway implements the Anthropic cache_control contract
(MiniMax, Zhipu GLM, LiteLLM's Anthropic proxy mode all do).
Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct
Alibaba (DashScope) also honour Anthropic-style ``cache_control``
markers on OpenAI-wire chat completions. Upstream pi-mono #3392 /
pi #3393 documented this for opencode-go Qwen. Without markers
these providers serve zero cache hits, re-billing the full prompt
on every turn.
Qwen models on OpenCode and direct Alibaba (DashScope), plus DeepSeek
models on OpenCode, also honour Anthropic-style ``cache_control`` markers
on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi #3393
documented this for opencode-go Qwen; #24617 reports the same gateway
contract for DeepSeek. Without markers these providers serve zero cache
hits, re-billing the full prompt on every turn.
If the operator has set ``prompt_caching.cache_ttl`` to a falsy value
(``false``, ``null``, ``"off"``, etc.) in config.yaml, prompt caching
is fully disabled — this early return ensures the disable survives
``/model`` switches, fallback re-derivation, and runtime snapshot
restoration (#33555). We check ``"_cache_disabled"`` (set by
init_agent when the disable is detected) rather than ``_cache_ttl``
directly, because ``_cache_ttl`` is not yet set when the policy runs
during the initial ``init_agent`` call.
"""
if getattr(agent, "_cache_disabled", False):
return (False, False)
eff_provider = (provider if provider is not None else agent.provider) or ""
eff_base_url = base_url if base_url is not None else (agent.base_url or "")
eff_api_mode = api_mode if api_mode is not None else (agent.api_mode or "")
@@ -1980,16 +2135,22 @@ def anthropic_prompt_cache_policy(
if is_minimax_provider or is_minimax_host:
return True, True
# Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire
# transport that accepts Anthropic-style cache_control markers and
# rewards them with real cache hits. Without this branch
# qwen3.6-plus on opencode-go reports 0% cached tokens and burns
# through the subscription on every turn.
# Qwen on OpenCode (Zen/Go) and native DashScope, plus DeepSeek on
# OpenCode only: OpenAI-wire transports that accept Anthropic-style
# cache_control markers and reward them with real cache hits. Keep direct
# Alibaba specific to Qwen; its catalog does not establish the same
# contract for DeepSeek.
model_is_qwen = "qwen" in model_lower
model_is_deepseek = "deepseek" in model_lower
provider_is_opencode = provider_lower in {
"opencode", "opencode-zen", "opencode-go",
}
provider_is_alibaba_family = provider_lower in {
"opencode", "opencode-zen", "opencode-go", "alibaba",
}
if provider_is_alibaba_family and model_is_qwen:
if (provider_is_alibaba_family and model_is_qwen) or (
provider_is_opencode and model_is_deepseek
):
# Envelope layout (native_anthropic=False): markers on inner
# content parts, not top-level tool messages. Matches
# pi-mono's "alibaba" cacheControlFormat.
@@ -2082,6 +2243,31 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo
# restore, request-scoped); auxiliary_client builds its own clients and keeps
# SDK retries because it is NOT wrapped by the conversation loop.
client_kwargs.setdefault("max_retries", 0)
# Defense-in-depth: guarantee Copilot requests carry the integration
# headers regardless of which build path we came through. The primary
# header wiring lives in `_apply_client_headers_for_base_url`, but two
# rebuild paths (`primary_recovery`, `restore_primary` in this module)
# reconstruct the client purely from a `_primary_runtime` snapshot and do
# NOT re-run that wiring. If the snapshot's client_kwargs ever lacks
# `default_headers` (older snapshot, header-less resolver result), the
# client goes out WITHOUT `Copilot-Integration-Id: vscode-chat`; the
# Copilot server then routes it to the "copilot-language-server" integrator
# whose model allowlist omits enterprise-only models (claude-opus-4.8) →
# HTTP 400 model_not_available_for_integrator on every turn. This chokepoint
# is the single place every primary OpenAI client passes through, so filling
# missing Copilot headers here closes the whole class. We only ADD missing
# keys — never override headers a caller deliberately set.
try:
if base_url_host_matches(str(client_kwargs.get("base_url", "")), "githubcopilot.com"):
from hermes_cli.models import copilot_default_headers
existing = dict(client_kwargs.get("default_headers") or {})
existing_lower = {k.lower() for k in existing}
for hk, hv in copilot_default_headers().items():
if hk.lower() not in existing_lower:
existing[hk] = hv
client_kwargs["default_headers"] = existing
except Exception:
_ra().logger.debug("Copilot default-header guard skipped", exc_info=True)
# Uses the module-level `OpenAI` name, resolved lazily on first
# access via __getattr__ below. Tests patch via `run_agent.OpenAI`.
client = _ra().OpenAI(**client_kwargs)
@@ -3773,6 +3959,9 @@ __all__ = [
"restore_primary_runtime",
"extract_reasoning",
"dump_api_request_debug",
"prompt_caching_disabled_from_config",
"blank_cache_policy_stub",
"plan_cache_sections_for_destination",
"anthropic_prompt_cache_policy",
"create_openai_client",
"switch_model",
+16 -4
View File
@@ -24,6 +24,18 @@ from urllib.parse import urlparse
from hermes_constants import get_hermes_home
from typing import Any, Dict, List, Optional, Tuple
from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars
from agent.secret_scope import get_secret as _get_secret
def _getenv(name: str, default: str = "") -> str:
"""Profile-scoped replacement for os.getenv on credential reads.
Routes through the secret scope (Workstream A): identical to os.getenv
when multiplexing is off, scope-aware (and fail-closed on an unscoped
read) when on. Mirrors the same wrapper in hermes_cli/runtime_provider.py.
"""
val = _get_secret(name, default)
return val if val is not None else default
# NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls
# ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.)
@@ -1358,7 +1370,7 @@ def resolve_anthropic_token() -> Optional[str]:
creds = read_claude_code_credentials()
# 1. Hermes-managed OAuth/setup token env var
token = os.getenv("ANTHROPIC_TOKEN", "").strip()
token = _getenv("ANTHROPIC_TOKEN").strip()
if token:
preferred = _prefer_refreshable_claude_code_token(token, creds)
if preferred:
@@ -1366,7 +1378,7 @@ def resolve_anthropic_token() -> Optional[str]:
return token
# 2. CLAUDE_CODE_OAUTH_TOKEN (used by Claude Code for setup-tokens)
cc_token = os.getenv("CLAUDE_CODE_OAUTH_TOKEN", "").strip()
cc_token = _getenv("CLAUDE_CODE_OAUTH_TOKEN").strip()
if cc_token:
preferred = _prefer_refreshable_claude_code_token(cc_token, creds)
if preferred:
@@ -1385,7 +1397,7 @@ def resolve_anthropic_token() -> Optional[str]:
# 5. Regular API key, or a legacy OAuth token saved in ANTHROPIC_API_KEY.
# This remains as a compatibility fallback for pre-migration Hermes configs.
api_key = os.getenv("ANTHROPIC_API_KEY", "").strip()
api_key = _getenv("ANTHROPIC_API_KEY").strip()
if api_key:
return api_key
@@ -1428,7 +1440,7 @@ def run_oauth_setup_token() -> Optional[str]:
# Check env vars that may have been set
for env_var in ("CLAUDE_CODE_OAUTH_TOKEN", "ANTHROPIC_TOKEN"):
val = os.getenv(env_var, "").strip()
val = _getenv(env_var).strip()
if val:
return val
+590 -71
View File
@@ -14,6 +14,10 @@ Resolution order for text tasks (auto mode):
6. Direct API-key providers (z.ai/GLM, Kimi/Moonshot, MiniMax, MiniMax-CN)
7. None
OpenRouter fallback cost guard: ``auxiliary.free_only: true`` restricts the
step-2 fallback to ``:free`` SKUs; ``auxiliary.openrouter_model`` overrides
the default. A one-time WARNING is logged for non-``:free`` models.
Resolution order for vision/multimodal tasks (auto mode):
1. Selected main provider, if it is one of the supported vision backends below
2. OpenRouter
@@ -42,6 +46,7 @@ Payment / credit exhaustion fallback:
import contextlib
import contextvars
import copy
import functools
import hashlib
import inspect
@@ -54,7 +59,7 @@ import time
import uuid
from pathlib import Path # noqa: F401 — used by test mocks
from types import SimpleNamespace
from typing import Any, Callable, Dict, List, Optional, Tuple, TYPE_CHECKING
from typing import Any, Callable, Dict, List, NamedTuple, Optional, Tuple, TYPE_CHECKING
from urllib.parse import urlparse, parse_qs, urlunparse
# NOTE: `from openai import OpenAI` is deliberately NOT at module top — the
@@ -223,30 +228,134 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any:
# part-way, compression falls back to a static "summary unavailable" marker
# and the real handoff is lost (#23975). A thread-local flag lets such a
# task mark its in-flight LLM call as interrupt-protected; the Codex
# Responses stream's cancellation check honors it. TIMEOUTS still fire
# Responses stream's cancellation check honors it. An explicit host cancel
# (CLI Ctrl+C or /stop) may install a cancel check that overrides protection;
# ordinary incoming-message interrupts remain protected. TIMEOUTS still fire
# (a hung call must die), and all OTHER aux tasks (vision, web_extract,
# title_generation, …) remain freely interruptible.
_aux_interrupt_protection = threading.local()
class AuxiliaryExplicitCancellation(BaseException):
"""Frozen signal that an auxiliary attempt was explicitly hard-cancelled.
This deliberately follows ``asyncio.CancelledError`` and inherits directly
from ``BaseException``: provider retry/fallback code catches ``Exception``
broadly and must never reinterpret an explicit host stop as a transport
failure. ``cause`` is immutable class data so downstream compression code
does not re-query a mutable host Event after the transport has unwound.
"""
cause = "explicit_host_cancel"
def __init__(self) -> None:
super().__init__("auxiliary request explicitly cancelled by host")
def _aux_interrupt_protected() -> bool:
return bool(getattr(_aux_interrupt_protection, "active", False))
def _aux_interrupt_cancel_requested() -> bool:
"""Return whether an explicit host cancel overrides aux protection."""
event = getattr(_aux_interrupt_protection, "cancel_event", None)
if event is not None:
try:
return bool(event.is_set())
except Exception:
logger.debug("aux interrupt cancel event check failed", exc_info=True)
return False
check = getattr(_aux_interrupt_protection, "cancel_check", None)
if not callable(check):
return False
try:
return bool(check())
except Exception:
logger.debug("aux interrupt cancel check failed", exc_info=True)
return False
@contextlib.contextmanager
def aux_interrupt_protection(active: bool = True):
def aux_interrupt_protection(
active: bool = True,
cancel_check=None,
cancel_event=None,
):
"""Mark the current thread's auxiliary LLM call as interrupt-protected.
Used by atomic aux tasks (compression) so a mid-flight gateway interrupt
doesn't abort the call and trigger a degraded fallback. Re-entrant-safe:
restores the previous value on exit.
restores the previous value on exit. ``cancel_check`` lets the host retain
an explicit hard-cancel path; ``cancel_event`` is preferred when the host
already owns an Event. Nested protection scopes inherit both values.
"""
prev = getattr(_aux_interrupt_protection, "active", False)
prev_cancel_check = getattr(_aux_interrupt_protection, "cancel_check", None)
prev_cancel_event = getattr(_aux_interrupt_protection, "cancel_event", None)
_aux_interrupt_protection.active = active
if callable(cancel_check):
_aux_interrupt_protection.cancel_check = cancel_check
if cancel_event is not None and callable(getattr(cancel_event, "is_set", None)):
_aux_interrupt_protection.cancel_event = cancel_event
try:
yield
finally:
_aux_interrupt_protection.active = prev
_aux_interrupt_protection.cancel_check = prev_cancel_check
_aux_interrupt_protection.cancel_event = prev_cancel_event
def _capture_aux_cancel_check() -> Optional[Callable[[], Any]]:
"""Capture the current explicit-cancel source on the owning request thread."""
event = getattr(_aux_interrupt_protection, "cancel_event", None)
is_set = getattr(event, "is_set", None)
if callable(is_set):
return is_set
check = getattr(_aux_interrupt_protection, "cancel_check", None)
if callable(check):
# Preserve callable identity so attempt-local decision objects retain
# methods such as begin_timeout_cleanup() when captured by adapters.
return check
return None
def _captured_aux_cancel_requested(cancel_check: Callable[[], Any]) -> bool:
"""Read a request-thread cancellation source without leaking its failures."""
try:
return bool(cancel_check())
except Exception:
logger.debug("captured aux cancel check failed", exc_info=True)
return False
class _AuxiliaryCancellationDecision:
"""Atomically choose explicit cancellation or provider timeout per attempt."""
def __init__(self, source_cancel_check: Callable[[], Any]) -> None:
self._source_cancel_check = source_cancel_check
self._lock = threading.Lock()
self._outcome = "active"
def __call__(self) -> bool:
with self._lock:
if self._outcome == "cancelled":
return True
if self._outcome == "timed_out":
return False
if _captured_aux_cancel_requested(self._source_cancel_check):
self._outcome = "cancelled"
return True
return False
def begin_timeout_cleanup(self) -> bool:
"""Return whether timeout won and destructive cleanup is permitted."""
with self._lock:
if self._outcome == "active":
if _captured_aux_cancel_requested(self._source_cancel_check):
self._outcome = "cancelled"
else:
self._outcome = "timed_out"
return self._outcome == "timed_out"
# ── Forward-progress hook for streamed auxiliary calls ───────────────────
@@ -293,6 +402,75 @@ def aux_progress_hook(hook):
_aux_progress.hook = prev
def _run_protected_sync_provider_call(
callback: Callable[[dict[str, Any]], Any],
kwargs: dict[str, Any],
) -> Any:
"""Run one protected provider callback in an attempt-isolated daemon.
A hard cancel must release the compression-owning thread promptly, but
auxiliary clients are process-shared and cannot safely be closed or evicted
to wake one request. Only protected calls with a captured hard-cancel source
use this seam. Their provider callback (including stream aggregation) runs
in a daemon worker while the owner polls cancellation. On cancel the owner
unwinds immediately; the worker is left to finish under the provider timeout
already present in ``kwargs``. It owns no transcript or compressor commit
state and never holds the session lock.
Ordinary auxiliary calls, and protected calls without a cancellation source,
retain the historical direct synchronous path with no extra thread.
"""
source_cancel_check = _capture_aux_cancel_check()
if not _aux_interrupt_protected() or not callable(source_cancel_check):
return callback(kwargs)
# Freeze one linearized outcome for this isolated attempt. The host Event is
# reused and cleared on a later turn, while the Codex timeout Timer may race
# owner polling. Both paths must decide under the same attempt-local lock.
cancel_check = _AuxiliaryCancellationDecision(source_cancel_check)
if cancel_check():
raise AuxiliaryExplicitCancellation()
progress_hook = getattr(_aux_progress, "hook", None)
provider_context = contextvars.copy_context()
done = threading.Event()
outcome: dict[str, Any] = {}
def _provider_worker() -> None:
try:
with aux_progress_hook(progress_hook), aux_interrupt_protection(
cancel_check=cancel_check
):
outcome["result"] = callback(kwargs)
except BaseException as exc:
outcome["exception"] = exc
finally:
done.set()
threading.Thread(
target=provider_context.run,
args=(_provider_worker,),
name="hermes-protected-aux-provider",
daemon=True,
).start()
while True:
# Cancellation is checked before and after every completion wait so it
# wins whenever result publication and the host Event become visible in
# the same polling interval.
if _captured_aux_cancel_requested(cancel_check):
raise AuxiliaryExplicitCancellation()
if not done.wait(0.02):
continue
if _captured_aux_cancel_requested(cancel_check):
raise AuxiliaryExplicitCancellation()
exception = outcome.get("exception")
if exception is not None:
raise exception
return outcome.get("result")
def _safe_isinstance(obj: Any, maybe_type: Any) -> bool:
"""Return False instead of raising when a patched symbol is not a type."""
try:
@@ -955,6 +1133,29 @@ def _nous_min_key_ttl_seconds() -> int:
return 1800
def _scoped_key_env(name: str) -> str:
"""Read a provider API key env var through the profile secret scope.
Auxiliary-client resolution runs both inside agent turns (secret scope
installed — its verdict is authoritative under multiplex, so a scoped
miss must NOT borrow another profile's process-env key) and on unscoped
startup/CLI probe paths, which keep the legacy ``os.environ`` read via
the ``UnscopedSecretError`` fallback (Slack pattern, #59739).
"""
if not name:
return ""
try:
from agent.secret_scope import UnscopedSecretError, get_secret
try:
return (get_secret(name) or "").strip()
except UnscopedSecretError:
pass
except Exception:
pass
return (os.getenv(name) or "").strip()
# ── Codex Responses → chat.completions adapter ─────────────────────────────
# All auxiliary consumers call client.chat.completions.create(**kwargs) and
# read response.choices[0].message.content. This adapter translates those
@@ -1150,12 +1351,53 @@ class _CodexCompletionsAdapter:
deadline = time.monotonic() + float(total_timeout) if total_timeout else None
timed_out = threading.Event()
timeout_timer: Optional[threading.Timer] = None
# A protected provider call may outlive its owning compression attempt:
# the owner returns promptly on hard cancellation while this adapter is
# still blocked in the SDK stream on its isolated worker. Timer threads
# do not inherit this worker's thread-local protection state, so freeze
# the hard-cancel source here, before creating the timer.
protected_cancel_check = (
_capture_aux_cancel_check() if _aux_interrupt_protected() else None
)
attempt_stream_lock = threading.Lock()
attempt_stream: List[Any] = []
def _timeout_message() -> str:
return f"Codex auxiliary Responses stream exceeded {float(total_timeout):.1f}s total timeout"
def _close_client_on_timeout() -> None:
begin_timeout_cleanup = getattr(
protected_cancel_check, "begin_timeout_cleanup", None
)
if callable(begin_timeout_cleanup):
timeout_won = bool(begin_timeout_cleanup())
else:
timeout_won = not (
callable(protected_cancel_check)
and _captured_aux_cancel_requested(protected_cancel_check)
)
# Publish transport timeout only after the attempt-local decision is
# fixed, so owner polling cannot observe completion in between.
timed_out.set()
if not timeout_won:
# The request owner already hard-cancelled this attempt. The
# OpenAI client is process-shared, so closing/evicting it here
# would disrupt unrelated sessions. Wake only this attempt's
# event stream when responses.create() returned one in time;
# otherwise rely on the bounded SDK/provider timeout.
with attempt_stream_lock:
stream = attempt_stream[0] if attempt_stream else None
close_stream = getattr(stream, "close", None)
if callable(close_stream):
try:
close_stream()
except Exception:
logger.debug(
"Codex auxiliary: cancelled attempt stream close "
"during timeout failed",
exc_info=True,
)
return
close = getattr(self._client, "close", None)
if callable(close):
try:
@@ -1182,11 +1424,14 @@ class _CodexCompletionsAdapter:
from tools.interrupt import is_interrupted
# Honor interrupt protection for atomic aux tasks (compression):
# a mid-flight gateway interrupt must NOT abort the summary call
# and trigger a degraded fallback marker (#23975). Timeouts above
# still fire; other aux tasks remain interruptible.
# and trigger a degraded fallback marker (#23975). Explicit host
# cancellation has its own frozen exception; timeouts above still
# fire and other aux tasks remain interruptible.
if _aux_interrupt_cancel_requested():
raise AuxiliaryExplicitCancellation()
if is_interrupted() and not _aux_interrupt_protected():
raise InterruptedError("Codex auxiliary Responses stream interrupted")
except InterruptedError:
except (InterruptedError, AuxiliaryExplicitCancellation):
raise
except Exception:
# Interrupt state is a best-effort UX hook; never make it a
@@ -1225,12 +1470,39 @@ class _CodexCompletionsAdapter:
_check_cancelled()
event_stream = self._client.responses.create(**stream_kwargs)
with attempt_stream_lock:
attempt_stream.append(event_stream)
# The timer can fire while responses.create() is blocked. If the
# cancelled attempt had no stream to close at that instant, close it
# now that it is safely attempt-owned; never touch the shared client.
if (
timed_out.is_set()
and callable(protected_cancel_check)
and _captured_aux_cancel_requested(protected_cancel_check)
):
close_fn = getattr(event_stream, "close", None)
if callable(close_fn):
try:
close_fn()
except Exception:
logger.debug(
"Codex auxiliary: late cancelled attempt stream close failed",
exc_info=True,
)
try:
final = _consume_codex_event_stream(
event_stream,
model=resp_kwargs.get("model"),
on_event=_on_each_event,
)
# Some Codex-compatible hosts accept ``stream=True`` but return
# a completed Responses object instead of an SSE iterator. Do
# not hand that object to the event consumer: typed Responses
# (and compatibility shims such as SimpleNamespace) are not
# event streams and may not be iterable at all.
if hasattr(event_stream, "output"):
final = event_stream
else:
final = _consume_codex_event_stream(
event_stream,
model=str(resp_kwargs.get("model") or model),
on_event=_on_each_event,
)
finally:
close_fn = getattr(event_stream, "close", None)
if callable(close_fn):
@@ -1238,6 +1510,8 @@ class _CodexCompletionsAdapter:
close_fn()
except Exception:
pass
with attempt_stream_lock:
attempt_stream.clear()
if final is None:
raise RuntimeError("Codex auxiliary Responses stream did not return a final response")
@@ -2150,8 +2424,63 @@ def _resolve_api_key_provider() -> Tuple[Optional[OpenAI], Optional[str]]:
# ── Provider resolution helpers ─────────────────────────────────────────────
_paid_lane_warned: set = set()
def _is_free_model(model: Optional[str]) -> bool:
"""True when ``model`` is an OpenRouter free SKU (``:free`` suffix)."""
return bool(model) and str(model).strip().endswith(":free")
def _aux_openrouter_settings() -> Tuple[bool, str]:
"""Read free_only and openrouter_model from config in one pass.
Returns (free_only, model) — defaults (False, _OPENROUTER_MODEL) on any
config-read failure.
"""
try:
from hermes_cli.config import cfg_get, load_config_readonly
cfg = load_config_readonly()
free_only = bool(cfg_get(cfg, "auxiliary", "free_only", default=False))
val = cfg_get(cfg, "auxiliary", "openrouter_model")
model = val.strip() if isinstance(val, str) and val.strip() else _OPENROUTER_MODEL
return free_only, model
except Exception:
return False, _OPENROUTER_MODEL
def _warn_paid_lane_once(model: str) -> None:
"""Log a WARNING the first time a non-:free OpenRouter model is engaged."""
if model in _paid_lane_warned:
return
_paid_lane_warned.add(model)
logger.warning(
"Auxiliary client: PAID lane engaged for auxiliary task — OpenRouter "
"fallback model %r is not a :free SKU and may incur real spend. Set "
"auxiliary.free_only: true to restrict auxiliary fallbacks to free "
"models, or auxiliary.openrouter_model to a :free model.",
model,
)
def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Optional[OpenAI], Optional[str]]:
free_only, cfg_model = _aux_openrouter_settings()
or_model = model or cfg_model
if free_only and not _is_free_model(or_model):
logger.warning(
"Auxiliary client: auxiliary.free_only is enabled but the "
"OpenRouter fallback model %r is not a :free SKU — skipping the "
"OpenRouter fallback. Set auxiliary.openrouter_model to a :free "
"model (e.g. nvidia/nemotron-3-ultra-550b-a55b:free) or disable "
"auxiliary.free_only.",
or_model,
)
_mark_provider_unhealthy("openrouter", ttl=60)
return None, None
if not _is_free_model(or_model):
_warn_paid_lane_once(or_model)
pool_present, entry = _select_pool_entry("openrouter")
if pool_present:
or_key = explicit_api_key or _pool_runtime_api_key(entry)
@@ -2159,18 +2488,18 @@ def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Op
base_url = _pool_runtime_base_url(entry, OPENROUTER_BASE_URL) or OPENROUTER_BASE_URL
logger.debug("Auxiliary client: OpenRouter via pool")
return _create_openai_client(api_key=or_key, base_url=base_url,
default_headers=build_or_headers()), model or _OPENROUTER_MODEL
default_headers=build_or_headers()), or_model
# Pool exists but is exhausted (no usable runtime key) — fall through to
# the OPENROUTER_API_KEY env-var path rather than failing outright.
logger.debug("Auxiliary client: OpenRouter pool exhausted, trying OPENROUTER_API_KEY")
or_key = explicit_api_key or os.getenv("OPENROUTER_API_KEY")
or_key = explicit_api_key or _scoped_key_env("OPENROUTER_API_KEY")
if not or_key:
_mark_provider_unhealthy("openrouter", ttl=60)
return None, None
logger.debug("Auxiliary client: OpenRouter")
return _create_openai_client(api_key=or_key, base_url=OPENROUTER_BASE_URL,
default_headers=build_or_headers()), model or _OPENROUTER_MODEL
default_headers=build_or_headers()), or_model
def _describe_openrouter_unavailable() -> str:
@@ -2181,7 +2510,7 @@ def _describe_openrouter_unavailable() -> str:
return "OpenRouter credential pool has no usable entries (credentials may be exhausted)"
if not _pool_runtime_api_key(entry):
return "OpenRouter credential pool entry is missing a runtime API key"
if not str(os.getenv("OPENROUTER_API_KEY") or "").strip():
if not _scoped_key_env("OPENROUTER_API_KEY"):
return "OPENROUTER_API_KEY not set"
return "no usable OpenRouter credentials found"
@@ -2607,14 +2936,17 @@ def _relay_sync_completion(
) -> Any:
callback = create or (lambda request: client.chat.completions.create(**request))
route = _relay_auxiliary_metadata(provider=provider, api_mode=api_mode)
# Protected compression calls isolate only the provider callback and stream
# aggregation. The owning thread remains free to unwind its lease/DB
# transaction on hard cancel without touching the process-shared client.
if route is None:
return callback(kwargs)
return _run_protected_sync_provider_call(callback, kwargs)
provider_name, fallback_model, metadata = route
from agent import relay_llm
return relay_llm.execute_current(
kwargs,
callback,
lambda request: _run_protected_sync_provider_call(callback, request),
name=provider_name,
model_name=str(kwargs.get("model") or fallback_model),
metadata=metadata,
@@ -2667,6 +2999,7 @@ def _relay_sync_stream(
model_name=str(kwargs.get("model") or fallback_model),
finalizer=dict,
metadata=metadata,
completed_response_predicate=lambda value: hasattr(value, "choices"),
)
_RUNTIME_MAIN_COMPAT_SNAPSHOT: Tuple[Any, ...] = ("", "", "", "", "", "")
_RUNTIME_MAIN_COMPAT_LOCK = threading.Lock()
@@ -2812,7 +3145,7 @@ def _resolve_custom_runtime() -> Tuple[Optional[str], Optional[str], Optional[st
if not isinstance(runtime, dict):
openai_base = os.getenv("OPENAI_BASE_URL", "").strip().rstrip("/")
openai_key = os.getenv("OPENAI_API_KEY", "").strip()
openai_key = _scoped_key_env("OPENAI_API_KEY")
if not openai_base:
return None, None, None
runtime = {
@@ -4174,6 +4507,27 @@ def _auth_refresh_provider_for_route(
return normalized
def _fallback_chain_entry(task: Optional[str], fb_label: str) -> Optional[Dict[str, Any]]:
"""Resolve the configured ``fallback_chain`` entry a label points at.
Labels minted by :func:`_try_configured_fallback_chain` carry the entry
index in our own stable format (``fallback_chain[<i>](<provider>)``).
Returns ``None`` when the label is not a configured-chain candidate or
the index no longer resolves to a dict entry.
"""
if not task or not fb_label:
return None
m = re.match(r"fallback_chain\[(\d+)\]", fb_label)
if not m:
return None
try:
chain = _get_auxiliary_task_config(task).get("fallback_chain")
entry = chain[int(m.group(1))] if isinstance(chain, list) else None
except Exception:
return None
return entry if isinstance(entry, dict) else None
def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[float]:
"""Resolve a per-entry ``timeout`` for a configured fallback candidate.
@@ -4185,29 +4539,113 @@ def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[floa
primary's 30s deadline every turn (#62452).
Entries in ``auxiliary.<task>.fallback_chain`` may declare their own
``timeout`` (seconds). This helper reads it by parsing the entry index
out of the label minted by :func:`_try_configured_fallback_chain`
(``fallback_chain[<i>](<provider>)`` — our own stable format). Returns
``None`` when the label is not a configured-chain candidate, the entry
has no ``timeout``, or the value is invalid — callers then keep the
task-level timeout, preserving existing behavior.
``timeout`` (seconds). Returns ``None`` when the label is not a
configured-chain candidate, the entry has no ``timeout``, or the value
is invalid — callers then keep the task-level timeout, preserving
existing behavior.
"""
if not task or not fb_label:
return None
m = re.match(r"fallback_chain\[(\d+)\]", fb_label)
if not m:
return None
try:
chain = _get_auxiliary_task_config(task).get("fallback_chain")
entry = chain[int(m.group(1))] if isinstance(chain, list) else None
raw = entry.get("timeout") if isinstance(entry, dict) else None
except Exception:
return None
entry = _fallback_chain_entry(task, fb_label)
raw = entry.get("timeout") if entry else None
if isinstance(raw, (int, float)) and not isinstance(raw, bool) and raw > 0:
return float(raw)
return None
def _fallback_provider_from_label(label: str) -> str:
"""Recover the provider identifier from a fallback display label."""
match = re.match(r"(?:fallback_chain\[\d+\]|main-agent)\(([^)]+)\)$", label or "")
return match.group(1).strip() if match else str(label or "").strip()
class _FallbackDestination(NamedTuple):
provider: str
base_url: str
api_mode: Optional[str]
model: Optional[str]
def _complete_fallback_destination(
provider: str,
base_url: str,
api_mode: Optional[str],
model: Optional[str],
) -> _FallbackDestination:
if not api_mode:
if _endpoint_speaks_anthropic_messages(base_url):
api_mode = "anthropic_messages"
else:
try:
from hermes_cli.runtime_provider import resolve_runtime_provider
runtime = resolve_runtime_provider(
requested=provider,
explicit_base_url=base_url or None,
target_model=model or "",
)
api_mode = str(runtime.get("api_mode") or "").strip() or None
except Exception:
pass
return _FallbackDestination(provider, base_url, api_mode, model)
def _fallback_destination_from_entry(
entry: Dict[str, Any],
fb_client: Any,
fb_model: Optional[str],
) -> _FallbackDestination:
provider = str(entry.get("provider") or "").strip()
base_url = str(
entry.get("base_url") or getattr(fb_client, "base_url", "") or ""
).strip()
api_mode = str(
entry.get("api_mode") or entry.get("transport") or ""
).strip() or None
model = fb_model or str(entry.get("model") or "").strip() or None
return _complete_fallback_destination(provider, base_url, api_mode, model)
def _fallback_destination(
task: Optional[str],
fb_client: Any,
fb_model: Optional[str],
fb_label: str,
) -> _FallbackDestination:
"""Return the resolved route identity used by a fallback request."""
attached = getattr(fb_client, "_hermes_fallback_destination", None)
if isinstance(attached, _FallbackDestination):
return attached
provider = _fallback_provider_from_label(fb_label)
base_url = str(getattr(fb_client, "base_url", "") or "")
api_mode = None
model = fb_model
entry = _fallback_chain_entry(task, fb_label)
if entry is not None:
return _fallback_destination_from_entry(entry, fb_client, fb_model)
return _complete_fallback_destination(provider, base_url, api_mode, model)
def _replan_synchronous_cache_sections(
messages: list,
tools: Optional[list],
*,
destination: _FallbackDestination,
) -> tuple[list, list]:
"""Strip source decoration and plan one synchronous destination locally."""
from agent.agent_runtime_helpers import plan_cache_sections_for_destination
return plan_cache_sections_for_destination(
messages,
tools,
provider=destination.provider,
base_url=destination.base_url,
api_mode=destination.api_mode or "",
model=destination.model or "",
)
def _call_fallback_candidate_sync(
fb_client: Any,
fb_model: Optional[str],
@@ -4250,36 +4688,70 @@ def _call_fallback_candidate_sync(
task or "call", fb_label, fb_timeout, effective_timeout,
)
effective_timeout = fb_timeout
fb_base = str(getattr(fb_client, "base_url", "") or "")
destination = _fallback_destination(task, fb_client, fb_model, fb_label)
fallback_messages, fallback_tools = _replan_synchronous_cache_sections(
messages,
tools,
destination=destination,
)
fb_kwargs = _build_call_kwargs(
fb_label, fb_model, messages,
destination.provider, destination.model, fallback_messages,
temperature=temperature, max_tokens=max_tokens,
tools=tools, timeout=effective_timeout,
tools=fallback_tools, timeout=effective_timeout,
extra_body=effective_extra_body, reasoning_config=reasoning_config,
base_url=fb_base, task=task)
base_url=destination.base_url, task=task)
try:
return _validate_llm_response(
_relay_sync_completion(fb_client, fb_kwargs, provider=fb_label), task)
_relay_sync_completion(
fb_client,
fb_kwargs,
provider=destination.provider,
api_mode=destination.api_mode,
),
task,
)
except Exception as fb_err:
if not _is_auth_error(fb_err):
raise
fb_provider = _auth_refresh_provider_for_route(fb_label, fb_base)
fb_provider = _auth_refresh_provider_for_route(
destination.provider, destination.base_url
)
if fb_provider not in {"auto", "", None} and _refresh_provider_credentials(fb_provider):
retry_client, retry_model = _get_cached_client(fb_provider, fb_model)
retry_client, retry_model = _get_cached_client(
fb_provider,
destination.model,
base_url=destination.base_url or None,
api_mode=destination.api_mode,
)
if retry_client is not None:
retry_destination = _FallbackDestination(
fb_provider,
destination.base_url
or str(getattr(retry_client, "base_url", "") or ""),
destination.api_mode,
retry_model or destination.model,
)
retry_messages, retry_tools = _replan_synchronous_cache_sections(
messages,
tools,
destination=retry_destination,
)
retry_kwargs = _build_call_kwargs(
fb_provider, retry_model or fb_model, messages,
retry_destination.provider,
retry_destination.model,
retry_messages,
temperature=temperature, max_tokens=max_tokens,
tools=tools, timeout=effective_timeout,
tools=retry_tools, timeout=effective_timeout,
extra_body=effective_extra_body,
reasoning_config=reasoning_config,
base_url=str(getattr(retry_client, "base_url", "") or fb_base), task=task)
base_url=retry_destination.base_url, task=task)
try:
return _validate_llm_response(
_relay_sync_completion(
retry_client,
retry_kwargs,
provider=fb_provider,
provider=retry_destination.provider,
api_mode=retry_destination.api_mode,
),
task,
)
@@ -4322,43 +4794,71 @@ async def _call_fallback_candidate_async(
task or "call", fb_label, fb_timeout, effective_timeout,
)
effective_timeout = fb_timeout
fb_base = str(getattr(fb_client, "base_url", "") or "")
destination = _fallback_destination(task, fb_client, fb_model, fb_label)
fallback_messages, fallback_tools = _replan_synchronous_cache_sections(
messages,
tools,
destination=destination,
)
fb_kwargs = _build_call_kwargs(
fb_label, fb_model, messages,
destination.provider, destination.model, fallback_messages,
temperature=temperature, max_tokens=max_tokens,
tools=tools, timeout=effective_timeout,
tools=fallback_tools, timeout=effective_timeout,
extra_body=effective_extra_body, reasoning_config=reasoning_config,
base_url=fb_base, task=task)
base_url=destination.base_url, task=task)
try:
return _validate_llm_response(
await _relay_async_completion(
fb_client,
fb_kwargs,
provider=fb_label,
provider=destination.provider,
api_mode=destination.api_mode,
),
task,
)
except Exception as fb_err:
if not _is_auth_error(fb_err):
raise
fb_provider = _auth_refresh_provider_for_route(fb_label, fb_base)
fb_provider = _auth_refresh_provider_for_route(
destination.provider, destination.base_url
)
if fb_provider not in {"auto", "", None} and _refresh_provider_credentials(fb_provider):
retry_client, retry_model = _get_cached_client(
fb_provider, fb_model, async_mode=True)
fb_provider,
destination.model,
async_mode=True,
base_url=destination.base_url or None,
api_mode=destination.api_mode,
)
if retry_client is not None:
retry_destination = _FallbackDestination(
fb_provider,
destination.base_url
or str(getattr(retry_client, "base_url", "") or ""),
destination.api_mode,
retry_model or destination.model,
)
retry_messages, retry_tools = _replan_synchronous_cache_sections(
messages,
tools,
destination=retry_destination,
)
retry_kwargs = _build_call_kwargs(
fb_provider, retry_model or fb_model, messages,
retry_destination.provider,
retry_destination.model,
retry_messages,
temperature=temperature, max_tokens=max_tokens,
tools=tools, timeout=effective_timeout,
tools=retry_tools, timeout=effective_timeout,
extra_body=effective_extra_body,
reasoning_config=reasoning_config,
base_url=str(getattr(retry_client, "base_url", "") or fb_base), task=task)
base_url=retry_destination.base_url, task=task)
try:
return _validate_llm_response(
await _relay_async_completion(
retry_client,
retry_kwargs,
provider=fb_provider,
provider=retry_destination.provider,
api_mode=retry_destination.api_mode,
),
task,
)
@@ -4735,14 +5235,15 @@ def _try_configured_fallback_for_unavailable_client(
def _fallback_entry_api_key(entry: Dict[str, Any]) -> Optional[str]:
"""Resolve inline or env-backed API key from a fallback-chain entry."""
explicit = str(entry.get("api_key") or "").strip()
if explicit:
return explicit
key_env = str(entry.get("key_env") or entry.get("api_key_env") or "").strip()
if key_env:
return os.getenv(key_env, "").strip() or None
return None
"""Resolve inline or env-backed API key from a fallback-chain entry.
Delegates to the centralized, secret-scope-aware resolver so this path
doesn't leak another profile's credential via a raw ``os.getenv`` under
gateway multiplexing (see ``hermes_cli.fallback_config.resolve_entry_api_key``).
"""
from hermes_cli.fallback_config import resolve_entry_api_key
return resolve_entry_api_key(entry)
def _resolve_fallback_entry(entry: Dict[str, Any]) -> Tuple[Optional[Any], Optional[str]]:
@@ -4754,13 +5255,21 @@ def _resolve_fallback_entry(entry: Dict[str, Any]) -> Tuple[Optional[Any], Optio
base_url = str(entry.get("base_url") or "").strip() or None
api_key = _fallback_entry_api_key(entry)
api_mode = str(entry.get("api_mode") or entry.get("transport") or "").strip() or None
return resolve_provider_client(
client, resolved_model = resolve_provider_client(
provider,
model=model,
explicit_base_url=base_url,
explicit_api_key=api_key,
api_mode=api_mode,
)
if client is not None:
try:
client._hermes_fallback_destination = _fallback_destination_from_entry(
entry, client, resolved_model
)
except Exception:
pass
return client, resolved_model
def _try_main_fallback_chain(
@@ -5442,7 +5951,7 @@ def resolve_provider_client(
custom_base = _to_openai_base_url(explicit_base_url).strip()
custom_key = (
(explicit_api_key or "").strip()
or os.getenv("OPENAI_API_KEY", "").strip()
or _scoped_key_env("OPENAI_API_KEY")
or _read_main_api_key_if_same_host(custom_base)
or "no-key-required" # local servers don't need auth
)
@@ -5538,7 +6047,7 @@ def resolve_provider_client(
custom_key = (custom_entry.get("api_key") or "").strip()
custom_key_env = (custom_entry.get("key_env") or custom_entry.get("api_key_env") or "").strip()
if not custom_key and custom_key_env:
custom_key = os.getenv(custom_key_env, "").strip()
custom_key = _scoped_key_env(custom_key_env)
custom_key = custom_key or "no-key-required"
if custom_key == "no-key-required":
logger.warning(
@@ -6360,7 +6869,7 @@ def auxiliary_max_tokens_param(value: int, *, model: Optional[str] = None) -> di
misses the case where a custom base URL serves e.g. ``gpt-5.4``.
"""
custom_base = _current_custom_base_url()
or_key = os.getenv("OPENROUTER_API_KEY")
or_key = _scoped_key_env("OPENROUTER_API_KEY")
# Use max_completion_tokens for direct OpenAI-compatible providers that reject
# max_tokens on newer GPT-4o/o-series/GPT-5-style models.
_custom_host = base_url_hostname(custom_base) or ""
@@ -6821,7 +7330,7 @@ def _resolve_task_provider_model(
task_config.get("key_env") or task_config.get("api_key_env") or ""
).strip()
if cfg_key_env:
cfg_api_key = os.getenv(cfg_key_env, "").strip() or None
cfg_api_key = _scoped_key_env(cfg_key_env) or None
cfg_api_mode = str(task_config.get("api_mode", "")).strip() or None
# 'auto' is a sentinel meaning "inherit from main runtime / auto-detect", not
@@ -8130,6 +8639,16 @@ def call_llm(
kwargs["stream"] = True
if stream_options:
kwargs["stream_options"] = stream_options
if task == "moa_aggregator" and isinstance(client, CodexAuxiliaryClient):
# CodexAuxiliaryClient (openai-codex, xai-oauth, and any other
# Responses-shim provider) consumes the provider stream internally
# and returns a completed response object. Routing that nested
# MoA stream through Relay's generic managed stream makes the
# manager iterate the completed SimpleNamespace itself (#55933).
# Return the provider call directly; the MoA facade converts a
# completed response into a one-chunk delta iterator at its
# boundary.
return client.chat.completions.create(**kwargs)
return _relay_sync_stream(
client,
kwargs,
+18 -2
View File
@@ -367,11 +367,27 @@ def describe_active_credential(config: Optional[EntraIdentityConfig] = None,
info["tenant_id_env"] = os.environ["AZURE_TENANT_ID"].strip()
# Surface which env-var sources are present without minting yet.
# Credential-bearing vars (AZURE_CLIENT_SECRET, AZURE_FEDERATED_TOKEN_FILE)
# are read through the profile secret scope so a multiplexed profile's
# diagnostics don't report another profile's env-bridged credentials;
# unscoped CLI probes keep the legacy env read (Slack pattern).
def _scoped_env(name: str) -> str:
try:
from agent.secret_scope import UnscopedSecretError, get_secret
try:
return (get_secret(name) or "").strip()
except UnscopedSecretError:
pass
except Exception:
pass
return os.environ.get(name, "").strip()
env_sources = []
if os.environ.get("AZURE_FEDERATED_TOKEN_FILE", "").strip():
if _scoped_env("AZURE_FEDERATED_TOKEN_FILE"):
env_sources.append("WorkloadIdentityCredential (AZURE_FEDERATED_TOKEN_FILE)")
if (os.environ.get("AZURE_CLIENT_ID", "").strip()
and os.environ.get("AZURE_CLIENT_SECRET", "").strip()
and _scoped_env("AZURE_CLIENT_SECRET")
and os.environ.get("AZURE_TENANT_ID", "").strip()):
env_sources.append("EnvironmentCredential (client secret)")
if os.environ.get("IDENTITY_ENDPOINT", "").strip() or os.environ.get("MSI_ENDPOINT", "").strip():
+56
View File
@@ -18,6 +18,7 @@ for invariants and PR review criteria.
from __future__ import annotations
import copy
import json
import logging
import os
@@ -284,6 +285,15 @@ _SKILL_REVIEW_PROMPT = (
" • One-off task narratives. A user asking 'summarize today's "
"market' or 'analyze this PR' is not a class of work that warrants "
"a skill.\n\n"
" • Unresolved failures: if the session ended WITHOUT actually "
"finding a working method — you tried several things, none worked, "
"and told the user to check manually — do NOT write those attempts "
"up as a 'reliable workflow' or 'recommended approach'. That presents "
"an untested sequence of failures as validated guidance a future "
"session will trust and repeat. Either say 'Nothing to save', or, "
"only if you are independently confident of a real working alternative "
"(not something you are merely guessing might work), capture ONLY that "
"alternative — never the dead ends, and never dressed up as best practice.\n\n"
"If a tool failed because of setup state, capture the FIX (install "
"command, config step, env var to set) under an existing setup or "
"troubleshooting skill — never 'this tool does not work' as a "
@@ -377,6 +387,15 @@ _COMBINED_REVIEW_PROMPT = (
" • One-off task narratives. A user asking 'summarize today's "
"market' or 'analyze this PR' is not a class of work that warrants "
"a skill.\n\n"
" • Unresolved failures: if the session ended WITHOUT actually "
"finding a working method — you tried several things, none worked, "
"and told the user to check manually — do NOT write those attempts "
"up as a 'reliable workflow' or 'recommended approach'. That presents "
"an untested sequence of failures as validated guidance a future "
"session will trust and repeat. Either say 'Nothing to save', or, "
"only if you are independently confident of a real working alternative "
"(not something you are merely guessing might work), capture ONLY that "
"alternative — never the dead ends, and never dressed up as best practice.\n\n"
"If a tool failed because of setup state, capture the FIX (install "
"command, config step, env var to set) under an existing setup or "
"troubleshooting skill — never 'this tool does not work' as a "
@@ -727,6 +746,43 @@ def _run_review_in_thread(
# _cached_system_prompt below.
if not _routed:
_fork_kwargs["reasoning_config"] = getattr(agent, "reasoning_config", None)
# Gateway session context is appended to the parent's cached
# system prompt at API-call time through this field. Preserve
# it on same-model forks so the complete effective system
# prompt remains byte-identical and can reuse the warm prefix.
_fork_kwargs["ephemeral_system_prompt"] = getattr(
agent, "ephemeral_system_prompt", None
)
# Prefill messages are inserted immediately after the system
# message at API-call time (chat_completion_helpers.py /
# conversation_loop.py), so a parent with prefill configured
# (gateway prefill_messages_file) would otherwise diverge
# from the warm prefix at message index 1 — same bug class
# as the ephemeral prompt above, one position later.
# Deep copy: the unicode-error recovery path mutates
# prefill entries IN PLACE (_sanitize_messages_surrogates
# via conversation_loop), so sharing dicts would let a
# fork-side sanitize rewrite the parent's prefill bytes.
_parent_prefill = copy.deepcopy(
getattr(agent, "prefill_messages", None) or []
)
if _parent_prefill:
_fork_kwargs["prefill_messages"] = _parent_prefill
# OpenRouter provider-routing pins: prompt caches live per
# UPSTREAM provider, so a fork without the parent's pins can
# be routed to a different upstream and miss the warm cache
# even with byte-identical prompt/tools bytes.
for _pref_attr in (
"providers_allowed",
"providers_ignored",
"providers_order",
"provider_sort",
"provider_require_parameters",
"provider_data_collection",
):
_pref_val = getattr(agent, _pref_attr, None)
if _pref_val:
_fork_kwargs[_pref_attr] = _pref_val
review_agent = AIAgent(
model=_rt.get("model") or agent.model,
max_iterations=16,
+2
View File
@@ -26,6 +26,7 @@ Session metadata contract (preserved from the legacy ``CloudBrowserProvider``)::
"session_name": str, # unique name for agent-browser --session
"bb_session_id": str, # provider session ID (for close/cleanup)
"cdp_url": str, # CDP websocket URL
"expires_at": str, # optional provider-authoritative ISO timestamp
"features": dict, # feature flags that were enabled
"external_call_id": str, # optional, managed-gateway billing key
}
@@ -96,6 +97,7 @@ class BrowserProvider(abc.ABC):
"session_name": str, # unique name for agent-browser --session
"bb_session_id": str, # provider session ID (for close/cleanup)
"cdp_url": str, # CDP websocket URL
"expires_at": str, # optional provider-authoritative ISO timestamp
"features": dict, # feature flags that were enabled
}
+13 -14
View File
@@ -1120,9 +1120,10 @@ def interruptible_api_call(agent, api_kwargs: dict):
def build_api_kwargs(agent, api_messages: list) -> dict:
def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict:
"""Build the keyword arguments dict for the active API mode."""
tools_for_api = agent.tools
if tools_for_api is None:
tools_for_api = agent.tools
if agent.api_mode == "anthropic_messages":
_transport = agent._get_transport()
@@ -1788,19 +1789,17 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool
# Pass base_url and api_key from fallback config so custom
# endpoints (e.g. Ollama Cloud) resolve correctly instead of
# falling through to OpenRouter defaults.
from hermes_cli.fallback_config import resolve_entry_api_key
fb_base_url_hint = (fb.get("base_url") or "").strip() or None
fb_api_key_hint = (fb.get("api_key") or "").strip() or None
if not fb_api_key_hint:
# key_env and api_key_env are both documented aliases (see
# _normalize_custom_provider_entry in hermes_cli/config.py).
fb_key_env = (fb.get("key_env") or fb.get("api_key_env") or "").strip()
if fb_key_env:
fb_api_key_hint = os.getenv(fb_key_env, "").strip() or None
fb_api_key_hint = resolve_entry_api_key(fb)
# For Ollama Cloud endpoints, pull OLLAMA_API_KEY from env
# when no explicit key is in the fallback config. Host match
# (not substring) — see GHSA-76xc-57q6-vm5m.
if fb_base_url_hint and base_url_host_matches(fb_base_url_hint, "ollama.com") and not fb_api_key_hint:
fb_api_key_hint = os.getenv("OLLAMA_API_KEY") or None
from agent.secret_scope import get_secret
fb_api_key_hint = get_secret("OLLAMA_API_KEY") or None
fb_client, _resolved_fb_model = resolve_provider_client(
fb_provider, model=fb_model, raw_codex=True,
explicit_base_url=fb_base_url_hint,
@@ -2384,7 +2383,7 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str:
final_response = "I reached the iteration limit and couldn't generate a summary."
except Exception as e:
logger.warning(f"Failed to get summary response: {e}")
logger.warning("Failed to get summary response: %s", e)
final_response = f"I reached the maximum iterations ({agent.max_iterations}) but couldn't summarize. Error: {str(e)}"
finally:
from agent import relay_llm
@@ -2425,7 +2424,7 @@ def cleanup_task_resources(agent, task_id: str) -> None:
_ra().cleanup_vm(task_id)
except Exception as e:
if agent.verbose_logging:
logger.warning(f"Failed to cleanup VM for task {task_id}: {e}")
logger.warning("Failed to cleanup VM for task %s: %s", task_id, e)
try:
headed = False
try:
@@ -2443,7 +2442,7 @@ def cleanup_task_resources(agent, task_id: str) -> None:
_ra().cleanup_browser(task_id)
except Exception as e:
if agent.verbose_logging:
logger.warning(f"Failed to cleanup browser for task {task_id}: {e}")
logger.warning("Failed to cleanup browser for task %s: %s", task_id, e)
def _build_partial_stream_stub(
@@ -3979,7 +3978,7 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
" To avoid this delay, set display.streaming: false "
"in config.yaml\n"
)
logger.info(
logger.exception(
"Streaming failed before delivery: %s",
e,
)
+103 -9
View File
@@ -14,6 +14,7 @@ import hashlib
import json
import logging
import re
import unicodedata
import uuid
from types import SimpleNamespace
from typing import Any, Dict, List, Optional
@@ -73,6 +74,79 @@ _TOOL_CALL_LEAK_PATTERN = re.compile(
)
# The ChatGPT Codex backend reserves these Harmony wire tokens. If their
# literal spellings are replayed anywhere in request text, the backend rejects
# the request before inference with ``invalid_prompt: Request blocked.``.
# Category-Cf handling covers persisted sessions from an earlier U+200B weak
# defang; fullwidth bars survive format-character stripping while keeping the
# inspected source legible.
_HARMONY_CONTROL_TOKEN_RE = re.compile(
r"<\|(start|end|channel|message|constrain|return|call)\|>"
)
_FULLWIDTH_PIPE = "\uff5c"
def _neutralize_harmony_tokens(text: str) -> str:
"""Keep Harmony source readable without emitting reserved wire tokens."""
if not text or "<" not in text or "|" not in text:
return text
replacement = rf"<{_FULLWIDTH_PIPE}\1{_FULLWIDTH_PIPE}>"
if not any(unicodedata.category(char) == "Cf" for char in text):
return _HARMONY_CONTROL_TOKEN_RE.sub(replacement, text)
# U+200B is confirmed to be stripped by the Codex backend before its
# reserved-token check. Treat every Unicode format control equivalently so
# moving the character elsewhere in the token (or swapping in another Cf)
# cannot recreate the same visually hidden form.
visible_chars: List[str] = []
original_positions: List[int] = []
for index, char in enumerate(text):
if unicodedata.category(char) == "Cf":
continue
visible_chars.append(char)
original_positions.append(index)
visible_text = "".join(visible_chars)
matches = list(_HARMONY_CONTROL_TOKEN_RE.finditer(visible_text))
if not matches:
return text
result: List[str] = []
original_cursor = 0
for match in matches:
original_start = original_positions[match.start()]
original_end = original_positions[match.end() - 1] + 1
result.append(text[original_cursor:original_start])
result.append(f"<{_FULLWIDTH_PIPE}{match.group(1)}{_FULLWIDTH_PIPE}>")
original_cursor = original_end
result.append(text[original_cursor:])
return "".join(result)
def _neutralize_harmony_structure(value: Any) -> Any:
"""Neutralize JSON-like values; normalize tuples and reject unsafe keys.
Rewriting an object key could desynchronize a tool schema from the executor
contract, so a reserved token there is rejected explicitly instead.
"""
if isinstance(value, str):
return _neutralize_harmony_tokens(value)
if isinstance(value, (list, tuple)):
return [_neutralize_harmony_structure(item) for item in value]
if isinstance(value, dict):
normalized = {}
for key, item in value.items():
if isinstance(key, str) and _neutralize_harmony_tokens(key) != key:
raise ValueError(
"Reserved Harmony tokens in a JSON object key cannot be "
"neutralized without changing its contract."
)
normalized[key] = _neutralize_harmony_structure(item)
return normalized
return value
# ---------------------------------------------------------------------------
# Multimodal content helpers
# ---------------------------------------------------------------------------
@@ -627,10 +701,16 @@ def _preflight_codex_input_items(
raw_items: Any,
*,
is_github_responses: bool = False,
sanitize_harmony_tokens: bool = False,
) -> List[Dict[str, Any]]:
if not isinstance(raw_items, list):
raise ValueError("Codex Responses input must be a list of input items.")
sanitize_text = (
_neutralize_harmony_tokens
if sanitize_harmony_tokens
else lambda text: text
)
normalized: List[Dict[str, Any]] = []
seen_ids: set = set()
for idx, item in enumerate(raw_items):
@@ -651,7 +731,7 @@ def _preflight_codex_input_items(
arguments = json.dumps(arguments, ensure_ascii=False)
elif not isinstance(arguments, str):
arguments = str(arguments)
arguments = arguments.strip() or "{}"
arguments = sanitize_text(arguments.strip() or "{}")
normalized.append(
{
@@ -685,7 +765,7 @@ def _preflight_codex_input_items(
if ptype == "input_text":
text = part.get("text")
if isinstance(text, str) and text:
cleaned.append({"type": "input_text", "text": text})
cleaned.append({"type": "input_text", "text": sanitize_text(text)})
elif ptype == "input_image":
url = part.get("image_url")
if isinstance(url, str) and url:
@@ -709,7 +789,7 @@ def _preflight_codex_input_items(
{
"type": "function_call_output",
"call_id": call_id.strip(),
"output": output,
"output": sanitize_text(output),
}
)
continue
@@ -722,14 +802,21 @@ def _preflight_codex_input_items(
if item_id in seen_ids:
continue
seen_ids.add(item_id)
reasoning_item = {"type": "reasoning", "encrypted_content": encrypted}
reasoning_item: Dict[str, Any] = {
"type": "reasoning",
"encrypted_content": encrypted,
}
# Do NOT include the "id" in the outgoing item — with
# store=False (our default) the API tries to resolve the
# id server-side and returns 404. The id is still used
# above for local deduplication via seen_ids.
summary = item.get("summary")
if isinstance(summary, list):
reasoning_item["summary"] = summary
reasoning_item["summary"] = (
_neutralize_harmony_structure(summary)
if sanitize_harmony_tokens
else summary
)
else:
reasoning_item["summary"] = []
normalized.append(reasoning_item)
@@ -758,7 +845,7 @@ def _preflight_codex_input_items(
text = ""
if not isinstance(text, str):
text = str(text)
normalized_content.append({"type": "output_text", "text": text})
normalized_content.append({"type": "output_text", "text": sanitize_text(text)})
if not normalized_content:
raise ValueError(f"Codex Responses input[{idx}] message item must contain at least one text part.")
normalized_item: Dict[str, Any] = {
@@ -798,7 +885,7 @@ def _preflight_codex_input_items(
for part_idx, part in enumerate(content):
if isinstance(part, str):
if part:
validated.append({"type": text_type, "text": part})
validated.append({"type": text_type, "text": sanitize_text(part)})
continue
if not isinstance(part, dict):
raise ValueError(
@@ -809,7 +896,7 @@ def _preflight_codex_input_items(
text = part.get("text", "")
if not isinstance(text, str):
text = str(text or "")
validated.append({"type": text_type, "text": text})
validated.append({"type": text_type, "text": sanitize_text(text)})
elif ptype in {"input_image", "image_url"}:
image_ref = part.get("image_url", "")
detail = part.get("detail")
@@ -833,7 +920,7 @@ def _preflight_codex_input_items(
if not isinstance(content, str):
content = str(content)
normalized.append({"role": role, "content": content})
normalized.append({"role": role, "content": sanitize_text(content)})
continue
raise ValueError(
@@ -848,6 +935,7 @@ def _preflight_codex_api_kwargs(
*,
allow_stream: bool = False,
is_github_responses: bool = False,
sanitize_harmony_tokens: bool = False,
) -> Dict[str, Any]:
if not isinstance(api_kwargs, dict):
raise ValueError("Codex Responses request must be a dict.")
@@ -868,10 +956,13 @@ def _preflight_codex_api_kwargs(
if not isinstance(instructions, str):
instructions = str(instructions)
instructions = instructions.strip() or DEFAULT_AGENT_IDENTITY
if sanitize_harmony_tokens:
instructions = _neutralize_harmony_tokens(instructions)
normalized_input = _preflight_codex_input_items(
api_kwargs.get("input"),
is_github_responses=is_github_responses,
sanitize_harmony_tokens=sanitize_harmony_tokens,
)
tools = api_kwargs.get("tools")
@@ -928,6 +1019,9 @@ def _preflight_codex_api_kwargs(
}
)
if sanitize_harmony_tokens and normalized_tools is not None:
normalized_tools = _neutralize_harmony_structure(normalized_tools)
store = api_kwargs.get("store", False)
if store is not False:
raise ValueError("Codex Responses contract requires 'store' to be false.")
+19 -6
View File
@@ -1365,12 +1365,6 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
on_event=_on_event,
interrupt_check=_interrupt_or_superseded,
)
# The terminal SSE frame is contractually last. Request the
# end-of-stream marker so Relay can run its response finalizer
# and close the physical attempt scope before Hermes returns.
if not agent._interrupt_requested:
for _ignored in event_stream:
pass
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
if attempt < max_stream_retries:
logger.debug(
@@ -1386,6 +1380,25 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
return event_stream.final_response
raise
# A terminal response has already been assembled at this point
# (``final`` is built), so a transport error while draining the
# rest of the iterator — done only to let Relay run its response
# finalizer — must NOT discard it or trigger a new physical
# request. Record it as a non-fatal finalization warning and
# still return the already-completed, already-billed response.
if not agent._interrupt_requested:
try:
for _ignored in event_stream:
pass
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
logger.warning(
"Codex Responses stream transport finalization failed "
"after a terminal response was already received; "
"returning the completed response instead of "
"retrying. %s error=%s",
agent._client_log_context(), exc,
)
if final.status in {"incomplete", "failed"}:
logger.warning(
"Codex Responses stream terminal status=%s "
+303 -26
View File
@@ -25,7 +25,12 @@ import time
import uuid
from typing import Any, Dict, List, Optional
from agent.auxiliary_client import call_llm, _is_connection_error, aux_interrupt_protection
from agent.auxiliary_client import (
AuxiliaryExplicitCancellation,
_is_connection_error,
aux_interrupt_protection,
call_llm,
)
from agent.context_engine import ContextEngine, sanitize_memory_context
from agent.error_classifier import FailoverReason, classify_api_error
from agent.model_metadata import (
@@ -636,6 +641,12 @@ _ACTIVE_TASK_MAX_CHARS = 1400
# high for small/light tails, but using all 20 as a hard floor here would bring
# back the old large-tool-output case where nothing can be compacted.
_MAX_TAIL_MESSAGE_FLOOR = 8
# Pre-LLM feasibility skip (#60451): when the compressible middle is below
# this fraction of threshold_tokens (and a prior real-usage ineffectiveness
# strike exists), skip the LLM summary call — deterministic dropping alone
# recovers the negligible savings such a summary could deliver.
_FEASIBILITY_SKIP_MIDDLE_FRACTION = 0.10
# Under context pressure (protected-tail tool bodies alone exceed the soft
# tail budget), demote large completed tool/file outputs even inside the
# protected region — but always keep this many trailing messages verbatim so
@@ -764,15 +775,52 @@ def _serialized_length_for_budget(value: Any) -> int:
# Responses sessions in particular carry ``codex_reasoning_items`` blobs of
# ``encrypted_content`` that can dominate the serialized session (a measured
# 214-turn session held ~115K tokens / 27% of its payload there — #55572).
#
# ``reasoning_details`` is handled separately (see
# ``_reasoning_details_text_chars``): its signed/base64 envelope is excluded
# from the budget, mirroring the preflight estimator's exclusion in
# ``model_metadata._estimate_message_tokens_without_images`` (#73298).
_REPLAY_BUDGET_KEYS = (
"reasoning",
"reasoning_content",
"reasoning_details",
"codex_reasoning_items",
"codex_message_items",
)
def _reasoning_details_text_chars(value: Any) -> int:
"""Textual thinking chars inside a ``reasoning_details`` envelope.
``reasoning_details`` carries provider thinking blocks: the actual
thinking TEXT plus opaque signed/base64 envelope blobs (Anthropic
``signature``, redacted ``data``, encrypted payloads). The envelope is
never billed at anything near chars/4 by the provider and — on every
transport except Codex Responses — is replayed for at most the newest
assistant turn, so charging it on every message inflated the tail-budget
walk and silently shrank the surviving tail (#73298, second site).
Count only the thinking text (the #51800 lesson: real reasoning text
MUST stay visible to the budget), skip everything else.
"""
if not value:
return 0
if isinstance(value, str):
return len(value)
total = 0
if isinstance(value, dict):
value = [value]
if isinstance(value, list):
for part in value:
if isinstance(part, str):
total += len(part)
elif isinstance(part, dict):
for text_key in ("thinking", "text", "summary"):
text = part.get(text_key)
if isinstance(text, str):
total += len(text)
return total
def _estimate_msg_budget_tokens(msg: dict) -> int:
"""Token estimate for one message in the tail-protection budget walks.
@@ -804,6 +852,17 @@ def _estimate_msg_budget_tokens(msg: dict) -> int:
tokens += estimate_tokens_rough(str(tc))
for key in _REPLAY_BUDGET_KEYS:
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
# reasoning_details: charge only the thinking TEXT, never the signed /
# base64 envelope (#73298 second site; mirrors the preflight estimator's
# exclusion in model_metadata). When the same thinking text already rides
# in ``reasoning``/``reasoning_content`` (measured byte-identical on
# Anthropic-wire sessions), skip it here entirely so the prose is not
# charged twice on top of the envelope exclusion.
if not (msg.get("reasoning") or msg.get("reasoning_content")):
tokens += (
_reasoning_details_text_chars(msg.get("reasoning_details"))
// _CHARS_PER_TOKEN
)
return tokens
@@ -1281,11 +1340,13 @@ class ContextCompressor(ContextEngine):
self._consecutive_timeout_failures = 0
self._last_summary_dropped_count = 0
self._last_summary_fallback_used = False
self._last_feasibility_skip = False
self._last_aux_model_failure_error = None
self._last_aux_model_failure_model = None
self._last_compression_savings_pct = 100.0
self._ineffective_compression_count = 0
self._anti_thrash_recovery_deadline = 0.0
self._prellm_skip_count = 0
self._fallback_compression_streak = 0
self._verify_compaction_cleared_threshold = False
self._last_compression_made_progress = False
@@ -1337,6 +1398,7 @@ class ContextCompressor(ContextEngine):
"protected_head_tokens": None,
"protected_tail_tokens": None,
"middle_window_tokens": None,
"prellm_skip_count": 0,
"aux_prompt_tokens": None,
"aux_output_reservation": None,
"aux_provider": "",
@@ -1547,11 +1609,13 @@ class ContextCompressor(ContextEngine):
self._consecutive_timeout_failures = 0
self._last_summary_dropped_count = 0
self._last_summary_fallback_used = False
self._last_feasibility_skip = False
self._last_aux_model_failure_error = None
self._last_aux_model_failure_model = None
self._last_compression_savings_pct = 100.0
self._ineffective_compression_count = 0
self._anti_thrash_recovery_deadline = 0.0
self._prellm_skip_count = 0
self._fallback_compression_streak = 0
self._verify_compaction_cleared_threshold = False
self._last_compression_made_progress = False
@@ -1578,6 +1642,7 @@ class ContextCompressor(ContextEngine):
self._consecutive_timeout_failures = 0
self._fallback_compression_streak = 0
self._ineffective_compression_count = 0
self._prellm_skip_count = 0
self._anti_thrash_recovery_deadline = 0.0
self.get_active_compression_failure_cooldown()
self._load_fallback_compression_streak()
@@ -1722,9 +1787,34 @@ class ContextCompressor(ContextEngine):
self._ineffective_compression_count = count
self._persist_ineffective_compression_count()
def record_completed_compaction(self, *, used_fallback: bool = False) -> None:
"""Record one completed boundary and its summary quality."""
def record_completed_compaction(
self, *, used_fallback: bool = False, feasibility_skip: bool = False,
) -> None:
"""Record one completed boundary and its summary quality.
``feasibility_skip=True`` marks a deliberate pre-LLM skip (#60451):
the boundary is streak-NEUTRAL for ``_fallback_compression_streak``
(neither incremented nor reset). It still arms the real-usage
effectiveness verdict (``_verify_compaction_cleared_threshold``) on
purpose — a skipped-summary drop that fails to clear the threshold is
exactly the incompressible-transcript case the ineffective-strike
breaker exists for, and its recovery probe bounds the block.
"""
self._verify_compaction_cleared_threshold = True
if feasibility_skip:
# A deliberate pre-LLM feasibility skip (#60451) is not a
# summary-quality verdict: it must neither extend a fallback
# streak (two skips would otherwise latch the >= 2 breaker and
# disable compression entirely — including the cheap deterministic
# dropping the skip exists to reach) nor reset one (a skip proves
# nothing about the summary model's health).
if not self.quiet_mode:
logger.info(
"Compaction completed via pre-LLM feasibility skip; "
"fallback_compression_streak unchanged (%d)",
self._fallback_compression_streak,
)
return
if used_fallback:
self._fallback_compression_streak += 1
if not self.quiet_mode:
@@ -1743,6 +1833,11 @@ class ContextCompressor(ContextEngine):
refresh: bool = False,
) -> Optional[Dict[str, Any]]:
"""Return the live compression-failure cooldown for the bound session."""
if refresh:
# Transaction rollback must distinguish an authoritative empty row
# from a failed/unavailable durable read. The public return value
# cannot do so because it deliberately falls back to local state.
self._last_cooldown_refresh_was_authoritative = None
now_mono = time.monotonic()
local_state = None
if self._summary_failure_cooldown_until > now_mono:
@@ -1767,10 +1862,16 @@ class ContextCompressor(ContextEngine):
try:
state = getter(session_id)
except sqlite3.Error as exc:
if refresh:
self._last_cooldown_refresh_was_authoritative = False
logger.debug("compression failure cooldown lookup failed: %s", exc)
return local_state
except Exception:
if refresh:
self._last_cooldown_refresh_was_authoritative = False
return local_state
if refresh:
self._last_cooldown_refresh_was_authoritative = True
if not state:
if refresh:
if local_state is not None and self._cooldown_persist_failed:
@@ -1831,7 +1932,43 @@ class ContextCompressor(ContextEngine):
self._cooldown_persist_failed = True
logger.debug("compression failure cooldown persist failed (non-sqlite): %s", exc)
def record_timeout_failure(self, error: str) -> None:
"""Record a consecutive timeout failure using the shared cooldown ladder.
Used by both the summary-LLM exception handler (inline at line ~3714)
and the host-level ``compress_context`` timeout wrapper in
``run_compress_context_with_progress_timeout``. Avoids re-implementing
the ladder at each call site (#62452).
"""
_TIMEOUT_COOLDOWN_LADDER = (60, 300, 900)
self._consecutive_timeout_failures = (
getattr(self, "_consecutive_timeout_failures", 0) + 1
)
cooldown = _TIMEOUT_COOLDOWN_LADDER[
min(self._consecutive_timeout_failures,
len(_TIMEOUT_COOLDOWN_LADDER)) - 1
]
self._record_compression_failure_cooldown(float(cooldown), error)
def _clear_compression_failure_cooldown(self) -> None:
# #76354 review F4: fence check BEFORE cooldown-clear. A late worker
# whose host already timed out (and recorded a timeout cooldown) must
# not undo that cooldown when its summary eventually succeeds. The
# hook is installed by compress_context for the duration of the
# fenced call; when it reports cancellation, keep the host's cooldown.
cancelled_check = getattr(self, "_compression_cancelled_check", None)
if callable(cancelled_check):
try:
if cancelled_check():
logger.info(
"Skipping compression cooldown clear: host already "
"cancelled this compression attempt"
)
return
except Exception:
logger.debug(
"compression cancellation check failed", exc_info=True
)
self._summary_failure_cooldown_until = 0.0
self._last_summary_error = None
self._consecutive_timeout_failures = 0
@@ -1934,6 +2071,7 @@ class ContextCompressor(ContextEngine):
# trigger invalidates them. Keep the durable copy in sync so a
# restart doesn't resurrect strikes this recalibration just voided.
self._record_ineffective_compression_verdict(0)
self._prellm_skip_count = 0
if runtime_changed:
self._fallback_compression_streak = 0
self._persist_fallback_compression_streak()
@@ -2166,6 +2304,10 @@ class ContextCompressor(ContextEngine):
self._micro_compact_consecutive_failures: int = 0
self._micro_compact_last_failure_cursor: int = -1
self._micro_compact_defrag_threshold_tokens: int = 2000
# Set by _defrag_rolling_summary when it pops _DB_PERSISTED_MARKER
# from a live dict in place; consumed by finalize_turn to invalidate
# the agent's bounded flush-scan cursor (sibling of the #75170 site).
self._flush_scan_cursor_invalidated: bool = False
self._micro_compact_passes: int = 0
self._micro_compact_tokens_saved_total: int = 0
# Cadence: run a pass every Nth completed turn. Each pass rewrites
@@ -2227,6 +2369,9 @@ class ContextCompressor(ContextEngine):
# restart with a persisted tripped counter (#69872) waits a full fresh
# window before probing (#54923: restart must never disarm a guard).
self._anti_thrash_recovery_deadline: float = 0.0
# Pre-LLM feasibility skips (#60451). Observability only; NEVER feeds
# the ineffectiveness strike latch or the fallback streak breaker.
self._prellm_skip_count: int = 0
# Consecutive completed deterministic-fallback boundaries. Unlike the
# real-usage effectiveness counter, ordinary fitting responses must not
# reset this breaker; only a healthy completed summary does.
@@ -2248,6 +2393,7 @@ class ContextCompressor(ContextEngine):
# (gateway hygiene, /compress) can surface a visible warning.
self._last_summary_dropped_count: int = 0
self._last_summary_fallback_used: bool = False
self._last_feasibility_skip: bool = False
# When summary generation fails we now ABORT compression entirely
# and return the original messages unchanged instead of dropping
# the middle window with a static placeholder. Callers inspect
@@ -4236,6 +4382,12 @@ This compaction should PRIORITISE preserving all information related to the focu
end: int,
) -> list[tuple[int, str]]:
"""Find handoff summaries inside a compression window."""
n = len(messages)
# Defensive: clamp bounds so a caller passing an out-of-range end
# (e.g. tail-cut returning len(messages)+1 when head_end >= n)
# cannot trigger IndexError. (#75588)
start = max(0, min(start, n))
end = max(start, min(end, n))
summaries: list[tuple[int, str]] = []
for idx in range(start, end):
content = messages[idx].get("content")
@@ -5005,7 +5157,7 @@ This compaction should PRIORITISE preserving all information related to the focu
# exists to prevent. Re-align FORWARD (never backward, which would give
# the floor's message back) so a raised cut skips to the end of the
# group and the whole call/result pair is summarised together.
return self._align_boundary_forward(messages, max(cut_idx, head_end + 1))
return min(n, self._align_boundary_forward(messages, max(cut_idx, head_end + 1)))
# ------------------------------------------------------------------
# ContextEngine: manual /compress preflight
@@ -5322,6 +5474,15 @@ This compaction should PRIORITISE preserving all information related to the focu
# Content changed after a possible flush — clear the persisted
# stamp so the DB sync/flush rewrites the row.
entry.pop(_DB_PERSISTED_MARKER, None)
# Sibling of the finalize_turn pop site (#75170): this pop
# also strips the marker from a LIVE dict in place, so the
# bounded flush-scan cursor would identity-skip the rewritten
# marker and the defragged summary would never reach state.db.
# The compressor holds no agent reference, so raise a flag the
# finalizer consumes to invalidate agent._db_flush_scan_prefix.
# (The pop sites at module scope — fresh copies in
# strip-marker helpers — break identity and need no flag.)
self._flush_scan_cursor_invalidated = True
break
logger.info(
"Micro-compaction defrag: rolling summary re-summarized "
@@ -5784,7 +5945,11 @@ This compaction should PRIORITISE preserving all information related to the focu
1. Prune old tool results (cheap pre-pass, no LLM call)
2. Protect head messages (system prompt + first exchange)
3. Find tail boundary by token budget (~20K tokens of recent context)
4. Summarize middle turns with structured LLM prompt
4. Summarize middle turns with structured LLM prompt (skipped
pre-LLM when the middle is below
``_FEASIBILITY_SKIP_MIDDLE_FRACTION`` of the threshold after a
prior real-usage ineffectiveness strike — the deterministic
fallback drop recovers the negligible savings instead)
5. On re-compression, iteratively update the previous summary
Blank platform-echo user rows trailing the latest actionable user
@@ -5804,7 +5969,9 @@ This compaction should PRIORITISE preserving all information related to the focu
everything else. Inspired by Claude Code's ``/compact``.
force: If True, clear any active summary-failure cooldown before
running so a manual ``/compress`` can retry immediately after
an auto-compression abort. Auto-compress callers pass False.
an auto-compression abort, and bypass the pre-LLM feasibility
skip so an explicit user request always exercises the full
summary path. Auto-compress callers pass False.
memory_context: Optional provider-supplied context to preserve in
the summary prompt. Whitespace-only values are ignored.
"""
@@ -5812,6 +5979,7 @@ This compaction should PRIORITISE preserving all information related to the focu
# after compress() returns to decide whether to surface a warning.
self._last_summary_dropped_count = 0
self._last_summary_fallback_used = False
self._last_feasibility_skip = False
self._last_summary_error = None
self._last_aux_model_failure_error = None
self._last_aux_model_failure_model = None
@@ -5940,6 +6108,7 @@ This compaction should PRIORITISE preserving all information related to the focu
# — take the narrow rescan, miss a beyond-window fossil, and discard the
# rehydrated state as cross-session leakage (#57835).
_previous_summary_before_scan = self._previous_summary
_summary_has_user_turn_before_scan = getattr(self, "_summary_has_user_turn", None)
# A persisted handoff summary can sit in the protected head after a
# resume (commonly immediately after the system prompt). Search from
# the first non-system message through the compression window. On the
@@ -6094,12 +6263,73 @@ This compaction should PRIORITISE preserving all information related to the focu
)
# Phase 3: Generate structured summary
summary_focus_topic = focus_topic or self._derive_auto_focus_topic(messages)
summary = self._generate_summary(
turns_to_summarize,
focus_topic=summary_focus_topic,
memory_context=memory_context,
)
# Pre-LLM feasibility check: if the middle section is too small to
# yield meaningful token savings, skip the expensive LLM summarization
# call and fall through to the deterministic message-dropping path
# (which is cheap and always applicable). Without this guard a
# tool-heavy session where the protected tail already holds most of
# the tokens can burn 500+ seconds on a summary call that replaces a
# few lightweight messages, leaving the total token count essentially
# unchanged.
#
# Only fires after at least one prior real-usage ineffectiveness
# strike. The check READS ``_ineffective_compression_count`` but
# never writes it: that strike counter is fed exclusively by real
# provider token counts (see the anti-thrashing verdict in
# _update_token_usage), and consumers latch at >= 2 to disable
# compression entirely. Feasibility skips are tracked separately
# in ``_prellm_skip_count`` for observability.
#
# Skipped when ``force=True`` (manual /compress) so auth/error
# handling paths are always exercised on explicit user request.
feasibility_skip = False
if not force and self._ineffective_compression_count >= 1:
# _record_compression_regions already estimated this exact window
# into the telemetry dict above; reuse it so the log line and
# telemetry can never disagree. The regions helper no-ops when the
# telemetry attr isn't a dict, so fall back to a fresh estimate
# when the key is absent/None (0 is a legitimate value).
middle_tokens = telemetry.get("middle_window_tokens")
if middle_tokens is None:
middle_tokens = estimate_messages_tokens_rough(turns_to_summarize)
if middle_tokens < int(
self.threshold_tokens * _FEASIBILITY_SKIP_MIDDLE_FRACTION
):
feasibility_skip = True
self._last_feasibility_skip = True
self._prellm_skip_count += 1
telemetry["prellm_skip_count"] = self._prellm_skip_count
if not self.quiet_mode:
logger.warning(
"Compression: middle section (%d tokens at indices "
"%d-%d) is below %.0f%% of threshold (%d tokens) — "
"skipping LLM summarization, proceeding with "
"deterministic message dropping. prellm_skip_count=%d",
middle_tokens, compress_start, compress_end,
_FEASIBILITY_SKIP_MIDDLE_FRACTION * 100,
self.threshold_tokens, self._prellm_skip_count,
)
if feasibility_skip:
summary = None # No LLM call; Phase 4 inserts the deterministic fallback
else:
# Deriving the auto focus topic scans recent user turns — only pay
# for it when a summary will actually be generated.
summary_focus_topic = focus_topic or self._derive_auto_focus_topic(messages)
try:
summary = self._generate_summary(
turns_to_summarize,
focus_topic=summary_focus_topic,
memory_context=memory_context,
)
except AuxiliaryExplicitCancellation:
# Explicit cancellation is a true no-op. Restore state mutated by
# the resume/handoff self-heal scan before the exception escapes to
# the outer transaction, which restores the transcript and lease.
self._previous_summary = _previous_summary_before_scan
self._summary_has_user_turn = _summary_has_user_turn_before_scan
raise
# If summary generation failed, behavior splits on
# ``abort_on_summary_failure`` (config: compression.abort_on_summary_failure):
@@ -6121,7 +6351,7 @@ This compaction should PRIORITISE preserving all information related to the focu
# of these cases, rotating into a child session with a placeholder
# summary degrades the conversation for zero benefit. Preserve it
# unchanged until access is restored or connectivity recovers.
if not summary and (
if not summary and not feasibility_skip and (
self.abort_on_summary_failure
or self._last_summary_auth_failure
or self._last_summary_network_failure
@@ -6201,15 +6431,26 @@ This compaction should PRIORITISE preserving all information related to the focu
# content-free "N messages were removed" marker.
if not summary:
if not self.quiet_mode:
logger.warning("Summary generation failed — inserting deterministic fallback context summary")
if feasibility_skip:
logger.info("Feasibility skip — inserting deterministic fallback context summary")
else:
logger.warning("Summary generation failed — inserting deterministic fallback context summary")
n_dropped = compress_end - compress_start
self._last_summary_dropped_count = n_dropped
self._last_summary_fallback_used = True
telemetry["fallback_used"] = True
telemetry["failure_class"] = telemetry.get("failure_class") or "summary_generation_failed"
if feasibility_skip:
# Deliberate optimization, not a summary failure — keep the
# telemetry class distinct so dashboards don't count skips
# as aux-model breakage.
telemetry["failure_class"] = telemetry.get("failure_class") or "feasibility_skip"
else:
telemetry["failure_class"] = telemetry.get("failure_class") or "summary_generation_failed"
summary = self._build_static_fallback_summary(
turns_to_summarize,
reason=self._last_summary_error,
# A stale error from an earlier real failure must not be
# embedded into a deliberate feasibility skip's fallback.
reason=None if feasibility_skip else self._last_summary_error,
)
tail_messages: List[Dict[str, Any]] = []
@@ -6258,16 +6499,18 @@ This compaction should PRIORITISE preserving all information related to the focu
None,
)
first_tail_role = None
first_tail_visible_idx: Optional[int] = None
if tail_messages:
first_tail_role = next(
first_tail_visible_idx, first_tail_role = next(
(
role
for role in (
_template_visible_role(m) for m in tail_messages
(idx, role)
for idx, role in (
(idx, _template_visible_role(m))
for idx, m in enumerate(tail_messages)
)
if role is not None
),
None,
(None, None),
)
# When the only protected head message is the system prompt, the
# summary becomes the first *visible* message in the API request
@@ -6294,11 +6537,27 @@ This compaction should PRIORITISE preserving all information related to the focu
# If no user-role message survives in either the protected head or the
# preserved tail, the summary MUST carry role="user" so the request
# always has at least one user turn.
#
# A bare role check is not enough: the tail's sole surviving user
# turn can be image-only (a screenshot with no caption). The newest
# image-bearing user message is the ``_strip_historical_media``
# anchor and is kept byte-for-byte, so it never gains a text
# placeholder — its role is "user" but its text content is empty,
# which backends checking for actual query text still reject. Count
# only user messages with non-empty text as "surviving"; when the
# guard fires, the real (never fabricated) summary text lands in a
# role="user" slot, which is always non-empty (falls back to
# ``_build_static_fallback_summary`` above when generation fails).
if not _force_user_leading:
def _is_nonempty_user_turn(message: Dict[str, Any]) -> bool:
return message.get("role") == "user" and bool(
_content_text_for_contains(message.get("content")).strip()
)
_user_survives = any(
message.get("role") == "user" for message in compressed
_is_nonempty_user_turn(message) for message in compressed
) or any(
message.get("role") == "user" for message in tail_messages
_is_nonempty_user_turn(message) for message in tail_messages
)
if not _user_survives:
_force_user_leading = True
@@ -6354,9 +6613,27 @@ This compaction should PRIORITISE preserving all information related to the focu
),
})
# Default merge target: literal tail index 0. For an ordinary
# alternation collision the summary only has to stay *invisible* to
# the template, and a leading template-exempt row (bare tool-call
# assistant message, tool result) is the ideal carrier — it absorbs
# the summary without adding a visible turn, and it leaves the live
# tail user message intact as the model's actual prompt. Retargeting
# to the first template-visible row here would convert that live
# request into the summary carrier for no benefit.
#
# The forced repair path is the exception. There the merge is not
# about alternation but about guaranteeing at least one genuinely
# non-empty role="user" message (an image-only or otherwise
# text-empty surviving user row). An exempt carrier cannot satisfy
# that invariant, so the summary text must land on the
# template-visible row itself.
_merge_target_idx = 0
if _force_user_leading and first_tail_visible_idx is not None:
_merge_target_idx = first_tail_visible_idx
for tail_idx, msg in enumerate(tail_messages):
if _merge_summary_into_tail and tail_idx == 0:
# Merge the summary into the first (post-strip) tail message.
if _merge_summary_into_tail and tail_idx == _merge_target_idx:
# Merge the summary into the tail message that collided.
old_content = msg.get("content", "")
if _force_user_leading and summary_role == "user":
# The summary must be part of the first user-visible
File diff suppressed because it is too large Load Diff
+232 -67
View File
@@ -70,8 +70,9 @@ from agent.model_metadata import (
)
from agent.process_bootstrap import _install_safe_stdio
from agent.prompt_caching import (
apply_anthropic_cache_control,
build_prompt_cache_plan,
strip_anthropic_cache_control,
strip_anthropic_tool_cache_control,
)
from agent.retry_utils import (
adaptive_rate_limit_backoff,
@@ -194,6 +195,54 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text
agent._stream_needs_break = True
def _is_copilot_provider(agent: Any) -> bool:
"""Delegate to ``AIAgent._is_copilot_provider`` (single owner of the check).
``agent.provider`` is not always the normalized ``copilot`` slug —
``/model`` and profile configs can leave the alias ``github-copilot`` (or
``github``) in place, and a bare ``provider == "copilot"`` gate silently
skips credential recovery for those spellings.
"""
try:
return bool(agent._is_copilot_provider())
except Exception:
return (getattr(agent, "provider", "") or "").strip().lower() in {
"copilot",
"github-copilot",
"github",
}
def _is_stale_copilot_credential_error(status_code: Optional[int], error_message: str) -> bool:
"""Detect a Copilot 400 that is really a STALE / DEGRADED credential.
Copilot surfaces a stale or degraded credential as an HTTP 400 rather than a
clean 401. Two body markers indicate this class:
- ``model_not_available_for_integrator`` — the request reached the
restricted ``copilot-language-server`` integrator (the server's fallback
when it receives a raw OAuth token instead of an exchanged API token),
whose model allowlist omits enterprise-only models.
- ``model_not_supported`` / "the requested model is not supported" — the
cached bearer's Copilot entitlement rotated out from under a long-lived
process.
Matched narrowly (status 400 AND a specific marker) so a genuinely wrong
model name — a real 400 — never triggers the single-shot re-exchange. The
caller enforces copilot-provider scoping and the single-shot guard.
"""
lowered = (error_message or "").lower()
is_400 = status_code == 400 or "error code: 400" in lowered
if not is_400:
return False
return (
"model_not_available_for_integrator" in lowered
or "not available for integrator" in lowered
or "model_not_supported" in lowered
or "the requested model is not supported" in lowered
)
def _image_error_max_dimension(error: Exception) -> Optional[int]:
"""Extract a provider-reported image dimension ceiling, if present."""
parts = []
@@ -689,6 +738,92 @@ _CONTENT_POLICY_RECOVERY_HINT = (
)
# Memo for the send-path tool-call argument canonicalization inside
# run_conversation(). That pass re-canonicalizes the arguments string of
# EVERY historical tool call on EVERY API-call iteration (quadratic in
# session tool-call count), and the api_messages copies share the exact
# argument string objects with the persisted history, so the same strings
# come through unchanged iteration after iteration.
#
# Soundness: canonicalization is a pure, deterministic function of the
# input string (fixed separators, sort_keys=True), so a value-keyed memo
# is exact — equal inputs always produce the canonical form computed the
# first time. Malformed strings raise out of json.loads BEFORE anything
# is stored, so the repair fallback below is never memoized and reruns on
# every occurrence, exactly as before. Bounded FIFO eviction mirrors the
# _MSG_TOKENS_CACHE idiom in agent/model_metadata.py.
_CANON_ARGS_CACHE: Dict[str, str] = {}
_CANON_ARGS_CACHE_MAX = 4096
# Count bound alone doesn't bound MEMORY: write_file/patch argument strings
# run 100KB+, so 4096 entries could pin ~800MB in a long-lived gateway
# process. The byte budget keeps the memo effective for the common case
# (args ~0.5-2KB) while bounding the worst case.
_CANON_ARGS_CACHE_MAX_BYTES = 32 * 1024 * 1024
_canon_args_cache_bytes = 0
def _canonicalize_tool_call_arguments(arg_str: str) -> str:
"""Return the canonical wire form of a tool-call arguments JSON string.
Raises whatever ``json.loads`` raises on malformed input; the caller
falls back to ``_repair_tool_call_arguments``, exactly as before.
"""
global _canon_args_cache_bytes
cached = _CANON_ARGS_CACHE.get(arg_str)
if cached is not None:
return cached
canonical = json.dumps(
json.loads(arg_str), separators=(",", ":"), sort_keys=True,
)
_CANON_ARGS_CACHE[arg_str] = canonical
_canon_args_cache_bytes += len(arg_str) + len(canonical)
while len(_CANON_ARGS_CACHE) > _CANON_ARGS_CACHE_MAX or (
_canon_args_cache_bytes > _CANON_ARGS_CACHE_MAX_BYTES
and len(_CANON_ARGS_CACHE) > 1
):
try:
evicted_key = next(iter(_CANON_ARGS_CACHE))
evicted_val = _CANON_ARGS_CACHE.pop(evicted_key)
_canon_args_cache_bytes -= len(evicted_key) + len(evicted_val)
except (StopIteration, KeyError, RuntimeError):
break
return canonical
def _canonicalize_api_tool_calls(api_messages) -> None:
"""Canonicalize tool-call argument JSON on the send-path message copy.
Rewrites each message's ``tool_calls`` in place (copy-on-write for the
tool-call dicts it canonicalizes; the persisted history is untouched).
The pass still traverses every message and tool call each iteration;
the memo above bounds the JSON parse/serialize work to one round-trip
per UNIQUE argument string instead of one per string per iteration —
the quadratic part of the cost. The remaining traversal is pointer
chasing and dict copies, cheap next to a json.loads + json.dumps.
"""
for am in api_messages:
tcs = am.get("tool_calls")
if not tcs:
continue
new_tcs = []
for tc in tcs:
if isinstance(tc, dict) and "function" in tc:
try:
tc = {**tc, "function": {
**tc["function"],
"arguments": _canonicalize_tool_call_arguments(
tc["function"]["arguments"]
),
}}
except Exception:
tc["function"]["arguments"] = _repair_tool_call_arguments(
tc["function"]["arguments"],
tc["function"].get("name", "?"),
)
new_tcs.append(tc)
am["tool_calls"] = new_tcs
def _invalid_tool_name_error_content(name: str, valid_tool_names) -> str:
"""Error-result content for a tool call whose name isn't a real tool.
@@ -893,7 +1028,8 @@ def _redecorate_prompt_cache_for_provider(
*,
system_message=None,
moa_prepared: Optional[Dict[str, Any]] = None,
) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]]:
tools_for_api: Optional[List[Dict[str, Any]]] = None,
) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]] | tuple[List[Dict[str, Any]], Optional[Dict[str, Any]], List[Dict[str, Any]]]:
"""Strip and re-apply cache_control for the *current* provider policy.
Decoration runs once per call block before the retry loop for the primary
@@ -903,10 +1039,9 @@ def _redecorate_prompt_cache_for_provider(
by reshaping at the top of each retry attempt.
The source list is the mutated in-flight request (image shrink / ASCII /
reasoning_details recoveries already applied) — never a pristine
pre-decoration snapshot. MoA guidance is peeled, the base is redecorated,
then ``rebase_prepared_request`` re-attaches guidance outside the cached
span.
reasoning_details recoveries already applied), never a pristine
pre-decoration snapshot. MoA guidance is peeled and rebased without
decoration; the acting aggregator plans its resolved destination later.
"""
messages: List[Dict[str, Any]] = [
dict(m) if isinstance(m, dict) else m for m in (api_messages or [])
@@ -917,6 +1052,21 @@ def _redecorate_prompt_cache_for_provider(
messages = _peel_moa_guidance(messages, guidance)
strip_anthropic_cache_control(messages)
planned_tools = strip_anthropic_tool_cache_control(
tools_for_api if tools_for_api is not None else getattr(agent, "tools", [])
)
if prepared is not None and getattr(agent, "provider", None) == "moa":
# Prepared MoA state is canonical: the synchronous acting-aggregator
# sender owns its destination-local cache plan after it resolves the slot.
completions = getattr(getattr(agent.client, "chat", None), "completions", None)
rebase = getattr(completions, "rebase_prepared_request", None)
if callable(rebase):
prepared = rebase(prepared, messages)
messages = prepared["messages"]
if tools_for_api is None:
return messages, prepared
return messages, prepared, planned_tools
# Direct attribute access matches the call-block decoration site — the
# flags are unconditionally initialized on AIAgent, and a getattr
@@ -924,31 +1074,25 @@ def _redecorate_prompt_cache_for_provider(
if agent._use_prompt_caching:
_ensure_cached_system_prompt_static(agent, system_message=system_message)
static = getattr(agent, "_cached_system_prompt_static", None)
messages = apply_anthropic_cache_control(
direct_tool_cache = getattr(
agent,
"_direct_native_anthropic_tool_cache_capability",
lambda: False,
)()
plan = build_prompt_cache_plan(
messages,
planned_tools,
cache_ttl=agent._cache_ttl,
native_anthropic=agent._use_native_cache_layout,
static_system_prefix=static if isinstance(static, str) else None,
direct_native_tool_cache=direct_tool_cache,
)
messages = plan.messages
planned_tools = plan.tools
if (
prepared is not None
and getattr(agent, "provider", None) == "moa"
):
# No `and guidance` here: guidance=None is a real prepared shape
# (all-references-failed / silent degraded policy builds the
# prepared request without attaching guidance), and the MoA facade
# sends prepared["messages"] — not api_kwargs["messages"] — so the
# rebase must refresh the prepared object even when there is no
# guidance to re-attach. rebase_prepared_request handles falsy
# guidance by copying the messages and skipping the attach.
completions = getattr(getattr(agent.client, "chat", None), "completions", None)
rebase = getattr(completions, "rebase_prepared_request", None)
if callable(rebase):
prepared = rebase(prepared, messages)
messages = prepared["messages"]
return messages, prepared
if tools_for_api is None:
return messages, prepared
return messages, prepared, planned_tools
def _apply_context_engine_selection(
@@ -1661,29 +1805,7 @@ def run_conversation(
for am in api_messages:
if isinstance(am.get("content"), str):
am["content"] = am["content"].strip()
for am in api_messages:
tcs = am.get("tool_calls")
if not tcs:
continue
new_tcs = []
for tc in tcs:
if isinstance(tc, dict) and "function" in tc:
try:
args_obj = json.loads(tc["function"]["arguments"])
tc = {**tc, "function": {
**tc["function"],
"arguments": json.dumps(
args_obj, separators=(",", ":"),
sort_keys=True,
),
}}
except Exception:
tc["function"]["arguments"] = _repair_tool_call_arguments(
tc["function"]["arguments"],
tc["function"].get("name", "?"),
)
new_tcs.append(tc)
am["tool_calls"] = new_tcs
_canonicalize_api_tool_calls(api_messages)
# Proactively strip any surrogate characters before the API call.
# Models served via Ollama (Kimi K2.5, GLM-5, Qwen) can return
@@ -1700,12 +1822,8 @@ def run_conversation(
# regardless of ordering (a single-space pad here previously had to
# be sequenced after normalization to survive, forking the concept).
# Apply Anthropic prompt caching for Claude models on native
# Anthropic, OpenRouter, and third-party Anthropic-compatible
# gateways. Auto-detected: if ``_use_prompt_caching`` is set, inject
# cache_control breakpoints for the static system prefix, full system
# prompt, and last two messages (or the legacy system-and-3 layout
# when no static prefix is available).
# Build the request-local cache sections only after every transcript
# mutation. The canonical tool registry stays undecorated.
#
# Runs LAST, after every message mutation above. Marking earlier
# defeats the prefix stability the mutations exist to create:
@@ -1720,10 +1838,12 @@ def run_conversation(
# exactly the point the breakpoints were meant to protect. Marking
# last also keeps breakpoints off messages that the orphan sweep or
# the thinking-only drop is about to remove or merge away.
if agent._use_prompt_caching:
tools_for_api = agent.tools
if agent._use_prompt_caching and agent.provider != "moa":
_static_system_prefix = getattr(agent, "_cached_system_prompt_static", None)
api_messages = apply_anthropic_cache_control(
_initial_cache_plan = build_prompt_cache_plan(
api_messages,
tools_for_api,
cache_ttl=agent._cache_ttl,
native_anthropic=agent._use_native_cache_layout,
static_system_prefix=(
@@ -1731,7 +1851,10 @@ def run_conversation(
if isinstance(_static_system_prefix, str)
else None
),
direct_native_tool_cache=agent._direct_native_anthropic_tool_cache_capability(),
)
api_messages = _initial_cache_plan.messages
tools_for_api = _initial_cache_plan.tools
# Build a persistent-MoA request before measuring compression pressure.
# MoA reference output is injected into the aggregator prompt, but it
@@ -2067,15 +2190,22 @@ def run_conversation(
# fallback refreshes the policy flags, but the decorated list
# still carries the primary's breakpoints (or none). Strip and
# re-render for the current provider before building kwargs.
api_messages, _moa_prepared_request = (
api_messages, _moa_prepared_request, tools_for_api = (
_redecorate_prompt_cache_for_provider(
agent,
api_messages,
system_message=system_message,
moa_prepared=_moa_prepared_request,
tools_for_api=tools_for_api,
)
)
api_kwargs = agent._build_api_kwargs(api_messages)
if tools_for_api == agent.tools:
api_kwargs = agent._build_api_kwargs(api_messages)
else:
api_kwargs = agent._build_api_kwargs(
api_messages,
tools_for_api=tools_for_api,
)
if agent._force_ascii_payload:
_sanitize_structure_non_ascii(api_kwargs)
if agent.api_mode == "codex_responses":
@@ -2083,6 +2213,7 @@ def run_conversation(
api_kwargs,
allow_stream=False,
is_github_responses=agent._is_copilot_url(),
sanitize_harmony_tokens=agent._is_codex_backend(),
)
# Copilot x-initiator: the first API call of a user turn is
# marked "user" so Copilot bills a premium request; tool-loop
@@ -2242,6 +2373,7 @@ def run_conversation(
next_api_kwargs,
allow_stream=False,
is_github_responses=agent._is_copilot_url(),
sanitize_harmony_tokens=agent._is_codex_backend(),
)
if _use_streaming:
return agent._interruptible_streaming_api_call(
@@ -2540,7 +2672,7 @@ def run_conversation(
# Terminal — flush buffered retry trace so user sees what happened.
agent._flush_status_buffer()
agent._emit_status(f"❌ Max retries ({max_retries}) exceeded for invalid responses. Giving up.")
logger.error(f"{agent.log_prefix}Invalid API response after {max_retries} retries.")
logger.error("%sInvalid API response after %d retries.", agent.log_prefix, max_retries)
agent._persist_session(messages, conversation_history)
_final_response = f"Invalid API response after {max_retries} retries: {_failure_hint}"
return {
@@ -2555,7 +2687,7 @@ def run_conversation(
# Backoff before retry — jittered exponential: 5s base, 120s cap
wait_time = jittered_backoff(retry_count, base_delay=5.0, max_delay=120.0)
agent._buffer_vprint(f"⏳ Retrying in {wait_time:.1f}s ({_failure_hint})...")
logger.warning(f"Invalid API response (retry {retry_count}/{max_retries}): {', '.join(error_details)} | Provider: {provider_name}")
logger.warning("Invalid API response (retry %d/%d): %s | Provider: %s", retry_count, max_retries, ', '.join(error_details), provider_name)
# Sleep in small increments to stay responsive to interrupts
sleep_end = time.time() + wait_time
@@ -3859,7 +3991,7 @@ def run_conversation(
print(f"{agent.log_prefix} • Verify stored credentials: {_dhh}/auth.json")
print(f"{agent.log_prefix} • Switch providers temporarily: /model <model> --provider openrouter")
if (
agent.provider == "copilot"
_is_copilot_provider(agent)
and status_code == 401
and not _retry.copilot_auth_retry_attempted
):
@@ -4467,7 +4599,7 @@ def run_conversation(
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached for payload-too-large error.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error(f"{agent.log_prefix}413 compression failed after {max_compression_attempts} attempts.")
logger.error("%s413 compression failed after %d attempts.", agent.log_prefix, max_compression_attempts)
agent._persist_session(messages, conversation_history)
_final_response = f"Request payload too large: max compression attempts ({max_compression_attempts}) reached."
return {
@@ -4536,7 +4668,7 @@ def run_conversation(
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Payload too large and cannot compress further.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error(f"{agent.log_prefix}413 payload too large. Cannot compress further.")
logger.error("%s413 payload too large. Cannot compress further.", agent.log_prefix)
agent._persist_session(messages, conversation_history)
_final_response = "Request payload too large (413). Cannot compress further."
return {
@@ -4609,7 +4741,7 @@ def run_conversation(
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error(f"{agent.log_prefix}Context compression failed after {max_compression_attempts} attempts.")
logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts)
agent._persist_session(messages, conversation_history)
_final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached."
return {
@@ -4698,6 +4830,13 @@ def run_conversation(
provider=agent.provider,
api_mode=agent.api_mode,
)
# Persist an explicit provider-reported limit before
# compression/retry. The next request can be rate
# limited, omit usage, or the process can restart; none
# of those should discard metadata the provider already
# confirmed. Keep the probe flags as a best-effort
# post-success retry if this write cannot complete.
save_context_length(agent.model, agent.base_url, new_ctx)
# Context probing flags — only set on built-in
# compressor (plugin engines manage their own). This
# value came from the provider, so it is safe to cache.
@@ -4721,7 +4860,7 @@ def run_conversation(
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True)
logger.error(f"{agent.log_prefix}Context compression failed after {max_compression_attempts} attempts.")
logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts)
agent._persist_session(messages, conversation_history)
_final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached."
return {
@@ -4779,7 +4918,7 @@ def run_conversation(
agent._flush_status_buffer()
agent._vprint(f"{agent.log_prefix}❌ Context length exceeded and cannot compress further.", force=True)
agent._vprint(f"{agent.log_prefix} 💡 The conversation has accumulated too much content. Try /new to start fresh, or /compress to manually trigger compression.", force=True)
logger.error(f"{agent.log_prefix}Context length exceeded: {new_tokens:,} tokens. Cannot compress further.")
logger.error("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}")
agent._persist_session(messages, conversation_history)
_final_response = f"Context length exceeded ({new_tokens:,} tokens). Cannot compress further."
return {
@@ -4865,6 +5004,32 @@ def run_conversation(
) and not is_context_length_error
if is_client_error:
# Copilot self-heal BEFORE fallback: a stale/degraded
# credential surfaces as a 400
# ``model_not_available_for_integrator`` /
# ``model_not_supported`` (not a clean 401), so the 401
# refresh path above never fired. Force a fresh token
# exchange + client rebuild and retry once on the SAME
# provider — a fresh 437-char API token routes to the
# correct integrator and the model becomes available again.
# Single-shot guard prevents looping on a genuinely
# unavailable model. Copilot-scoped so other providers'
# real 400s are untouched.
if (
_is_copilot_provider(agent)
and not _retry.copilot_stale_cred_retry_attempted
and _is_stale_copilot_credential_error(
status_code, str(getattr(api_error, "message", "") or api_error)
)
):
_retry.copilot_stale_cred_retry_attempted = True
if agent._try_recover_stale_copilot_credential():
agent._buffer_vprint(
"🔐 Copilot credential re-exchanged after "
"model_not_available 400. Retrying request..."
)
retry_count = 0
continue
# Try fallback before aborting — a different provider may
# not have the same issue (rate limit, auth, etc.). Only
# announce the attempt when a fallback chain actually
@@ -5019,7 +5184,7 @@ def run_conversation(
f"{agent.log_prefix} for localhost, or add the server's cert to your trust store.",
force=True,
)
logger.error(f"{agent.log_prefix}Non-retryable client error: {api_error}")
logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error)
# Skip session persistence when the error is likely
# context-overflow related (status 400 + large session).
# Persisting the failed user message would make the
+117 -34
View File
@@ -28,10 +28,13 @@ from hermes_cli.auth import (
_auth_store_lock,
_codex_access_token_is_expiring,
_decode_jwt_claims,
_global_auth_file_path,
_load_auth_store,
_load_provider_state,
_load_provider_state_with_source,
_resolve_kimi_base_url,
_resolve_zai_base_url,
_same_path,
_save_auth_store,
_save_provider_state,
_store_provider_state,
@@ -792,7 +795,7 @@ class CredentialPool:
device_code-sourced entries; env/API-key-sourced entries have no
auth.json shadow to sync from.
"""
if self.provider != "openai-codex" or entry.source != "device_code":
if self.provider != "openai-codex" or entry.source not in ("device_code", "manual:device_code"):
return entry
try:
with _auth_store_lock():
@@ -808,19 +811,43 @@ class CredentialPool:
# Adopt auth.json tokens when either side differs. Codex refresh
# tokens are single-use too, so a fresh refresh_token from
# another process means our entry's pair is consumed/stale.
#
# Also adopt when the store has a refresh_token but no
# access_token — another process may have rotated the pair
# and the store entry's access_token was already consumed;
# the important signal is the refresh_token difference.
entry_access = entry.access_token or ""
entry_refresh = entry.refresh_token or ""
should_adopt = False
if store_access and (
store_access != entry_access
or (store_refresh and store_refresh != entry_refresh)
):
should_adopt = True
elif (
store_refresh
and store_refresh != entry_refresh
and not store_access
):
# Store has only a refresh_token (no access_token) —
# another process rotated the pair. Adopt the
# refresh_token so we don't replay the consumed one.
logger.info(
"Pool entry %s: auth.json has newer refresh_token "
"but no access_token; adopting refresh_token to "
"avoid replaying consumed token",
entry.id,
)
should_adopt = True
if should_adopt:
logger.debug(
"Pool entry %s: syncing Codex tokens from auth.json "
"(refreshed by another process)",
entry.id,
)
field_updates: Dict[str, Any] = {
"access_token": store_access,
"access_token": store_access or entry.access_token,
"refresh_token": store_refresh or entry.refresh_token,
"last_status": None,
"last_status_at": None,
@@ -1039,32 +1066,58 @@ class CredentialPool:
try:
with _auth_store_lock():
auth_store = _load_auth_store()
# Decide BEFORE writing whether this profile is reading the
# grant from the global root (no own providers.<id> block) vs.
# genuinely shadowing it. A pool refresh rotates single-use
# OAuth refresh tokens, so a profile that resolved the grant
# from root MUST write the rotated chain back to root too —
# otherwise root keeps a revoked refresh token and every other
# profile reading the stale root grant dies with
# refresh_token_reused / invalid_grant once its access token
# expires. This mirrors the xAI write-through in
# hermes_cli.auth._save_xai_oauth_tokens (#43589); the pool
# refresh path is the Codex/xAI analog reported in #48415.
_wt_provider_id = {
"nous": "nous",
"openai-codex": "openai-codex",
"xai-oauth": "xai-oauth",
}.get(self.provider)
write_through_to_root = bool(_wt_provider_id) and not (
isinstance(auth_store.get("providers"), dict)
and isinstance(
auth_store["providers"].get(_wt_provider_id), dict
)
)
# Resolve state and track which store it came from — the
# source path tells us whether this profile genuinely owns
# its provider block or is reading from the global root.
# #74339: the old key-presence check decided write-through
# on whether the profile had ``providers.<id>`` BEFORE the
# save — correct for the first refresh but self-sealing
# because ``_store_provider_state`` unconditionally creates
# that key inside the same function. Once the profile has
# the key, every subsequent refresh silently disables the
# root write-through and root keeps a revoked refresh token.
#
# Fix: use ``_load_provider_state_with_source`` to learn
# where the state was resolved from. When the grant was
# resolved from the global root, write back *only* to root
# and skip ``_store_provider_state`` for the profile so the
# profile does not accrue a shadowing ``providers.<id>``
# key that blocks both the root fallback and the write-through
# on subsequent calls.
if self.provider == "nous":
state = _load_provider_state(auth_store, "nous")
state, source_path = _load_provider_state_with_source(
auth_store, "nous"
)
if state is None:
return
elif self.provider == "openai-codex":
state, source_path = _load_provider_state_with_source(
auth_store, "openai-codex"
)
if not isinstance(state, dict):
return
elif self.provider == "xai-oauth":
state, source_path = _load_provider_state_with_source(
auth_store, "xai-oauth"
)
if not isinstance(state, dict):
return
else:
return
global_root = _global_auth_file_path()
is_from_root = bool(
source_path is not None
and global_root is not None
and _same_path(source_path, global_root)
)
if self.provider == "nous":
state["access_token"] = entry.access_token
if entry.refresh_token:
state["refresh_token"] = entry.refresh_token
@@ -1082,12 +1135,8 @@ class CredentialPool:
state[extra_key] = val
if entry.inference_base_url:
state["inference_base_url"] = entry.inference_base_url
_store_provider_state(auth_store, "nous", state, set_active=False)
elif self.provider == "openai-codex":
state = _load_provider_state(auth_store, "openai-codex")
if not isinstance(state, dict):
return
tokens = state.get("tokens")
if not isinstance(tokens, dict):
return
@@ -1096,12 +1145,8 @@ class CredentialPool:
tokens["refresh_token"] = entry.refresh_token
if entry.last_refresh:
state["last_refresh"] = entry.last_refresh
_store_provider_state(auth_store, "openai-codex", state, set_active=False)
elif self.provider == "xai-oauth":
state = _load_provider_state(auth_store, "xai-oauth")
if not isinstance(state, dict):
return
tokens = state.get("tokens")
if not isinstance(tokens, dict):
return
@@ -1110,16 +1155,26 @@ class CredentialPool:
tokens["refresh_token"] = entry.refresh_token
if entry.last_refresh:
state["last_refresh"] = entry.last_refresh
_store_provider_state(auth_store, "xai-oauth", state, set_active=False)
else:
return
_save_auth_store(auth_store)
if write_through_to_root and _wt_provider_id:
if is_from_root and _wt_provider_id:
# Grant was resolved from root — write back to root
# only. Do NOT call _store_provider_state on the
# profile auth_store (it would create a shadowing
# providers.<id> key that disables write-through on
# the next refresh — #74339).
# _load_provider_state has root fallback, so the
# profile can always read fresh tokens from root
# without needing its own providers block.
_write_through_provider_state_to_global_root(
_wt_provider_id, state
)
else:
# Profile genuinely owns this provider — write to
# the profile store as normal.
_store_provider_state(
auth_store, self.provider, state, set_active=False
)
_save_auth_store(auth_store)
except Exception as exc:
logger.debug("Failed to sync %s pool entry back to auth store: %s", self.provider, exc)
@@ -2336,6 +2391,20 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup
token, source = resolve_copilot_token()
if token:
api_token, enterprise_base_url = get_copilot_api_token(token)
# Observability: get_copilot_api_token falls back to returning
# the RAW token when the exchange fails. A raw ~40-char token
# sent to the Copilot API is routed to the fallback
# "copilot-language-server" integrator, whose allowlist omits
# enterprise-only models (claude-opus-4.8) → HTTP 400 on every
# turn. exchange_copilot_token now retries + reuses a persisted
# JWT, so this should be rare; surface it at WARNING so a
# recurrence is visible in logs instead of failing silently.
if api_token == token and not enterprise_base_url:
logger.warning(
"Copilot token exchange degraded to RAW token (exchange "
"unavailable); enterprise-only models may 400 with "
"model_not_available_for_integrator until exchange recovers."
)
source_name = "gh_cli" if "gh" in source.lower() else f"env:{source}"
if not _is_suppressed(provider, source_name):
active_sources.add(source_name)
@@ -2506,6 +2575,20 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
changed = False
active_sources: Set[str] = set()
# Copilot has its own dedicated seeding branch (see `_seed_credentials`
# for provider == "copilot") which exchanges the raw ghu_ OAuth token
# for the ~437-char api token via `get_copilot_api_token`. If we let
# the generic env-var loop below run for copilot, it re-reads
# COPILOT_GITHUB_TOKEN from .env and shoves the RAW 40-char token in
# as `access_token`, overwriting the correctly-exchanged token. That
# bypasses the Copilot token exchange entirely and causes 400s with
# "not available for integrator copilot-language-server" (the server's
# fallback integrator when it receives a raw OAuth token instead of
# an api token). Skip the generic loop here — the copilot-specific
# branch is authoritative.
if provider == "copilot":
return False, active_sources
# Prefer ~/.hermes/.env over os.environ — the user's config file is the
# authoritative source for Hermes credentials. Stale env vars from parent
# processes (Codex CLI, test scripts, etc.) should not override deliberate
+49 -8
View File
@@ -541,6 +541,33 @@ def _restore_cron_skill_links(snapshot_dir: Path) -> Dict[str, Any]:
def _unstage(moved: List[Tuple[Path, Path]]) -> List[str]:
"""Move staged entries back to their original paths.
``shutil.move`` moves *into* an existing destination directory rather than
replacing it, so a partially-completed extract leaves debris that would
otherwise bury the user's real skill one level deeper
(``skills/foo/foo/``) while the tree still looks populated. Clear whatever
the failed extract created at each original path first. The staged copy is
authoritative, and the pre-rollback safety snapshot is the undo handle for
the extract's own output.
Returns the names that could not be restored, so the caller can report an
incomplete recovery instead of claiming the state was restored.
"""
failed: List[str] = []
for orig, dest in moved:
try:
if orig.is_dir() and not orig.is_symlink():
shutil.rmtree(orig)
elif orig.exists() or orig.is_symlink():
orig.unlink()
shutil.move(str(dest), str(orig))
except OSError:
failed.append(orig.name)
return failed
def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]]:
"""Restore ``~/.hermes/skills/`` from a snapshot.
@@ -609,11 +636,7 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
moved.append((entry, dest))
except OSError as e:
# Best-effort rollback of the move
for orig, dest in moved:
try:
shutil.move(str(dest), str(orig))
except OSError:
pass
_unstage(moved)
try:
shutil.rmtree(staged, ignore_errors=True)
except OSError:
@@ -638,12 +661,30 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
# Python < 3.12 — no filter kwarg
tf.extractall(str(skills))
except (OSError, tarfile.TarError) as e:
# Best-effort recover: move staged contents back
for orig, dest in moved:
# Best-effort recover. A partial extract can leave entries the
# original tree never had, so drop those first, otherwise the
# "restored" tree is the user's skills plus a slice of the snapshot.
staged_names = {orig.name for orig, _ in moved}
for entry in list(skills.iterdir()):
if entry.name in _EXCLUDE_TOP_LEVEL or entry.name in staged_names:
continue
try:
shutil.move(str(dest), str(orig))
if entry.is_dir() and not entry.is_symlink():
shutil.rmtree(entry)
else:
entry.unlink()
except OSError:
pass
unrestored = _unstage(moved)
if unrestored:
# Do not claim a clean restore we did not achieve, and keep the
# staging dir so the entries can be recovered by hand.
return (
False,
f"snapshot extract failed: {e} - could not restore "
f"{', '.join(sorted(unrestored))}; staged copies kept at {staged}",
None,
)
try:
shutil.rmtree(staged, ignore_errors=True)
except OSError:
+13 -3
View File
@@ -31,11 +31,21 @@ KANBAN_ENV_KEYS: tuple[str, ...] = (
@contextmanager
def delegated_child_context() -> Iterator[None]:
"""Mark the current execution context as a delegate_task child."""
def delegated_child_context(session_id: str | None = None) -> Iterator[None]:
"""Mark child execution and isolate its task-local session identity.
Child construction calls ``set_current_session_id`` internally, so even a
context entered without an id must restore the parent's ContextVar. Child
execution passes its explicit id and receives it only for this scope.
"""
token = _DELEGATED_CHILD_CONTEXT.set(True)
try:
yield
# Import lazily: session_context calls is_delegated_child_context() when
# deciding whether the compatibility os.environ mirror is safe.
from gateway.session_context import scoped_current_session_id
with scoped_current_session_id(session_id):
yield
finally:
_DELEGATED_CHILD_CONTEXT.reset(token)
+13
View File
@@ -836,6 +836,19 @@ def classify_api_error(
if classified is not None:
return classified
# Local MoA streaming compatibility errors are adapter-shape bugs, not a
# provider outage. Falling back to another model would silently switch the
# user's selected MoA route to a single-model answer (#55933 follow-up).
if provider_lower == "moa" and (
"'types.SimpleNamespace' object is not iterable" in str(error)
or "'types.SimpleNamespace' object has no attribute 'index'" in str(error)
):
return _result(
FailoverReason.format_error,
retryable=False,
should_fallback=False,
)
# Local MoA config drift is deterministic: a persisted session can retain
# a preset name that was later renamed/deleted. Retrying the same lookup
# cannot recover and makes a clear config error look like an API outage.
+35
View File
@@ -0,0 +1,35 @@
"""Compatibility helper for explicit agent stop producers."""
from __future__ import annotations
import inspect
from typing import Any
def request_hard_interrupt(agent: Any, message: str | None = None) -> bool:
"""Request an explicit stop, falling back to the legacy interrupt ABI.
New agents expose ``hard_interrupt(message=None)``. Third-party agents and
old test doubles may only expose ``interrupt(message=None)``; keep those
usable without sending the newer ``hard_cancel=`` keyword they do not know.
Returns ``False`` only when neither callable is available.
"""
# Avoid treating a dynamic ``__getattr__`` proxy (notably an unspecced
# ``MagicMock`` or a third-party RPC facade) as if it genuinely implements
# the new ABI. Static lookup proves the attribute exists on the instance or
# its type before normal descriptor binding retrieves the callable.
try:
inspect.getattr_static(agent, "hard_interrupt")
except AttributeError:
interrupt = None
else:
interrupt = getattr(agent, "hard_interrupt", None)
if not callable(interrupt):
interrupt = getattr(agent, "interrupt", None)
if not callable(interrupt):
return False
if message is None:
interrupt()
else:
interrupt(message)
return True
+6 -2
View File
@@ -35,6 +35,7 @@ from pathlib import Path
from typing import Any, Dict, Optional
from hermes_cli._subprocess_compat import windows_hide_flags
from hermes_constants import find_node_executable
logger = logging.getLogger("agent.lsp.install")
@@ -249,9 +250,12 @@ def _install_npm(
peer deps that npm doesn't auto-pull (typescript-language-server
needs ``typescript`` next to it; intelephense ships standalone).
"""
npm = shutil.which("npm")
# Managed npm first: $HERMES_HOME/node is not on an arbitrary process's
# PATH, so a bare which() misses the Node that Hermes installed and
# reports "npm not on PATH" on a machine that has a perfectly good one.
npm = find_node_executable("npm")
if npm is None:
logger.info("[install] cannot install %s: npm not on PATH", pkg)
logger.info("[install] cannot install %s: no usable npm found", pkg)
return None
staging = hermes_lsp_bin_dir().parent # <HERMES_HOME>/lsp/
install_targets = [pkg] + list(extra_pkgs or [])
+162 -10
View File
@@ -13,6 +13,7 @@ import logging
import re
import threading
from concurrent.futures import ThreadPoolExecutor, wait as _futures_wait
from types import SimpleNamespace
from typing import Any
from agent.auxiliary_client import call_llm
@@ -382,6 +383,8 @@ def _merge_slot_extra_body(
def _maybe_apply_moa_cache_control(
messages: list[dict[str, Any]],
runtime: dict[str, Any],
*,
cache_disabled: bool | None = None,
) -> list[dict[str, Any]]:
"""Decorate an advisor or aggregator request with cache_control when its
route honors it.
@@ -395,17 +398,27 @@ def _maybe_apply_moa_cache_control(
Returns the messages unchanged on any resolution error or when the
policy says the route doesn't honor markers.
``cache_disabled`` (or the live config when omitted) is stamped onto the
policy stub so ``prompt_caching.cache_ttl: off`` is not bypassed by the
blank-agent pattern (#76085).
"""
try:
from types import SimpleNamespace
from agent.agent_runtime_helpers import anthropic_prompt_cache_policy
from agent.agent_runtime_helpers import (
anthropic_prompt_cache_policy,
blank_cache_policy_stub,
)
from agent.prompt_caching import apply_anthropic_cache_control
# Prefer an explicit kwarg, then a snapshot on the runtime dict
# (threaded from the live agent), else config via the stub factory.
if cache_disabled is None and "_cache_disabled" in runtime:
cache_disabled = runtime.get("_cache_disabled")
# The policy function reads agent.* only as fallbacks for kwargs we
# don't pass; provide a stub so the slot is judged purely on its own
# resolved runtime.
stub = SimpleNamespace(provider="", base_url="", api_mode="", model="")
# don't pass; blank_cache_policy_stub is the only sanctioned stub
# so _cache_disabled cannot be left off again (#76085).
stub = blank_cache_policy_stub(cache_disabled)
should_cache, native_layout = anthropic_prompt_cache_policy(
stub,
provider=runtime.get("provider") or "",
@@ -431,6 +444,7 @@ def _run_reference(
max_tokens: int | None = None,
reference_timeout: float | None = None,
context_length_cache: Any = None,
cache_disabled: bool | None = None,
) -> tuple[str, str, Any]:
"""Call one reference model and return ``(label, text, accounting)``.
@@ -492,7 +506,12 @@ def _run_reference(
# caching is opt-in per request. OpenAI-family advisors are untouched
# (their caching is automatic; markers are ignored harmlessly, but we
# only decorate when the policy says the route honors them).
messages = _maybe_apply_moa_cache_control(messages, runtime)
# Pin the live agent disable onto the runtime so advisor decoration
# tracks conversation state, not a fresh config re-read (#76085).
cache_runtime = runtime
if cache_disabled is not None:
cache_runtime = {**runtime, "_cache_disabled": cache_disabled}
messages = _maybe_apply_moa_cache_control(messages, cache_runtime)
# Per-slot max_tokens takes precedence over the preset-level
# reference_max_tokens passed in by the caller. This lets each
# reference model have its own output cap independently.
@@ -796,6 +815,9 @@ def _run_references_parallel(
# instead of re-probing metadata sources per reference (dict get/set is
# GIL-atomic; a rare duplicate probe on a first-use race is harmless).
_ctx_len_cache: dict[tuple[str, str], int | None] = {}
cache_disabled = (
getattr(agent, "_cache_disabled", None) if agent is not None else None
)
try:
for idx, slot in enumerate(reference_models):
if slot.get("provider") == "moa":
@@ -814,6 +836,7 @@ def _run_references_parallel(
max_tokens=max_tokens,
reference_timeout=reference_timeout,
context_length_cache=_ctx_len_cache,
cache_disabled=cache_disabled,
)
] = idx
@@ -1261,6 +1284,19 @@ def aggregate_moa_context(
agg_label = _slot_label(aggregator)
agg_runtime = _slot_runtime(aggregator)
# Pin the live agent disable onto synthesis decoration so mid-session
# config flips cannot re-enable markers on this path alone (#76085).
# Same not-None guard as _run_reference: stamping None would be a no-op
# (present-None falls through to the config fallback anyway).
agg_cache_runtime = agg_runtime
_agg_cache_disabled = (
getattr(agent, "_cache_disabled", None) if agent is not None else None
)
if _agg_cache_disabled is not None:
agg_cache_runtime = {
**agg_runtime,
"_cache_disabled": _agg_cache_disabled,
}
try:
# Same cache_control decoration as _run_reference's advisor calls
# (see _maybe_apply_moa_cache_control) — this synthesis call is a
@@ -1273,7 +1309,7 @@ def aggregate_moa_context(
# breakpoints, even when the resolved aggregator slot is a
# cache-honoring route (e.g. Claude on OpenRouter/native Anthropic).
agg_messages = _maybe_apply_moa_cache_control(
[{"role": "user", "content": synth_prompt}], agg_runtime
[{"role": "user", "content": synth_prompt}], agg_cache_runtime
)
response = call_llm(
task="moa_aggregator",
@@ -1300,6 +1336,53 @@ def aggregate_moa_context(
)
def _completed_response_as_stream_chunk(response: Any) -> Any:
"""Convert a completed Chat Completions response into one delta stream chunk.
MoA's outer streaming consumer expects ``choices[0].delta`` chunks. A
completed aggregator response carries ``choices[0].message`` instead; adapt
it here, at the MoA facade boundary, so provider-specific Relay behavior and
other transports remain untouched.
"""
choices = getattr(response, "choices", None)
first_choice = choices[0] if isinstance(choices, (list, tuple)) and choices else None
message = getattr(first_choice, "message", None)
raw_tool_calls = getattr(message, "tool_calls", None)
tool_call_deltas = None
if isinstance(raw_tool_calls, (list, tuple)) and raw_tool_calls:
tool_call_deltas = []
for index, tc in enumerate(raw_tool_calls):
function = getattr(tc, "function", None)
tool_call_deltas.append(SimpleNamespace(
index=getattr(tc, "index", index),
id=getattr(tc, "id", None),
type=getattr(tc, "type", None) or "function",
function=SimpleNamespace(
name=getattr(function, "name", None),
arguments=getattr(function, "arguments", None),
),
))
delta = SimpleNamespace(
content=getattr(message, "content", None),
tool_calls=tool_call_deltas,
reasoning_content=getattr(message, "reasoning_content", None),
reasoning=getattr(message, "reasoning", None),
reasoning_details=getattr(message, "reasoning_details", None),
)
choice = SimpleNamespace(
index=getattr(first_choice, "index", 0),
delta=delta,
finish_reason=getattr(first_choice, "finish_reason", None) or "stop",
)
return SimpleNamespace(
id=getattr(response, "id", None),
model=getattr(response, "model", None),
choices=[choice],
usage=getattr(response, "usage", None),
)
def _attach_reference_guidance(agg_messages: list[dict[str, Any]], guidance: str) -> None:
"""Attach the per-turn reference block at the END of the aggregator prompt.
@@ -1609,6 +1692,52 @@ class MoAChatCompletions:
max_tokens: Any = agg_kwargs.get("max_tokens")
tools: Any = agg_kwargs.get("tools")
extra_body: Any = agg_kwargs.get("extra_body")
agg_runtime = _slot_runtime(aggregator)
try:
from agent.agent_runtime_helpers import (
plan_cache_sections_for_destination,
)
guidance = prepared.get("guidance")
planning_messages = agg_messages
if guidance:
planning_messages = peel_reference_guidance(
agg_messages,
str(guidance),
)
# plan_cache_sections_for_destination never mutates its inputs
# and always returns request-local copies, so the prepared
# state stays canonical.
# Tri-state: only pass a bool when a live agent snapshot exists.
# Prepared-aggregator facades built via __new__ have no _agent;
# getattr(self._agent, ...) raises and bool(None-agent) would
# force False and suppress the planner's config fallback (#76085).
_agent = getattr(self, "_agent", None)
_cache_disabled = (
getattr(_agent, "_cache_disabled", None)
if _agent is not None
else None
)
agg_messages, tools = plan_cache_sections_for_destination(
planning_messages,
tools,
provider=agg_runtime.get("provider") or "",
base_url=agg_runtime.get("base_url") or "",
api_mode=agg_runtime.get("api_mode") or "",
model=agg_runtime.get("model") or "",
cache_disabled=_cache_disabled,
)
if guidance:
_attach_reference_guidance(agg_messages, str(guidance))
except Exception as exc: # pragma: no cover - cache planning must not block MoA
# Warning, not debug: since the call-block site skips MoA, this
# block is the aggregator's ONLY decoration path — a silent
# failure here ships an undecorated request and regresses the
# exact 0%-cache MoA failure the planning exists to prevent.
logger.warning(
"MoA aggregator cache plan failed — sending undecorated "
"request (cache misses expected): %s", exc,
)
# Record the exact aggregator INPUT (incl. the injected reference
# context) into the pending trace so a trace captures what the
# aggregator actually saw, not a reconstruction. Traces are a
@@ -1649,7 +1778,6 @@ class MoAChatCompletions:
# actually governs the aggregator stream, not just call_llm's default.
if api_kwargs.get("timeout") is not None:
stream_kwargs["timeout"] = api_kwargs["timeout"]
agg_runtime = _slot_runtime(aggregator)
# _slot_runtime may carry the provider's request_overrides.extra_body;
# pop it and merge with the caller's extra_body (caller wins) so the
# explicit kwarg below never collides with **agg_runtime.
@@ -1685,6 +1813,14 @@ class MoAChatCompletions:
self._pending_trace["aggregator_output"] = _extract_text(_agg_response)
except Exception: # pragma: no cover - defensive
self._pending_trace["aggregator_output"] = None
if stream and hasattr(_agg_response, "choices"):
# Some aggregator adapters (notably openai-codex Responses) consume
# their provider stream internally and return a completed response
# object even when the acting consumer requested token streaming.
# The outer chat-completions streaming loop expects delta chunks;
# hand it a one-chunk iterator instead of letting it iterate the
# SimpleNamespace response itself (#55933).
return iter((_completed_response_as_stream_chunk(_agg_response),))
return _agg_response
def create(self, **api_kwargs: Any) -> Any:
@@ -2174,8 +2310,24 @@ def build_moa_facade(agent, preset_name: Any = None) -> MoAClient:
except Exception:
pass
resolved_preset = preset_name
if resolved_preset is None and getattr(agent, "provider", None) == "moa":
resolved_preset = getattr(agent, "model", None)
resolved_preset = str(resolved_preset or "default")
try:
from hermes_cli.config import load_config
from hermes_cli.moa_config import normalize_moa_config
moa_cfg = normalize_moa_config(load_config().get("moa") or {})
presets = moa_cfg.get("presets") or {}
if resolved_preset not in presets:
resolved_preset = moa_cfg.get("default_preset") or "default"
except Exception:
resolved_preset = "default"
return MoAClient(
str(preset_name or getattr(agent, "model", None) or "default"),
resolved_preset,
reference_callback=_moa_reference_relay,
# Thread the agent through so the reference fan-out wait can be
# aborted on a user interrupt (see _run_references_parallel).
+129 -50
View File
@@ -273,6 +273,27 @@ CONTEXT_PROBE_TIERS = [
# Default context length when no detection method succeeds.
DEFAULT_FALLBACK_CONTEXT = CONTEXT_PROBE_TIERS[0]
# (model, base_url) pairs that already emitted the fallback warning.
# The fallback result itself is deliberately never cached, so without this
# the warning would repeat on every resolution for the same unknown model.
_FALLBACK_WARNED: set = set()
def _warn_context_length_fallback(model: str, base_url: str) -> None:
"""Warn (once per model+endpoint) that context detection failed and the
hard default is being used, so small-context models (8K, 32K) don't
silently get 256K and cause hard-to-debug API failures."""
key = (model, base_url or "")
if key in _FALLBACK_WARNED:
return
_FALLBACK_WARNED.add(key)
logger.warning(
"Could not determine context length for model %r (base_url=%s) "
"— falling back to %s tokens. Set model.context_length in "
"config.yaml to override.",
model, base_url or "default", f"{DEFAULT_FALLBACK_CONTEXT:,}",
)
# Minimum context length required to run Hermes Agent. Models with fewer
# tokens cannot maintain enough working memory for tool-calling workflows.
# Sessions, model switches, and cron jobs should reject models below this.
@@ -634,12 +655,17 @@ def _is_known_provider_base_url(base_url: str) -> bool:
def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]:
"""Return metadata confirmed only for the Kimi Coding endpoint.
"""Return context metadata confirmed for one provider endpoint.
Kimi Coding serves K3 under the bare slug ``k3``, but users may also
configure or select the public-facing aliases ``kimi-k3`` and
``kimi-k3-cot``. Only canonical ``https://api.kimi.com/coding`` endpoints
(legacy Moonshot keys do not serve K3) get the 1 Mi context window.
NVIDIA NIM serves ``deepseek-ai/deepseek-v4-pro`` with a 262,144-token
window even though DeepSeek's native endpoint serves the V4 family with a
1M window. Keep the lower limit scoped to NVIDIA instead of weakening the
global model-family metadata.
"""
normalized = _normalize_base_url(base_url)
try:
@@ -659,6 +685,18 @@ def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]:
and model.strip().lower() in {"k3", "kimi-k3", "kimi-k3-cot"}
):
return 1_048_576
if (
parsed.scheme.lower() == "https"
and (parsed.hostname or "").lower() == "integrate.api.nvidia.com"
and port in (None, 443)
and parsed.username is None
and parsed.password is None
and parsed.path.rstrip("/") == "/v1"
and not parsed.query
and not parsed.fragment
and model.strip().lower() == "deepseek-ai/deepseek-v4-pro"
):
return 262_144
return None
@@ -1046,7 +1084,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any
return cache
except Exception as e:
logger.warning(f"Failed to fetch model metadata from OpenRouter: {e}")
logger.warning("Failed to fetch model metadata from OpenRouter: %s", e)
if _model_metadata_cache:
return _model_metadata_cache
disk_cache = _load_model_metadata_disk_cache()
@@ -1147,8 +1185,22 @@ def fetch_endpoint_model_metadata(
for candidate in candidates:
url = candidate.rstrip("/") + "/models"
response = None
try:
response = requests.get(url, headers=headers, timeout=(5, 10), verify=_resolve_requests_verify())
response = requests.get(
url,
headers=headers,
timeout=(5, 10),
verify=_resolve_requests_verify(),
stream=True,
)
if response.status_code in (401, 403):
logger.debug(
"Model metadata probe received HTTP %s from %s; stopping candidate probing",
response.status_code,
url,
)
break
response.raise_for_status()
payload = response.json()
cache: Dict[str, Dict[str, Any]] = {}
@@ -1198,6 +1250,9 @@ def fetch_endpoint_model_metadata(
return cache
except Exception as exc:
last_error = exc
finally:
if response is not None:
response.close()
if last_error:
logger.debug("Failed to fetch model metadata from %s/models: %s", normalized, last_error)
@@ -2582,6 +2637,9 @@ def get_model_context_length(
f"{length:,}", model, default_model,
)
return length
# Same silent-256K bug class as the step-9 fallback below —
# warn here too so custom/local endpoints aren't left invisible.
_warn_context_length_fallback(model, base_url)
return DEFAULT_FALLBACK_CONTEXT
# 4. Anthropic /v1/models API (only for regular API keys, not OAuth)
@@ -2754,7 +2812,10 @@ def get_model_context_length(
if default_model in model_lower:
return length
# 9. Default fallback — 256K
# 9. Default fallback — warn (deduped per model+endpoint) so
# small-context models don't silently get 256K. See
# _warn_context_length_fallback for rationale.
_warn_context_length_fallback(model, base_url)
return DEFAULT_FALLBACK_CONTEXT
@@ -2965,6 +3026,68 @@ def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
return count * cost_per_image
def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
"""Shadow of a message holding only what the provider actually receives.
Two adjustments to the raw persisted dict:
* ``api_content`` is a SUBSTITUTE for ``content``, not an addition to it.
``turn_context.substitute_api_content()`` pops the sidecar and overwrites
``content`` at every API-bound build site, so exactly one of the two is
ever sent. Counting both double-counts any message whose sidecar differs
from its clean stored content (2.00x on a 40KB sidecar).
The substitution mirrors that helper's guard exactly: only a non-empty
STRING sidecar on a ``user``/``assistant`` row displaces ``content``.
Any other sidecar shape is popped and discarded on the wire without
touching ``content``, so a shadow that substituted unconditionally
would UNDERcount those rows — the dangerous direction, since it makes
compaction fire too late and the turn dies on a hard context error.
* Base64 image payloads are replaced with a placeholder; they are charged
separately at a flat rate by ``_count_image_tokens``, and counting their
raw chars here would massively overestimate usage.
"""
sidecar = msg.get("api_content")
sidecar_wins = (
isinstance(sidecar, str)
and bool(sidecar)
and msg.get("role") in ("user", "assistant")
)
shadow: Dict[str, Any] = {}
for k, v in msg.items():
if k in ("_anthropic_content_blocks", "reasoning_details"):
continue
if k == "api_content":
# Always popped before the request is built; only counted when it
# actually replaces ``content``.
if sidecar_wins:
shadow["content"] = v
continue
if k == "content":
if sidecar_wins:
# The sidecar wins on the wire; skip the clean copy so the
# same logical content is not counted twice.
continue
if isinstance(v, list):
cleaned = []
for part in v:
if isinstance(part, dict):
if part.get("type") in {"image", "image_url", "input_image"}:
cleaned.append({"type": part.get("type"), "image": "[stripped]"})
else:
cleaned.append(part)
else:
cleaned.append(part)
shadow[k] = cleaned
elif isinstance(v, dict) and v.get("_multimodal"):
shadow[k] = v.get("text_summary", "")
else:
shadow[k] = v
else:
shadow[k] = v
return shadow
def _estimate_message_chars(msg: Dict[str, Any]) -> int:
"""Char count for token estimation, excluding base64 image data.
@@ -2973,58 +3096,14 @@ def _estimate_message_chars(msg: Dict[str, Any]) -> int:
"""
if not isinstance(msg, dict):
return len(str(msg))
shadow: Dict[str, Any] = {}
for k, v in msg.items():
if k == "_anthropic_content_blocks":
continue
if k == "content":
if isinstance(v, list):
cleaned = []
for part in v:
if isinstance(part, dict):
if part.get("type") in {"image", "image_url", "input_image"}:
cleaned.append({"type": part.get("type"), "image": "[stripped]"})
else:
cleaned.append(part)
else:
cleaned.append(part)
shadow[k] = cleaned
elif isinstance(v, dict) and v.get("_multimodal"):
shadow[k] = v.get("text_summary", "")
else:
shadow[k] = v
else:
shadow[k] = v
return len(str(shadow))
return len(str(_wire_message_shadow(msg)))
def _estimate_message_tokens_without_images(msg: Dict[str, Any]) -> int:
"""Token estimate for a message shadow with image payloads stripped."""
if not isinstance(msg, dict):
return estimate_tokens_rough(str(msg))
shadow: Dict[str, Any] = {}
for k, v in msg.items():
if k == "_anthropic_content_blocks":
continue
if k == "content":
if isinstance(v, list):
cleaned = []
for part in v:
if isinstance(part, dict):
if part.get("type") in {"image", "image_url", "input_image"}:
cleaned.append({"type": part.get("type"), "image": "[stripped]"})
else:
cleaned.append(part)
else:
cleaned.append(part)
shadow[k] = cleaned
elif isinstance(v, dict) and v.get("_multimodal"):
shadow[k] = v.get("text_summary", "")
else:
shadow[k] = v
else:
shadow[k] = v
return estimate_tokens_rough(str(shadow))
return estimate_tokens_rough(str(_wire_message_shadow(msg)))
def estimate_request_tokens_rough(
+569
View File
@@ -0,0 +1,569 @@
"""
Outbound webhook notifications.
Reads the ``hooks.outbound:`` list from ``config.yaml`` and registers
notify-only callbacks on the existing plugin hook manager, so every
``invoke_hook()`` site can push lifecycle events to external HTTP
endpoints — CI systems, dashboards, other agents — with zero changes to
call sites and zero polling on the receiving end.
This is the outbound mirror of the inbound webhook platform
(``gateway/platforms/webhook.py``): inbound wakes Hermes when the world
changes; outbound tells the world when Hermes does something.
Design notes
------------
* Delivery is fire-and-forget through a bounded in-process queue and a
single daemon worker thread. ``invoke_hook()`` runs inside the agent
loop, so callbacks must never block on network I/O — they serialize,
enqueue, and return ``None`` immediately. Outbound targets can never
block a tool call, inject context, or otherwise influence agent flow.
* Payloads are signed with HMAC-SHA256 (GitHub-style
``X-Hermes-Signature-256: sha256=<hexdigest>`` over the raw body) when
a secret is configured. Receivers verify exactly like they verify
GitHub webhooks.
* No consent prompt: unlike shell hooks, an outbound target executes no
code on this machine — it POSTs JSON to a URL the user themselves put
in config. ``HERMES_SAFE_MODE=1`` still skips registration, matching
plugins / MCP / shell hooks.
* Registration is idempotent — safe to invoke from both the CLI entry
point and the gateway entry point.
Config schema (``~/.hermes/config.yaml``)::
hooks:
outbound:
- url: https://ci.example.com/hermes-events
events: [on_session_end, subagent_stop]
# secret literal (discouraged) or env var name (preferred):
secret_env: HERMES_OUTBOUND_WEBHOOK_SECRET
# optional regex, honored for pre/post_tool_call only:
matcher: "terminal|delegate_task"
timeout: 10 # per-attempt seconds, clamped to [1, 60]
name: ci-notify # optional label for logs / `hermes hooks list`
Wire format (POST body)::
{
"hook_event_name": "on_session_end",
"tool_name": null,
"tool_input": null,
"session_id": "sess_abc123",
"cwd": "/home/user/project",
"extra": {...}, # event-specific kwargs
"delivery_id": "3f2c...", # uuid4, unique per POST
"timestamp": "2026-07-22T14:00:00Z"
}
Headers::
Content-Type: application/json
User-Agent: Hermes-Agent-Outbound-Webhook
X-Hermes-Event: <hook event name>
X-Hermes-Delivery: <delivery_id>
X-Hermes-Signature-256: sha256=<hmac hexdigest> # only when secret set
"""
from __future__ import annotations
import atexit
import hashlib
import hmac
import json
import logging
import os
import queue
import re
import threading
import time
import uuid
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional, Set, Tuple
from urllib import error as urlerror
from urllib import request as urlrequest
logger = logging.getLogger(__name__)
DEFAULT_TIMEOUT_SECONDS = 10
MAX_TIMEOUT_SECONDS = 60
MAX_DELIVERY_ATTEMPTS = 2
RETRY_BACKOFF_SECONDS = 1.0
QUEUE_MAX_SIZE = 256
# Events whose ``matcher`` field is honored (mirrors shell hooks).
_TOOL_SCOPED_EVENTS = {"pre_tool_call", "post_tool_call"}
# kwargs promoted to top-level payload keys (mirrors shell hooks wire).
_TOP_LEVEL_PAYLOAD_KEYS = {"tool_name", "args", "session_id", "parent_session_id"}
# (event, url) pairs already wired to the plugin manager in this process.
_registered: Set[Tuple[str, str]] = set()
_registered_lock = threading.Lock()
_delivery_queue: "queue.Queue[Optional[Dict[str, Any]]]" = queue.Queue(
maxsize=QUEUE_MAX_SIZE
)
_worker_lock = threading.Lock()
_worker: Optional[threading.Thread] = None
@dataclass
class WebhookTarget:
"""Parsed and validated representation of one ``hooks.outbound`` entry."""
url: str
events: List[str]
name: str = ""
secret: Optional[str] = None
matcher: Optional[str] = None
timeout: int = DEFAULT_TIMEOUT_SECONDS
compiled_matcher: Optional[re.Pattern] = field(default=None, repr=False)
def __post_init__(self) -> None:
if isinstance(self.matcher, str):
stripped = self.matcher.strip()
self.matcher = stripped if stripped else None
if self.matcher:
try:
self.compiled_matcher = re.compile(self.matcher)
except re.error as exc:
logger.warning(
"outbound webhook matcher %r is invalid (%s) — treating "
"as literal equality", self.matcher, exc,
)
self.compiled_matcher = None
@property
def label(self) -> str:
return self.name or self.url
def matches_tool(self, tool_name: Optional[str]) -> bool:
if not self.matcher:
return True
if tool_name is None:
return False
if self.compiled_matcher is not None:
return self.compiled_matcher.fullmatch(tool_name) is not None
return tool_name == self.matcher
# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------
def register_from_config(cfg: Optional[Dict[str, Any]]) -> List[WebhookTarget]:
"""Register every configured outbound webhook on the plugin manager.
``cfg`` is the full parsed config dict. Missing, empty, or malformed
``hooks.outbound`` is treated as zero targets — config parsing never
raises, because a broken webhook entry must not crash the agent.
Returns the targets that ended up wired (deduplicated across repeat
calls, so the CLI and gateway can both invoke this safely).
"""
if not isinstance(cfg, dict):
return []
from utils import env_var_enabled
if env_var_enabled("HERMES_SAFE_MODE"):
logger.info("HERMES_SAFE_MODE=1 — outbound webhook registration skipped")
return []
hooks_cfg = cfg.get("hooks")
targets = _parse_outbound_block(
hooks_cfg.get("outbound") if isinstance(hooks_cfg, dict) else None
)
if not targets:
return []
from hermes_cli.plugins import get_plugin_manager
manager = get_plugin_manager()
registered: List[WebhookTarget] = []
with _registered_lock:
for target in targets:
wired_any = False
for event in target.events:
key = (event, target.url)
if key in _registered:
continue
manager._hooks.setdefault(event, []).append(
_make_callback(event, target)
)
_registered.add(key)
wired_any = True
logger.info(
"outbound webhook registered: %s -> %s (matcher=%s, "
"timeout=%ds)",
event, target.label, target.matcher, target.timeout,
)
if wired_any:
registered.append(target)
return registered
def iter_configured_targets(cfg: Optional[Dict[str, Any]]) -> List[WebhookTarget]:
"""Parse ``hooks.outbound`` without registering anything.
Used by ``hermes hooks list``."""
if not isinstance(cfg, dict):
return []
hooks_cfg = cfg.get("hooks")
return _parse_outbound_block(
hooks_cfg.get("outbound") if isinstance(hooks_cfg, dict) else None
)
def flush(timeout: float = 5.0) -> bool:
"""Block until all queued deliveries are done (or *timeout* elapses).
Returns ``True`` when the queue fully drained. Test/shutdown helper."""
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
with _delivery_queue.all_tasks_done:
if _delivery_queue.unfinished_tasks == 0:
return True
time.sleep(0.02)
with _delivery_queue.all_tasks_done:
return _delivery_queue.unfinished_tasks == 0
def reset_for_tests() -> None:
"""Clear the idempotence set and drain the queue. Test-only helper."""
with _registered_lock:
_registered.clear()
try:
while True:
_delivery_queue.get_nowait()
_delivery_queue.task_done()
except queue.Empty:
pass
# ---------------------------------------------------------------------------
# Config parsing
# ---------------------------------------------------------------------------
def _parse_outbound_block(raw: Any) -> List[WebhookTarget]:
if raw is None:
return []
if not isinstance(raw, list):
logger.warning(
"hooks.outbound must be a list of webhook targets; got %s",
type(raw).__name__,
)
return []
targets: List[WebhookTarget] = []
for i, entry in enumerate(raw):
target = _parse_single_target(i, entry)
if target is not None:
targets.append(target)
return targets
def _parse_single_target(index: int, raw: Any) -> Optional[WebhookTarget]:
from hermes_cli.plugins import VALID_HOOKS
if not isinstance(raw, dict):
logger.warning(
"hooks.outbound[%d] must be a mapping with 'url' and 'events' "
"keys; got %s", index, type(raw).__name__,
)
return None
url = raw.get("url")
if not isinstance(url, str) or not url.strip():
logger.warning("hooks.outbound[%d] is missing a non-empty 'url'", index)
return None
url = url.strip()
if not url.lower().startswith(("http://", "https://")):
logger.warning(
"hooks.outbound[%d].url must be http(s); got %r — skipped",
index, url,
)
return None
if url.lower().startswith("http://"):
logger.warning(
"hooks.outbound[%d].url uses plain http:// — payloads (including "
"tool inputs) travel unencrypted. Prefer https.", index,
)
events_raw = raw.get("events")
if not isinstance(events_raw, list) or not events_raw:
logger.warning(
"hooks.outbound[%d] needs a non-empty 'events' list (valid: %s)",
index, ", ".join(sorted(VALID_HOOKS)),
)
return None
events: List[str] = []
for ev in events_raw:
if ev in VALID_HOOKS:
events.append(ev)
else:
logger.warning(
"hooks.outbound[%d]: unknown event %r ignored (valid: %s)",
index, ev, ", ".join(sorted(VALID_HOOKS)),
)
if not events:
logger.warning(
"hooks.outbound[%d] has no valid events — skipped", index,
)
return None
matcher = raw.get("matcher")
if matcher is not None and not isinstance(matcher, str):
logger.warning(
"hooks.outbound[%d].matcher must be a string regex; ignoring",
index,
)
matcher = None
if matcher is not None and not any(e in _TOOL_SCOPED_EVENTS for e in events):
logger.warning(
"hooks.outbound[%d].matcher=%r will be ignored — matcher is only "
"honored for pre_tool_call / post_tool_call.", index, matcher,
)
matcher = None
timeout_raw = raw.get("timeout", DEFAULT_TIMEOUT_SECONDS)
try:
timeout = int(timeout_raw)
except (TypeError, ValueError):
logger.warning(
"hooks.outbound[%d].timeout must be an int (got %r); using "
"default %ds", index, timeout_raw, DEFAULT_TIMEOUT_SECONDS,
)
timeout = DEFAULT_TIMEOUT_SECONDS
timeout = max(1, min(timeout, MAX_TIMEOUT_SECONDS))
secret = _resolve_secret(index, raw)
name = raw.get("name")
if not isinstance(name, str):
name = ""
return WebhookTarget(
url=url,
events=events,
name=name.strip(),
secret=secret,
matcher=matcher,
timeout=timeout,
)
def _resolve_secret(index: int, raw: Dict[str, Any]) -> Optional[str]:
"""``secret_env`` (env var name, preferred) wins over inline ``secret``."""
secret_env = raw.get("secret_env")
if isinstance(secret_env, str) and secret_env.strip():
value = os.environ.get(secret_env.strip(), "")
if value:
return value
logger.warning(
"hooks.outbound[%d].secret_env=%r is not set in the environment "
"— deliveries will be UNSIGNED", index, secret_env.strip(),
)
return None
secret = raw.get("secret")
if isinstance(secret, str) and secret:
return secret
return None
# ---------------------------------------------------------------------------
# Callback + delivery
# ---------------------------------------------------------------------------
def _make_callback(event: str, target: WebhookTarget):
"""Build the notify-only closure ``invoke_hook()`` calls per firing."""
def _callback(**kwargs: Any) -> None:
if event in _TOOL_SCOPED_EVENTS:
if not target.matches_tool(kwargs.get("tool_name")):
return None
delivery_id = uuid.uuid4().hex
try:
body = _serialize_payload(event, kwargs, delivery_id)
except Exception: # defensive — a bad payload must not hurt the loop
logger.warning(
"outbound webhook payload serialization failed (event=%s "
"target=%s)", event, target.label, exc_info=True,
)
return None
_enqueue(_build_delivery(event, target, body, delivery_id))
return None
_callback.__name__ = f"outbound_webhook[{event}:{target.label}]"
_callback.__qualname__ = _callback.__name__
return _callback
def _serialize_payload(
event: str, kwargs: Dict[str, Any], delivery_id: str,
) -> bytes:
"""Render the POST body. Same top-level shape as shell hooks' stdin
(documented in :mod:`agent.shell_hooks`), plus delivery metadata.
``delivery_id`` is shared with the ``X-Hermes-Delivery`` header so
receivers can dedupe on either — and since it (plus ``timestamp``)
lives inside the HMAC-signed body, it doubles as replay protection.
"""
extras = {k: v for k, v in kwargs.items() if k not in _TOP_LEVEL_PAYLOAD_KEYS}
try:
cwd = str(Path.cwd())
except OSError:
cwd = ""
payload = {
"hook_event_name": event,
"tool_name": kwargs.get("tool_name"),
"tool_input": kwargs.get("args") if isinstance(kwargs.get("args"), dict) else None,
"session_id": kwargs.get("session_id") or kwargs.get("parent_session_id") or "",
"cwd": cwd,
"extra": extras,
"delivery_id": delivery_id,
"timestamp": datetime.now(tz=timezone.utc)
.isoformat()
.replace("+00:00", "Z"),
}
return json.dumps(payload, ensure_ascii=False, default=str).encode("utf-8")
def _build_delivery(
event: str, target: WebhookTarget, body: bytes, delivery_id: str,
) -> Dict[str, Any]:
headers = {
"Content-Type": "application/json",
"User-Agent": "Hermes-Agent-Outbound-Webhook",
"X-Hermes-Event": event,
"X-Hermes-Delivery": delivery_id,
}
if target.secret:
digest = hmac.new(
target.secret.encode("utf-8"), body, hashlib.sha256
).hexdigest()
headers["X-Hermes-Signature-256"] = f"sha256={digest}"
return {
"url": target.url,
"label": target.label,
"event": event,
"body": body,
"headers": headers,
"timeout": target.timeout,
}
def _enqueue(delivery: Dict[str, Any]) -> None:
_ensure_worker()
try:
_delivery_queue.put_nowait(delivery)
except queue.Full:
logger.warning(
"outbound webhook queue full (%d pending) — dropping %s event "
"for %s", QUEUE_MAX_SIZE, delivery["event"], delivery["label"],
)
def _ensure_worker() -> None:
global _worker
if _worker is not None and _worker.is_alive():
return
with _worker_lock:
if _worker is not None and _worker.is_alive():
return
_worker = threading.Thread(
target=_worker_loop, name="outbound-webhooks", daemon=True,
)
_worker.start()
# The worker is a daemon thread, so a short-lived process (a `-q`
# CLI run, a cron session) can exit right after enqueuing the
# final events — silently dropping on_session_end, the headline
# use case. Drain the queue at interpreter shutdown, bounded so
# a dead endpoint can only delay exit, never hang it.
atexit.register(flush, timeout=5.0)
def _worker_loop() -> None:
while True:
delivery = _delivery_queue.get()
try:
if delivery is not None:
_deliver(delivery)
except Exception: # pragma: no cover — defensive
logger.warning(
"outbound webhook delivery crashed (target=%s)",
delivery.get("label") if isinstance(delivery, dict) else "?",
exc_info=True,
)
finally:
_delivery_queue.task_done()
class _NoRedirectHandler(urlrequest.HTTPRedirectHandler):
"""Refuse to follow redirects.
urllib's default handler converts a redirected POST into a body-less
GET — the signed payload would be silently dropped and the headers
re-sent to a location the user never configured. Treat any 3xx as a
delivery failure instead (surfaced as HTTPError by returning None).
"""
def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D102
return None
_opener = urlrequest.build_opener(_NoRedirectHandler)
def _deliver(delivery: Dict[str, Any]) -> None:
"""POST with bounded retries. Retries on connection errors and 5xx;
4xx is the receiver telling us the request itself is wrong — no retry.
3xx redirects are never followed (misconfiguration — fix the URL)."""
last_error = ""
for attempt in range(1, MAX_DELIVERY_ATTEMPTS + 1):
req = urlrequest.Request(
delivery["url"],
data=delivery["body"],
headers=delivery["headers"],
method="POST",
)
try:
with _opener.open(req, timeout=delivery["timeout"]) as resp:
status = getattr(resp, "status", 200)
if 200 <= status < 300:
logger.debug(
"outbound webhook delivered: %s -> %s (HTTP %d)",
delivery["event"], delivery["label"], status,
)
return
last_error = f"HTTP {status}"
except urlerror.HTTPError as exc:
last_error = f"HTTP {exc.code}"
if 300 <= exc.code < 400:
logger.warning(
"outbound webhook target redirected (event=%s target=%s): "
"%s -> %s — redirects are not followed; update the "
"configured url", delivery["event"], delivery["label"],
last_error, exc.headers.get("Location", "?"),
)
return
if 400 <= exc.code < 500:
logger.warning(
"outbound webhook rejected (event=%s target=%s): %s — "
"not retrying", delivery["event"], delivery["label"],
last_error,
)
return
except Exception as exc:
last_error = str(exc) or type(exc).__name__
if attempt < MAX_DELIVERY_ATTEMPTS:
time.sleep(RETRY_BACKOFF_SECONDS * attempt)
logger.warning(
"outbound webhook delivery failed after %d attempt(s) (event=%s "
"target=%s): %s",
MAX_DELIVERY_ATTEMPTS, delivery["event"], delivery["label"], last_error,
)
+1
View File
@@ -1083,6 +1083,7 @@ def _probe_remote_backend(env_type: str) -> str | None:
"docker_env": config.get("docker_env", {}),
"docker_run_as_host_user": config.get("docker_run_as_host_user", False),
"docker_extra_args": config.get("docker_extra_args", []),
"docker_shm_size": config.get("docker_shm_size", "1g"),
"docker_persist_across_processes": config.get("docker_persist_across_processes", True),
"docker_orphan_reaper": config.get("docker_orphan_reaper", True),
}
+176 -3
View File
@@ -11,9 +11,27 @@ Pure functions -- no class state, no AIAgent dependency.
"""
import copy
from dataclasses import dataclass
from typing import Any, Dict, List
@dataclass(frozen=True)
class PromptCachePlan:
"""Request-local message and tool sections with their cache markers."""
messages: List[Dict[str, Any]]
tools: List[Dict[str, Any]]
@property
def marker_count(self) -> int:
"""Wire-visible cache markers in this plan (computed on demand).
Only tests consume this; keeping it lazy avoids walking every
message part and tool schema on the per-request hot path.
"""
return _count_cache_markers(self.messages, self.tools)
def _apply_cache_marker(msg: dict, cache_marker: dict, native_anthropic: bool = False) -> None:
"""Add cache_control to a single message, handling all format variations."""
role = msg.get("role", "")
@@ -89,12 +107,28 @@ def _apply_system_cache_markers(
static_system_prefix: str | None,
*,
native_anthropic: bool,
mark_suffix: bool = True,
fallback_to_whole: bool = True,
) -> int:
"""Mark the static system prefix and full prompt when they can be split.
"""Mark the static system prefix (and optionally the full prompt).
The system prompt remains one stored string. Splitting it only in the
outgoing request keeps session persistence and non-Anthropic transports
unchanged while making the stable prefix independently cacheable.
``mark_suffix=False`` is the tool-cache-plan layout: only the static
prefix carries a marker, the volatile suffix rides unmarked (its
breakpoint budget is spent on the tools array instead).
``fallback_to_whole=False`` skips marking entirely when the prefix
split is not possible (no prefix, mismatched prefix, non-string
content) instead of marking the whole message.
When the prompt IS exactly the static prefix (empty suffix), the whole
message is marked as a single block — never a two-part split with an
empty text block, which Anthropic rejects.
Returns the number of markers applied (0, 1, or 2).
"""
content = message.get("content")
if (
@@ -105,16 +139,26 @@ def _apply_system_cache_markers(
):
suffix = content[len(static_system_prefix):]
if suffix:
suffix_part: dict = {"type": "text", "text": suffix}
if mark_suffix:
suffix_part["cache_control"] = cache_marker
message["content"] = [
{
"type": "text",
"text": static_system_prefix,
"cache_control": cache_marker,
},
{"type": "text", "text": suffix, "cache_control": cache_marker},
suffix_part,
]
return 2
return 2 if mark_suffix else 1
# Empty suffix: the stored prompt IS the static prefix. Mark it as
# one whole block — a [marked-prefix, ""] split would put an empty
# text block on the wire (HTTP 400 on native Anthropic).
_apply_cache_marker(message, cache_marker, native_anthropic=native_anthropic)
return 1
if not fallback_to_whole:
return 0
_apply_cache_marker(message, cache_marker, native_anthropic=native_anthropic)
return 1
@@ -172,6 +216,135 @@ def strip_anthropic_cache_control(
return api_messages
def strip_anthropic_tool_cache_control(tools: List[Dict[str, Any]] | None) -> List[Dict[str, Any]]:
"""Return copied tools without request-local Anthropic cache markers."""
cleaned = copy.deepcopy(tools or [])
for tool in cleaned:
if isinstance(tool, dict):
tool.pop("cache_control", None)
return cleaned
def _count_cache_markers(messages: List[Dict[str, Any]], tools: List[Dict[str, Any]]) -> int:
"""Count the wire-visible cache markers in a request-local plan."""
count = sum(
1
for message in messages
if isinstance(message, dict) and "cache_control" in message
)
count += sum(
1
for message in messages
if isinstance(message, dict) and isinstance(message.get("content"), list)
for part in message["content"]
if isinstance(part, dict) and "cache_control" in part
)
return count + sum(
1 for tool in tools if isinstance(tool, dict) and "cache_control" in tool
)
def _completed_transaction_endpoint_indexes(
messages: List[Dict[str, Any]], *, native_anthropic: bool,
) -> List[int]:
"""Select legal ends of completed tool runs and ordinary turns."""
endpoints: List[int] = []
index = 0
while index < len(messages):
message = messages[index]
if not isinstance(message, dict) or message.get("role") == "system":
index += 1
continue
if message.get("role") == "assistant" and message.get("tool_calls"):
result_start = index + 1
result_end = result_start
while result_end < len(messages):
result = messages[result_end]
if not isinstance(result, dict) or result.get("role") != "tool":
break
result_end += 1
if result_end > result_start:
endpoint = result_end - 1
if _can_carry_marker(messages[endpoint], native_anthropic):
endpoints.append(endpoint)
index = result_end
continue
if message.get("role") == "tool":
while index < len(messages):
result = messages[index]
if not isinstance(result, dict) or result.get("role") != "tool":
break
index += 1
continue
if message.get("role") == "user" and index + 1 < len(messages):
index += 1
continue
if (
message.get("role") == "assistant"
and message.get("content") in (None, "")
):
index += 1
continue
if _can_carry_marker(message, native_anthropic):
endpoints.append(index)
index += 1
return endpoints
def build_prompt_cache_plan(
api_messages: List[Dict[str, Any]],
tools: List[Dict[str, Any]] | None,
*,
cache_ttl: str = "5m",
native_anthropic: bool = False,
static_system_prefix: str | None = None,
direct_native_tool_cache: bool = False,
) -> PromptCachePlan:
"""Build isolated cache sections for one resolved request destination."""
messages = copy.deepcopy(api_messages or [])
strip_anthropic_cache_control(messages)
planned_tools = strip_anthropic_tool_cache_control(tools)
if not direct_native_tool_cache or not planned_tools:
planned_messages = apply_anthropic_cache_control(
messages,
cache_ttl=cache_ttl,
native_anthropic=native_anthropic,
static_system_prefix=static_system_prefix,
)
return PromptCachePlan(messages=planned_messages, tools=planned_tools)
marker = _build_marker(cache_ttl)
if (
messages
and isinstance(messages[0], dict)
and messages[0].get("role") == "system"
):
# Tool-cache layout: only the static prefix carries a system-side
# marker; the volatile suffix's budget is spent on the tools array.
_apply_system_cache_markers(
messages[0],
marker,
static_system_prefix,
native_anthropic=True,
mark_suffix=False,
fallback_to_whole=False,
)
planned_tools[-1]["cache_control"] = dict(marker)
for endpoint in _completed_transaction_endpoint_indexes(
messages,
native_anthropic=True,
)[-2:]:
_apply_cache_marker(messages[endpoint], marker, native_anthropic=True)
return PromptCachePlan(messages=messages, tools=planned_tools)
def apply_anthropic_cache_control(
api_messages: List[Dict[str, Any]],
cache_ttl: str = "5m",
+26 -3
View File
@@ -111,6 +111,23 @@ _PREFIX_PATTERNS = [
r"fw-[A-Za-z0-9]{30,}", # Fireworks AI API key
r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key
r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key
# GitLab token families (each pattern keeps a full literal prefix so the
# _PREFIX_SUBSTRINGS pre-screen stays false-negative-free). Ported from
# openclaw/openclaw#112954; follow-up invited in #4541.
r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token
r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret
r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token
r"glrt-[A-Za-z0-9_.\-]{10,}", # GitLab runner authentication token (routable tokens are dotted)
r"glrtr-[A-Za-z0-9_.\-]{10,}", # GitLab runner registration token (routable)
r"glcbt-[A-Za-z0-9_\-]{10,}", # GitLab CI/CD job token
r"glptt-[A-Za-z0-9_\-]{10,}", # GitLab pipeline trigger token
r"glft-[A-Za-z0-9_\-]{10,}", # GitLab feed token
r"glimt-[A-Za-z0-9_\-]{10,}", # GitLab incoming mail token
r"glagent-[A-Za-z0-9_\-]{10,}", # GitLab agent (KAS) token
r"glsoat-[A-Za-z0-9_\-]{10,}", # GitLab service-account access token
r"glffct-[A-Za-z0-9_\-]{10,}", # GitLab feature-flags client token
r"glwt-[A-Za-z0-9_\-]{10,}", # GitLab workspace token
r"GR1348941[A-Za-z0-9_\-]{10,}", # GitLab legacy runner registration token
]
# ENV assignment patterns: KEY=value where KEY contains a secret-like name.
@@ -154,9 +171,13 @@ _ENV_LOOKUP_VALUE_RE = re.compile(
r"^(?:os\.(?:getenv|environ)|process\.env|\$ENV\{)"
)
# Namespaced (dotted) key: the secret word may sit anywhere in a dotted path.
# NOTE(perf): possessive quantifiers (py3.11+) replace the nested quantifier
# ``(?:[A-Za-z0-9_\-]+\.)+`` (exponential backtracking on long dotted runs).
# The ``*`` runs bordering {_SECRET_CFG_NAMES} must stay backtrackable
# (secret words are matchable by the class, e.g. ``app.api.key=…``).
_CFG_DOTTED_RE = re.compile(
rf"((?:[A-Za-z0-9_\-]+\.)+[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*"
rf"|[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*\.[A-Za-z0-9_.\-]+)"
rf"([A-Za-z0-9_\-]++\.[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*+"
rf"|[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*\.[A-Za-z0-9_.\-]++)"
rf"={_CFG_VALUE}",
re.IGNORECASE,
)
@@ -175,8 +196,10 @@ _CFG_ANCHORED_RE = re.compile(
# is masked by _AUTH_HEADER_RE); ``auth_token``/``auth-token`` still match via
# the ``token`` keyword. Quoted values defer to _JSON_FIELD_RE via the lookahead.
_YAML_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential)"
# NOTE(perf): possessive quantifiers wherever the successor is disjoint; the
# leading ``[A-Za-z0-9_.\-]*`` stays backtrackable (see _CFG_DOTTED_RE note).
_YAML_ASSIGN_RE = re.compile(
rf"(^[ \t]*[A-Za-z0-9_.\-]*{_YAML_CFG_NAMES}[A-Za-z0-9_.\-]*)(:[ \t]*)(?!['\"])([^\s&]+)",
rf"(^[ \t]*+[A-Za-z0-9_.\-]*{_YAML_CFG_NAMES}[A-Za-z0-9_.\-]*+)(:[ \t]*+)(?!['\"])([^\s&]++)",
re.IGNORECASE | re.MULTILINE,
)
+42 -2
View File
@@ -247,6 +247,14 @@ async def execute_current_async(
)
def _has_running_event_loop() -> bool:
try:
asyncio.get_running_loop()
except RuntimeError:
return False
return True
def stream_current(
request: dict[str, Any],
stream_factory: Callable[[dict[str, Any]], Any],
@@ -256,12 +264,34 @@ def stream_current(
finalizer: Callable[[], Any],
metadata: dict[str, Any] | None = None,
defer_logical_completion: bool = False,
completed_response_predicate: Callable[[Any], bool] | None = None,
) -> Any:
"""Run a provider stream under the inherited Hermes turn when present."""
"""Run a provider stream under the inherited Hermes turn when present.
When ``completed_response_predicate`` is set and the stream_factory returns
a complete response instead of an iterator (e.g. AnthropicAuxiliaryClient
and other shims that ignore ``stream=True``), unwrap and return the
completed response directly. This mirrors the pre-Relay behavior where
``call_llm(stream=True)`` returned the raw response and the consumer's
own ``hasattr(stream, "choices")`` check handled it (#11732, #55933) —
without the unwrap the response stays trapped as ``final_response`` on the
inner ManagedLlmStream and the outer consumer sees an empty stream.
"""
turn = relay_runtime.active_turn()
if turn is None:
return stream_factory(request)
return stream(
if _has_running_event_loop():
# Managed provider callbacks execute on the Relay session's event
# loop. A nested ManagedLlmStream built here would be synchronously
# iterated on that same loop thread, which asyncio forbids
# ("Cannot run the event loop while another loop is running").
# Return the raw factory result instead: the outer managed stream
# already provides Relay tracking for the enclosing attempt, and its
# own completed_response_predicate traps a completed response (e.g.
# the MoA facade's auxiliary ``call_llm(stream=True)`` returning a
# full response when an adapter ignores ``stream=True``).
return stream_factory(request)
managed = stream(
request,
stream_factory,
session_id=turn.lease.session_id,
@@ -270,7 +300,17 @@ def stream_current(
finalizer=finalizer,
metadata=metadata,
defer_logical_completion=defer_logical_completion,
completed_response_predicate=completed_response_predicate,
)
# In the non-managed path the factory already ran eagerly during __init__,
# so a completed response is visible immediately and must surface raw.
# In the managed path the factory runs lazily on first pull, so
# final_response is still None here and the managed stream is returned.
if completed_response_predicate is not None:
completed = getattr(managed, "final_response", None)
if completed is not None:
return completed
return managed
def stream(
+66 -8
View File
@@ -23,6 +23,7 @@ Design rationale lives in ``docs/design/multiplexing-gateway.md`` (Workstream A)
from __future__ import annotations
import os
import re
from contextvars import ContextVar, Token
from pathlib import Path
from typing import Dict, Mapping, Optional
@@ -105,6 +106,14 @@ _GLOBAL_ENV_EXACT = frozenset({
"VIRTUAL_ENV", "PYTHONPATH", "SSL_CERT_FILE",
# Kanban paths (per-board, not per-profile-secret)
"HERMES_KANBAN_DB", "HERMES_KANBAN_WORKSPACES_ROOT", "HERMES_KANBAN_BOARD",
# API-server LISTENER settings — deployment config (Docker compose
# ``environment:`` block, systemd ``Environment=``), not profile secrets.
# The scoped runner reload (#64674) must keep seeing them or container
# deployments silently lose the api_server platform (#69379). NOTE:
# API_SERVER_KEY is deliberately NOT here — it IS a credential and stays
# profile-scoped.
"API_SERVER_ENABLED", "API_SERVER_HOST", "API_SERVER_PORT",
"API_SERVER_CORS_ORIGINS",
})
_GLOBAL_ENV_PREFIXES = (
"HERMES_KANBAN_",
@@ -177,20 +186,72 @@ def get_secret(name: str, default: Optional[str] = None) -> Optional[str]:
return val if val is not None else default
def _strip_inline_comment(value: str) -> str:
"""Strip a dotenv-style inline comment from a raw ``.env`` value.
Mirrors python-dotenv (1.2.2) semantics, verified empirically:
- Quoted values: scan for the matching close quote
(backslash-escape-aware for double quotes, since ``save_env_value``
writes ``\\"``/``\\\\`` escapes). Everything through the close quote is
kept; a trailing ``# ...`` remainder after it is discarded, so
``KEY="has # inside" # trailing`` yields ``has # inside``. Non-comment
trailing junk leaves the value untouched (lenient, unlike dotenv's
hard parse error).
- Unquoted values: truncate only at a ``#`` PRECEDED BY WHITESPACE, so
``KEY=foo#bar`` keeps ``foo#bar`` while ``KEY=value # comment`` keeps
``value``. A value that *starts* with ``#`` (``KEY=#leading``) is kept.
"""
value = value.strip()
if not value:
return value
quote = value[0]
if quote in ("'", '"'):
i = 1
while i < len(value):
ch = value[i]
if quote == '"' and ch == "\\":
i += 2 # skip the escaped character
continue
if ch == quote:
remainder = value[i + 1:].lstrip()
if remainder.startswith("#"):
return value[: i + 1]
return value
i += 1
return value # unterminated quote: leave as-is
return re.split(r"\s+#", value, maxsplit=1)[0].strip()
def load_env_file(env_path: Path) -> Dict[str, str]:
"""Parse a ``.env`` file into a plain dict WITHOUT touching ``os.environ``.
Used to load a profile's secrets into an isolated mapping for
``set_secret_scope``. Mirrors python-dotenv's basic parsing (KEY=VALUE,
``export`` prefix, ``#`` comments, optional matching quotes) but never
mutates the process environment — that isolation is the whole point.
``set_secret_scope``. Parses the small KEY=VALUE subset Hermes writes
itself (``export`` prefix, ``#`` comments — full-line and
dotenv-compatible inline, matching quotes with the
writer's ``\\"``/``\\\\`` escapes reversed — the same semantics as
``hermes_cli.config._parse_env_value``) but never mutates the process
environment — that isolation is the whole point.
Encoding is ``utf-8-sig`` so a leading UTF-8 BOM (Windows Notepad /
PowerShell ``Set-Content -Encoding UTF8``) does not prefix the first
key as ``\\ufeffNAME`` and make ``get_secret('NAME')`` miss under scope.
"""
secrets: Dict[str, str] = {}
try:
text = env_path.read_text(encoding="utf-8")
text = env_path.read_text(encoding="utf-8-sig")
except (FileNotFoundError, OSError, UnicodeDecodeError):
return secrets
# Parse values with the canonical Hermes parser: save_env_value
# escapes " and \ inside double quotes, and every other reader
# (load_env, python-dotenv) reverses those escapes. Stripping only
# the outer quotes here would corrupt credentials containing "
# or \ — they work interactively but fail in scoped (cron /
# multiplex) resolution.
from hermes_cli.config import _parse_env_value
for raw in text.splitlines():
line = raw.strip()
if not line or line.startswith("#"):
@@ -203,10 +264,7 @@ def load_env_file(env_path: Path) -> Dict[str, str]:
key = key.strip()
if not key:
continue
value = value.strip()
if len(value) >= 2 and value[0] == value[-1] and value[0] in ("'", '"'):
value = value[1:-1]
secrets[key] = value
secrets[key] = _parse_env_value(_strip_inline_comment(value))
return secrets
+20 -1
View File
@@ -39,17 +39,36 @@ from __future__ import annotations
import os
import re
import subprocess
from contextvars import ContextVar, Token
from abc import ABC, abstractmethod
from dataclasses import dataclass, field
from enum import Enum
from pathlib import Path
from typing import Dict, FrozenSet, List, Optional, Sequence
from typing import Dict, FrozenSet, List, MutableMapping, Optional, Sequence
# Bump ONLY for breaking changes to the required contract surface
# (abstract-method signatures, FetchResult required fields). Additive
# optional hooks must ship with defaults and must NOT bump this.
SECRET_SOURCE_API_VERSION = 1
_SOURCE_ENVIRONMENT: ContextVar[Optional[MutableMapping[str, str]]]
_SOURCE_ENVIRONMENT = ContextVar("hermes_secret_source_environment", default=None)
def set_source_environment(environ: MutableMapping[str, str]) -> Token:
"""Install a per-fetch environment view without changing ``os.environ``."""
return _SOURCE_ENVIRONMENT.set(environ)
def reset_source_environment(token: Token) -> None:
_SOURCE_ENVIRONMENT.reset(token)
def get_source_environment() -> MutableMapping[str, str]:
"""Return the active per-fetch environment, or the process environment."""
environ = _SOURCE_ENVIRONMENT.get()
return environ if environ is not None else os.environ
# Timeout the orchestrator enforces around fetch() when the source's
# config section doesn't override it. Generous because a first run may
# include a one-time CLI binary auto-install (e.g. bws download+verify).
+11 -5
View File
@@ -58,6 +58,7 @@ from agent.secret_sources._cache import (
is_valid_env_name as _is_valid_env_name,
)
from agent.secret_sources.base import ErrorKind, SecretSource
from agent.secret_sources.base import get_source_environment
logger = logging.getLogger(__name__)
@@ -667,10 +668,15 @@ def _run_bws_list(
bws: Path, access_token: str, project_id: str, server_url: str = ""
) -> Tuple[Dict[str, str], List[str]]:
cmd = [str(bws), "secret", "list", project_id, "--output", "json"]
# bws child intentionally receives the access token; exact preservation
# (BWS_SERVER_URL manual overrides etc. must survive untouched).
from tools.environments.local import build_subprocess_env
env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False)
# bws child intentionally receives the access token. Under a profile-local
# fetch it must not inherit sibling credentials from process-global env.
source_env = get_source_environment()
if source_env is os.environ:
from tools.environments.local import build_subprocess_env
env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False)
else:
env = dict(source_env)
env["BWS_ACCESS_TOKEN"] = access_token
# Make sure we're not echoing telemetry / colour codes into json.
env.setdefault("NO_COLOR", "1")
@@ -908,7 +914,7 @@ class BitwardenSource(SecretSource):
result = FetchResult()
access_token_env = str(cfg.get("access_token_env") or "BWS_ACCESS_TOKEN")
access_token = os.environ.get(access_token_env, "").strip()
access_token = get_source_environment().get(access_token_env, "").strip()
if not access_token:
result.error = (
f"secrets.bitwarden.enabled is true but {access_token_env} is "
+12 -2
View File
@@ -44,6 +44,7 @@ from typing import Dict, Optional
# Reuse the exact result shape the bitwarden source returns so
# hermes_cli.env_loader can consume both providers identically.
from agent.secret_sources.base import ErrorKind, SecretSource
from agent.secret_sources.base import get_source_environment
from agent.secret_sources.bitwarden import FetchResult
__all__ = [
@@ -181,8 +182,17 @@ def _run_helper(
# User-configured secret-helper command: runs with the user's full shell
# env by design (it may need any credential to resolve the secret).
from tools.environments.local import build_subprocess_env
env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False)
source_env = get_source_environment()
if source_env is os.environ:
# Legacy single-profile startup intentionally preserves the existing
# helper contract, which may rely on the user's full environment.
from tools.environments.local import build_subprocess_env
env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False)
else:
# A multiplex profile must never inherit sibling secrets from the
# process-global environment. hydrate_profile_secret_sources seeds
# only global-safe values plus this profile's own .env.
env = dict(source_env)
env["HERMES_SECRET_KEY"] = secret_key
try:
+12 -9
View File
@@ -55,6 +55,7 @@ from agent.secret_sources._cache import (
is_valid_env_name,
)
from agent.secret_sources.base import ErrorKind, SecretSource
from agent.secret_sources.base import get_source_environment
logger = logging.getLogger(__name__)
@@ -182,15 +183,16 @@ def _auth_fingerprint(token_env: str) -> str:
previous identity is never served under a new one. Never logged or
displayed; the raw token never leaves this hash.
"""
source_env = get_source_environment()
parts: List[str] = [
f"token={os.environ.get(token_env, '')}",
f"account={os.environ.get('OP_ACCOUNT', '')}",
f"connect_host={os.environ.get('OP_CONNECT_HOST', '')}",
f"connect_token={os.environ.get('OP_CONNECT_TOKEN', '')}",
f"token={source_env.get(token_env, '')}",
f"account={source_env.get('OP_ACCOUNT', '')}",
f"connect_host={source_env.get('OP_CONNECT_HOST', '')}",
f"connect_token={source_env.get('OP_CONNECT_TOKEN', '')}",
]
for key in sorted(os.environ):
for key in sorted(source_env):
if key.startswith("OP_SESSION_"):
parts.append(f"{key}={os.environ[key]}")
parts.append(f"{key}={source_env[key]}")
material = "\n".join(parts)
return hashlib.sha256(material.encode("utf-8")).hexdigest()[:16]
@@ -238,13 +240,14 @@ def _scrub(text: str) -> str:
def _op_child_env(token_value: str) -> Dict[str, str]:
"""Build a minimal allowlisted environment for the ``op`` child process."""
source_env = get_source_environment()
env: Dict[str, str] = {}
for key in _OP_ENV_ALLOWLIST:
val = os.environ.get(key)
val = source_env.get(key)
if val is not None:
env[key] = val
# Desktop / interactive session credentials.
for key, val in os.environ.items():
for key, val in source_env.items():
if key.startswith("OP_SESSION_"):
env[key] = val
# `op` reads OP_SERVICE_ACCOUNT_TOKEN regardless of which env var the user
@@ -340,7 +343,7 @@ def fetch_onepassword_secrets(
if not valid:
return {}, warnings
token_value = os.environ.get(token_env, "").strip()
token_value = get_source_environment().get(token_env, "").strip()
cache_key: _CacheKey = (
_auth_fingerprint(token_env),
account or "",
+15 -5
View File
@@ -32,7 +32,7 @@ import logging
import os
from dataclasses import dataclass, field
from pathlib import Path
from typing import Dict, List, Optional
from typing import Dict, List, MutableMapping, Optional
from agent.secret_sources.base import (
SECRET_SOURCE_API_VERSION,
@@ -40,6 +40,8 @@ from agent.secret_sources.base import (
FetchResult,
SecretSource,
is_valid_env_name,
reset_source_environment,
set_source_environment,
)
logger = logging.getLogger(__name__)
@@ -196,7 +198,8 @@ def _reset_registry_for_tests() -> None:
def _fetch_with_timeout(
source: SecretSource, cfg: dict, home_path: Path
source: SecretSource, cfg: dict, home_path: Path,
environ: MutableMapping[str, str],
) -> FetchResult:
"""Run source.fetch() under a wall-clock budget; never raises.
@@ -211,7 +214,14 @@ def _fetch_with_timeout(
max_workers=1, thread_name_prefix=f"secret-src-{source.name}"
)
try:
future = executor.submit(source.fetch, cfg, home_path)
def _fetch() -> FetchResult:
token = set_source_environment(environ)
try:
return source.fetch(cfg, home_path)
finally:
reset_source_environment(token)
future = executor.submit(_fetch)
try:
result = future.result(timeout=timeout)
except concurrent.futures.TimeoutError:
@@ -321,7 +331,7 @@ def _profile_alias_target(var: str, profile: str) -> Optional[str]:
def apply_all(secrets_cfg: dict, home_path: Path,
environ: Optional[Dict[str, str]] = None) -> ApplyReport:
environ: Optional[MutableMapping[str, str]] = None) -> ApplyReport:
"""Fetch from every enabled source and apply the merged result to env.
``environ`` defaults to ``os.environ``; injectable for tests.
@@ -376,7 +386,7 @@ def apply_all(secrets_cfg: dict, home_path: Path,
for source in ordered:
cfg = secrets_cfg.get(source.name)
cfg = cfg if isinstance(cfg, dict) else {}
result = _fetch_with_timeout(source, cfg, home_path)
result = _fetch_with_timeout(source, cfg, home_path, env)
fetches.append((source, cfg, result))
try:
for var in source.protected_env_vars(cfg):
+106
View File
@@ -0,0 +1,106 @@
"""Shared session activity observation contract (#72016 / #72039).
Observation-only: timestamp + bounded description/provenance.
Notification, timeout, kill, and retry policy stay in their own components.
Consumers distinguish work (API / tool / compacting / stalled) from the
description text itself — there is no separate phase enum.
Provenance is a small closed enum of *noun* sources (where the stamp came
from). The default agent activity clock (``_touch_activity``) stamps
``unknown`` unless a caller passes an explicit ``provenance=``; named
values are for special writers.
"""
from __future__ import annotations
from enum import Enum
from typing import Any, Mapping, Optional
ACTIVITY_DESCRIPTION_MAX = 120
# Durable SessionDB activity heartbeat cadence (seconds between writes per
# session). Contract: MUST stay >= 30s — the SessionDB write path is
# contended (deadline/patience retry, compression-lock patience), and the
# heartbeat is an observation-only projection that never justifies extra
# write pressure. This cadence is deliberately a code constant, independent
# of any compression.* or agent.* config, so no configuration can turn the
# heartbeat into a high-frequency writer. Matches the kanban auto-heartbeat
# cadence. force_persist (terminal stamps) is the only bypass.
SESSION_ACTIVITY_HEARTBEAT_MIN_INTERVAL_SECONDS = 60.0
class ActivityProvenance(str, Enum):
"""Where a durable/in-memory activity stamp came from."""
UNKNOWN = "unknown"
# Compression writers (#72424 / activity contract): heartbeat, host timeout, cooldown.
AGENT_COMPRESSION = "agent.compression"
AGENT_COMPRESSION_TIMEOUT = "agent.compression_timeout"
AGENT_COMPRESSION_COOLDOWN = "agent.compression_cooldown"
def bound_activity_description(description: Optional[str]) -> str:
"""Clamp free-form activity text to the shared description budget."""
text = (description or "").strip()
if len(text) <= ACTIVITY_DESCRIPTION_MAX:
return text
return text[: ACTIVITY_DESCRIPTION_MAX - 1] + "…"
def normalize_activity_provenance(
provenance: Optional[ActivityProvenance | str],
) -> ActivityProvenance:
"""Return a known provenance, or ``UNKNOWN`` when unset/unrecognized."""
if isinstance(provenance, ActivityProvenance):
return provenance
value = (provenance or "").strip()
try:
return ActivityProvenance(value)
except ValueError:
return ActivityProvenance.UNKNOWN
def reset_session_activity_persist_window(agent: Any) -> None:
"""Clear the agent's durable SessionDB activity persist rate-limit window.
The next ``_touch_activity`` / ``_persist_session_activity_if_due`` will
write through even if a stamp landed within the last 60s. Used for
terminal compression labels that must not stay stuck on mid-compress
text (e.g. "context compression in progress" after /compress).
"""
try:
agent._session_activity_last_persist_mono = 0.0
except Exception:
pass
def build_activity_snapshot(
*,
last_activity_at: Optional[float],
last_activity_description: Optional[str],
last_activity_provenance: Optional[ActivityProvenance | str] = None,
now: Optional[float] = None,
extra: Optional[Mapping[str, Any]] = None,
) -> dict[str, Any]:
"""Build the shared activity snapshot (plus optional caller extras)."""
import time as _time
when = float(last_activity_at) if last_activity_at is not None else None
clock = float(now if now is not None else _time.time())
desc = bound_activity_description(last_activity_description)
prov = normalize_activity_provenance(last_activity_provenance)
elapsed = round(clock - when, 1) if when is not None else None
snap: dict[str, Any] = {
"last_activity_at": when,
"last_activity_description": desc,
"last_activity_provenance": prov.value,
"seconds_since_activity": elapsed,
# Short aliases used by existing gateway/delegate readers.
"last_activity_ts": when,
"last_activity_desc": desc,
"description": desc,
"provenance": prov.value,
}
if extra:
snap.update(dict(extra))
return snap
+3 -2
View File
@@ -319,8 +319,9 @@ def _parse_hooks_block(hooks_cfg: Any) -> List[ShellHookSpec]:
for event_name, entries in hooks_cfg.items():
# Reserved sub-keys that aren't event names — skip silently. These
# are config sub-sections nested under `hooks:` for related
# functionality (e.g. output-spill budgets).
if event_name in ("output_spill",):
# functionality (e.g. output-spill budgets, outbound webhooks —
# the latter parsed by agent/outbound_webhooks.py).
if event_name in ("output_spill", "outbound"):
continue
if event_name not in VALID_HOOKS:
suggestion = difflib.get_close_matches(
+9 -2
View File
@@ -21,6 +21,7 @@ from contextlib import contextmanager
from concurrent.futures import Future, TimeoutError
from typing import Any, Callable, Mapping, Optional
from agent.interrupt_compat import request_hard_interrupt
PUBLIC_CONTRACT_VERSION = 1
_MAX_GOAL_CHARS = 16_000
@@ -300,16 +301,22 @@ class SubagentLifecycleService:
agent = record.agent
record.state = SubagentState.CANCEL_REQUESTED
record.updated_at = time.time()
if agent is None or not hasattr(agent, "interrupt"):
if agent is None:
return SubagentCancelResult(
False, unsupported=True, state=SubagentState.CANCEL_REQUESTED
)
try:
agent.interrupt(f"Lifecycle cancellation requested: {reason[:500]}")
accepted = request_hard_interrupt(
agent, f"Lifecycle cancellation requested: {reason[:500]}"
)
except Exception:
return SubagentCancelResult(
False, unsupported=True, state=SubagentState.CANCEL_REQUESTED
)
if not accepted:
return SubagentCancelResult(
False, unsupported=True, state=SubagentState.CANCEL_REQUESTED
)
return SubagentCancelResult(True, state=SubagentState.CANCEL_REQUESTED)
def result(self, handle: SubagentHandle) -> SubagentResult:
+102 -24
View File
@@ -4,9 +4,10 @@ Pure module-level utilities extracted from ``run_agent.py``:
* ``_is_destructive_command`` — terminal-command heuristic used to gate
parallel batch dispatch.
* ``_should_parallelize_tool_batch`` / ``_extract_parallel_scope_path`` /
``_paths_overlap`` — the rules engine deciding when a multi-tool batch
can run concurrently.
* ``_should_parallelize_tool_batch`` / ``_extract_parallel_scope_paths`` /
``_extract_parallel_scope_path`` / ``_paths_overlap`` — the rules engine
deciding when a multi-tool batch can run concurrently (V4A patch scope
uses patch-body file headers, not a decoy ``path=``).
* ``_is_multimodal_tool_result`` / ``_multimodal_text_summary`` /
``_append_subdir_hint_to_multimodal`` — envelope helpers for the
``{"_multimodal": True, "content": [...], "text_summary": ...}`` dict
@@ -57,8 +58,17 @@ _PARALLEL_SAFE_TOOLS = frozenset({
"web_search",
})
# Filesystem tools whose parallel admission is decided by path overlap.
# Readers may share a subtree with other readers; a writer conflicts with
# ANY overlapping reservation (reader or writer). This is what keeps a
# batched ``search_files``/``read_file`` from observing pre-mutation file
# state when the model batches it alongside the ``patch``/``write_file``
# it depends on (the classic same-block write→read race).
_PATH_SCOPED_READERS = frozenset({"read_file", "search_files"})
_PATH_SCOPED_WRITERS = frozenset({"write_file", "patch"})
# File tools can run concurrently when they target independent paths.
_PATH_SCOPED_TOOLS = frozenset({"read_file", "write_file", "patch"})
_PATH_SCOPED_TOOLS = _PATH_SCOPED_READERS | _PATH_SCOPED_WRITERS
# Patterns that indicate a terminal command may modify/delete files.
_DESTRUCTIVE_PATTERNS = re.compile(
@@ -116,10 +126,18 @@ def _plan_tool_batch_segments(tool_calls, *, execution_cwd: Optional[Path] = Non
* ``_NEVER_PARALLEL_TOOLS`` (interactive tools) → barrier.
* Unparseable / non-dict arguments → barrier.
* Path-scoped tools (``read_file``/``write_file``/``patch``) join a
parallel run only when their target path does not overlap another
path already reserved in the same run; an overlap closes the run so
the conflicting call starts a NEW run after the first completes.
* Path-scoped tools (``read_file``/``search_files``/``write_file``/
``patch``) join a parallel run only when their target path(s) do not
CONFLICT with a path already reserved in the same run. Reservations
carry a reader/writer role: reader↔reader overlap is harmless (two
reads of the same file commute) and stays parallel; any overlap
involving a writer closes the run so the conflicting call starts a
NEW run after the first completes. ``search_files`` reserves its
search root (default ``.``) as a reader — a search batched after a
write into the searched subtree is ordered behind that write instead
of racing it. For V4A ``patch(mode="patch")`` the reserved paths are
the file headers in the patch body, not a possibly-stale ``path=``
argument.
* Anything not in ``_PARALLEL_SAFE_TOOLS`` and not an opted-in MCP
tool → barrier.
@@ -129,7 +147,8 @@ def _plan_tool_batch_segments(tool_calls, *, execution_cwd: Optional[Path] = Non
"""
segments: list[list] = [] # [kind, calls] pairs, normalized to tuples on return
current: list = []
reserved_paths: list[Path] = []
# (canonical_path, is_writer) reservations for the current parallel run.
reserved_paths: list[tuple[Path, bool]] = []
def _close_parallel() -> None:
nonlocal current, reserved_paths
@@ -173,15 +192,25 @@ def _plan_tool_batch_segments(tool_calls, *, execution_cwd: Optional[Path] = Non
continue
if tool_name in _PATH_SCOPED_TOOLS:
scoped_path = _extract_parallel_scope_path(tool_name, function_args, execution_cwd=execution_cwd)
if scoped_path is None:
scoped_paths = _extract_parallel_scope_paths(
tool_name, function_args, execution_cwd=execution_cwd
)
if not scoped_paths:
_add_sequential(tool_call)
continue
if any(_paths_overlap(scoped_path, existing) for existing in reserved_paths):
is_writer = tool_name in _PATH_SCOPED_WRITERS
if any(
(is_writer or existing_is_writer)
and _paths_overlap(scoped_path, existing)
for scoped_path in scoped_paths
for existing, existing_is_writer in reserved_paths
):
# Same-subtree conflict inside this run: close it so this
# call starts a fresh run AFTER the conflicting one lands.
# Reader↔reader overlap never conflicts — concurrent reads
# of the same subtree commute.
_close_parallel()
reserved_paths.append(scoped_path)
reserved_paths.extend((p, is_writer) for p in scoped_paths)
current.append(tool_call)
continue
@@ -233,33 +262,77 @@ def _canonical_path(raw_path: str, execution_cwd: Optional[Path] = None) -> Path
return Path(resolved)
def _extract_parallel_scope_path(
def _extract_parallel_scope_paths(
tool_name: str,
function_args: dict,
execution_cwd: Optional[Path] = None,
) -> Optional[Path]:
"""Return the canonical file target for path-scoped tools.
) -> List[Path]:
"""Return every canonical path this call reserves for overlap checks.
*execution_cwd* should be the working directory that the tool will
actually use at runtime. When omitted the process cwd is used,
which may differ from the tool execution environment on some
platforms (e.g. WSL, sandboxed sub-processes).
For ``patch`` in V4A ``mode=patch``, scope comes from patch-body
``*** Update/Add/Delete/Move File:`` headers (not a possibly-decoy
``path=``). An empty result means the planner cannot determine the
scope and must treat the call as a sequential barrier.
"""
if tool_name not in _PATH_SCOPED_TOOLS:
return None
return []
raw_path = function_args.get("path")
if not isinstance(raw_path, str) or not raw_path.strip():
return None
raw_paths: List[str] = []
if tool_name == "patch" and (function_args.get("mode") or "replace") == "patch":
raw_paths.extend(_extract_file_mutation_targets(tool_name, function_args))
else:
raw_path = function_args.get("path")
if isinstance(raw_path, str) and raw_path.strip():
raw_paths.append(raw_path)
elif tool_name == "search_files":
# ``search_files`` defaults its search root to the cwd when
# ``path`` is omitted — reserve that root rather than falling
# back to a sequential barrier (an empty result here would
# demote every bare search to a barrier and destroy read
# parallelism).
raw_paths.append(".")
return _canonical_path(raw_path, execution_cwd)
scoped: List[Path] = []
seen: set[str] = set()
for raw in raw_paths:
if not isinstance(raw, str) or not raw.strip():
continue
canonical = _canonical_path(raw, execution_cwd)
key = str(canonical)
if key in seen:
continue
seen.add(key)
scoped.append(canonical)
return scoped
def _extract_parallel_scope_path(
tool_name: str,
function_args: dict,
execution_cwd: Optional[Path] = None,
) -> Optional[Path]:
"""Return the primary canonical file target for path-scoped tools.
Thin view over ``_extract_parallel_scope_paths`` kept for callers/tests
that only need a single representative path. For multi-file V4A
patches this is the first header target.
"""
scoped = _extract_parallel_scope_paths(
tool_name, function_args, execution_cwd=execution_cwd
)
return scoped[0] if scoped else None
def _paths_overlap(left: Path, right: Path) -> bool:
"""Return True when two paths may refer to the same subtree.
Both *left* and *right* must already be canonical (as returned by
``_extract_parallel_scope_path`` / ``_canonical_path``) so that
``_extract_parallel_scope_paths`` / ``_canonical_path``) so that
symlink aliases and case differences are already normalised.
"""
left_parts = left.parts
@@ -354,8 +427,10 @@ def _extract_file_mutation_targets(tool_name: str, args: Dict[str, Any]) -> List
if not isinstance(body, str) or not body:
return []
paths: List[str] = []
# ``\s*`` (not ``\s+``) after ``***`` matches patch_parser / file_tools:
# they accept ``***Update File:`` with no space after the asterisks.
for _m in re.finditer(
r'^\*\*\*\s+(?:Update|Add|Delete)\s+File:\s*(.+)$',
r'^\*\*\*\s*(?:Update|Add|Delete)\s+File:\s*(.+)$',
body,
re.MULTILINE,
):
@@ -363,7 +438,7 @@ def _extract_file_mutation_targets(tool_name: str, args: Dict[str, Any]) -> List
if p:
paths.append(p)
for _m in re.finditer(
r'^\*\*\*\s+Move\s+File:\s*(.+?)\s*->\s*(.+)$',
r'^\*\*\*\s*Move\s+File:\s*(.+?)\s*->\s*(.+)$',
body,
re.MULTILINE,
):
@@ -634,6 +709,8 @@ __all__ = [
"_NEVER_PARALLEL_TOOLS",
"_PARALLEL_SAFE_TOOLS",
"_PATH_SCOPED_TOOLS",
"_PATH_SCOPED_READERS",
"_PATH_SCOPED_WRITERS",
"_DESTRUCTIVE_PATTERNS",
"_REDIRECT_OVERWRITE",
"_is_destructive_command",
@@ -641,6 +718,7 @@ __all__ = [
"_should_parallelize_tool_batch",
"_canonical_path",
"_extract_parallel_scope_path",
"_extract_parallel_scope_paths",
"_paths_overlap",
"_is_multimodal_tool_result",
"_multimodal_text_summary",
+49 -20
View File
@@ -1236,8 +1236,8 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
logging.debug("file-mutation verifier record failed: %s", _ver_err)
if agent.verbose_logging:
logging.debug(f"Tool {function_name} completed in {tool_duration:.2f}s")
logging.debug(f"Tool result ({len(function_result)} chars): {function_result}")
logging.debug("Tool %s completed in %.2fs", function_name, tool_duration)
logging.debug("Tool result (%d chars): %s", len(function_result), function_result)
agent._current_tool = None
_status_suffix = " (error)" if is_error else ""
@@ -1296,7 +1296,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
result=display_function_result,
)
except Exception as cb_err:
logging.debug(f"Tool progress callback error: {cb_err}")
logging.debug("Tool progress callback error: %s", cb_err)
# Print cute message per tool
if agent._should_emit_quiet_tool_messages():
@@ -1320,7 +1320,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
tc.id, name, display_args, display_function_result,
)
except Exception as cb_err:
logging.debug(f"Tool complete callback error: {cb_err}")
logging.debug("Tool complete callback error: %s", cb_err)
if (
risk_metadata is not None
@@ -1339,12 +1339,10 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
except Exception as cb_err:
logging.debug("Tool output risk callback error: %s", cb_err)
# ── Per-tool /steer drain ───────────────────────────────────
# Same as the sequential path: drain between each collected
# result so the steer lands as early as possible.
agent._apply_pending_steer_to_tool_results(messages, 1)
# ── Per-turn aggregate budget enforcement ─────────────────────────
# Keep /steer pending until the final post-budget drain below. The model
# cannot observe a partial batch, while an early drain can be discarded
# when aggregate budget enforcement replaces that tool result.
num_tools = len(parsed_calls)
if finalize and num_tools > 0:
turn_tool_msgs = messages[-num_tools:]
@@ -1359,6 +1357,26 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
def _append_cancelled_tool_results(messages: list, tool_calls, *, reason: str) -> None:
"""Append a cancelled ``tool`` result for each call in ``tool_calls``.
Used when a hard interrupt (KeyboardInterrupt / BaseException) aborts the
sequential executor mid-batch. Without this, the loop re-raises leaving the
assistant tool-call turn with no matching tool results — a message-role
alternation violation that malforms the next provider request. Mirrors the
cooperative-interrupt skip block and the concurrent path, both of which
already emit a result for every call_id.
"""
for tc in tool_calls:
name = getattr(getattr(tc, "function", None), "name", "") or "tool"
messages.append(make_tool_result_message(
name,
f"[Tool execution cancelled — {name} was skipped due to {reason}]",
getattr(tc, "id", "") or "",
effect_disposition="none",
))
def execute_tool_calls_sequential(agent, assistant_message, messages: list, effective_task_id: str, api_call_count: int = 0, *, finalize: bool = True) -> None:
"""Execute tool calls sequentially (original behavior). Used for single calls or interactive tools.
@@ -1439,7 +1457,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
stage=f"invalid tool arguments {function_name}",
):
return
agent._apply_pending_steer_to_tool_results(messages, 1)
continue
# Tool Search unwrap — see execute_tool_calls_concurrent for full
@@ -1799,6 +1816,14 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
agent.interrupt("keyboard interrupt")
except Exception:
pass
# Emit a tool result for THIS call and every remaining call in
# the batch before re-raising, so the assistant tool-call turn
# is never left without matching tool results (alternation).
_append_cancelled_tool_results(
messages,
assistant_message.tool_calls[i - 1:],
reason="keyboard interrupt",
)
raise
except Exception as tool_error:
function_result = f"Error executing tool '{function_name}': {tool_error}"
@@ -1868,6 +1893,13 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
agent.interrupt("keyboard interrupt")
except Exception:
pass
# Emit a tool result for THIS call and every remaining call in
# the batch before re-raising (see interactive branch above).
_append_cancelled_tool_results(
messages,
assistant_message.tool_calls[i - 1:],
reason="keyboard interrupt",
)
raise
except Exception as tool_error:
function_result = f"Error executing tool '{function_name}': {tool_error}"
@@ -1944,9 +1976,9 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
agent._touch_activity(f"tool completed: {function_name} ({tool_duration:.1f}s){_status_suffix}")
if agent.verbose_logging:
logging.debug(f"Tool {function_name} completed in {tool_duration:.2f}s")
logging.debug("Tool %s completed in %.2fs", function_name, tool_duration)
_log_result = _multimodal_text_summary(function_result)
logging.debug(f"Tool result ({len(_log_result)} chars): {_log_result}")
logging.debug("Tool result (%d chars): %s", len(_log_result), _log_result)
display_function_result = function_result
function_result = maybe_persist_tool_result(
@@ -1988,7 +2020,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
result=display_function_result,
)
except Exception as cb_err:
logging.debug(f"Tool progress callback error: {cb_err}")
logging.debug("Tool progress callback error: %s", cb_err)
if not _execution_blocked and agent.tool_complete_callback:
try:
@@ -2003,7 +2035,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
display_function_result,
)
except Exception as cb_err:
logging.debug(f"Tool complete callback error: {cb_err}")
logging.debug("Tool complete callback error: %s", cb_err)
if (
risk_metadata is not None
@@ -2022,12 +2054,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
except Exception as cb_err:
logging.debug("Tool output risk callback error: %s", cb_err)
# ── Per-tool /steer drain ───────────────────────────────────
# Drain pending steer BETWEEN individual tool calls so the
# injection lands as soon as a tool finishes — not after the
# entire batch. The model sees it on the next API iteration.
agent._apply_pending_steer_to_tool_results(messages, 1)
if not agent.quiet_mode and getattr(agent, "tool_progress_mode", "all") != "off":
if agent.verbose_logging:
print(f" ✅ Tool {i} completed in {tool_duration:.2f}s")
@@ -2057,6 +2083,9 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
break
# ── Per-turn aggregate budget enforcement ─────────────────────────
# Keep /steer pending until the final post-budget drain below. The model
# only receives this batch after all calls finish, and an early drain can
# be discarded when aggregate budget enforcement replaces a tool result.
num_tools_seq = len(assistant_message.tool_calls)
if finalize and num_tools_seq > 0:
enforce_turn_budget(messages[-num_tools_seq:], env=get_active_env(effective_task_id), config=_tool_budget)
+4
View File
@@ -544,10 +544,13 @@ class ResponsesApiTransport(ProviderTransport):
*,
allow_stream: bool = False,
is_github_responses: bool = False,
sanitize_harmony_tokens: bool = False,
) -> dict:
"""Validate and sanitize Codex API kwargs before the call.
Normalizes input items, strips unsupported fields, validates structure.
``sanitize_harmony_tokens`` is enabled only for the ChatGPT Codex
backend, which rejects literal reserved Harmony wire tokens in text.
"""
from agent.codex_responses_adapter import _preflight_codex_api_kwargs
@@ -555,6 +558,7 @@ class ResponsesApiTransport(ProviderTransport):
api_kwargs,
allow_stream=allow_stream,
is_github_responses=is_github_responses,
sanitize_harmony_tokens=sanitize_harmony_tokens,
)
if "prompt_cache_key" in normalized:
bounded = _bounded_prompt_cache_key(normalized["prompt_cache_key"])
+25
View File
@@ -339,6 +339,12 @@ def finalize_turn(
# otherwise ``/resume`` reloads ``content=""`` and the bug
# resurfaces cross-session.
_tail.pop("_db_persisted", None)
# The bounded flush-scan cursor (run_agent.py) skips the
# identity-matched prefix of its previous snapshot on the
# assumption that no live dict loses the marker in place —
# this pop is the one place that does. Invalidate it so the
# filled row is re-examined instead of skipped.
agent._db_flush_scan_prefix = None
# The model has completed its request, so replace API-local
# voice/model/skill guidance with the clean user input before writing the
@@ -378,6 +384,18 @@ def finalize_turn(
):
_before = len(messages)
_compacted = _compressor._micro_compact(messages)
# Micro-compaction defrag rewrites the newest MICRO
# marker's content and pops _db_persisted from the live
# dict in place — the sibling of the pop site above. The
# compressor has no agent reference, so it raises a flag
# for us to invalidate the bounded flush-scan cursor;
# otherwise the rewritten marker row is identity-skipped
# and the stale summary persists to state.db.
if getattr(
_compressor, "_flush_scan_cursor_invalidated", False
):
_compressor._flush_scan_cursor_invalidated = False
agent._db_flush_scan_prefix = None
if isinstance(_compacted, list) and _compacted:
messages[:] = _compacted
_after = len(messages)
@@ -647,6 +665,13 @@ def finalize_turn(
}
if agent._tool_guardrail_halt_decision is not None:
result["guardrail"] = agent._tool_guardrail_halt_decision.to_metadata()
# Persistence failures already set failed=True + an explanation in
# final_response; also stamp `error` so gateway surfaces status="error"
# (and desktop can toast disk-full) instead of a quiet complete frame.
if failed and str(_turn_exit_reason) == "session_persistence_failed":
result["error"] = final_response or (
"session storage could not be written — free disk space and try again"
)
# Surface any post-loop cleanup failures so the caller can distinguish a
# clean turn from one whose trajectory/session/resource teardown raised
# (the response is still returned either way — #8049).
+8
View File
@@ -45,6 +45,14 @@ class TurnRetryState:
nous_auth_retry_attempted: bool = False
nous_paid_entitlement_refresh_attempted: bool = False
copilot_auth_retry_attempted: bool = False
# Copilot surfaces a stale/degraded credential as a 400
# ``model_not_available_for_integrator`` / ``model_not_supported`` instead
# of a clean 401 (e.g. a raw OAuth token seeded when the token exchange
# degraded at startup, routing the request to the restricted
# ``copilot-language-server`` integrator). Guard a single-shot forced
# re-exchange + client rebuild for that case, separate from the 401 guard
# so both can fire within one attempt if needed.
copilot_stale_cred_retry_attempted: bool = False
vertex_auth_retry_attempted: bool = False
# ── Format / payload recovery guards ─────────────────────────────────
+26 -34
View File
@@ -19,41 +19,33 @@
"fix": "npm run lint:fix"
},
"dependencies": {
"@nous-research/ui": "0.16.0",
"@tailwindcss/vite": "^4.2.4",
"@tailwindcss/typography": "^0.5.19",
"@tauri-apps/api": "^2.0.0",
"@tauri-apps/plugin-dialog": "^2.0.0",
"@tauri-apps/plugin-opener": "^2.0.0",
"@tauri-apps/plugin-process": "^2.0.0",
"@tauri-apps/plugin-shell": "^2.0.0",
"@vscode/codicons": "^0.0.45",
"class-variance-authority": "^0.7.1",
"clsx": "^2.1.1",
"katex": "^0.16.45",
"lucide-react": "^0.577.0",
"nanostores": "^1.3.0",
"radix-ui": "^1.4.3",
"react": "^19.2.4",
"react-dom": "^19.2.4",
"tailwind-merge": "^3.5.0",
"tailwindcss": "^4.2.1",
"tw-shimmer": "^0.4.11"
"@nous-research/ui": "0.18.2",
"@tailwindcss/typography": "0.5.20",
"@tailwindcss/vite": "4.3.3",
"@tauri-apps/api": "2.11.1",
"@tauri-apps/plugin-dialog": "2.7.1",
"@tauri-apps/plugin-opener": "2.5.4",
"@tauri-apps/plugin-process": "2.3.1",
"@tauri-apps/plugin-shell": "2.3.5",
"@vscode/codicons": "0.0.45",
"class-variance-authority": "0.7.1",
"clsx": "2.1.1",
"katex": "0.16.47",
"lucide-react": "0.577.0",
"nanostores": "1.4.0",
"radix-ui": "1.6.7",
"react": "19.2.7",
"react-dom": "19.2.7",
"tailwind-merge": "3.6.0",
"tailwindcss": "4.3.3",
"tw-shimmer": "0.4.12"
},
"devDependencies": {
"@eslint/js": "^9.39.4",
"@tauri-apps/cli": "^2.0.0",
"@types/react": "^19.2.14",
"@types/react-dom": "^19.2.3",
"@vitejs/plugin-react": "^6.0.2",
"eslint": "^9.39.4",
"eslint-plugin-perfectionist": "^5.9.0",
"eslint-plugin-react": "^7.37.5",
"eslint-plugin-react-hooks": "^7.1.1",
"eslint-plugin-unused-imports": "^4.4.1",
"globals": "^17.4.0",
"typescript": "^6.0.3",
"typescript-eslint": "^8.56.1",
"vite": "^8.0.16"
"@tauri-apps/cli": "2.11.4",
"@types/react": "19.2.17",
"@types/react-dom": "19.2.3",
"@vitejs/plugin-react": "6.0.3",
"typescript": "6.0.3",
"vite": "8.2.0"
}
}
@@ -98,6 +98,12 @@ pub fn update_in_progress_marker() -> PathBuf {
/// that path), where copying onto ourselves would be a Windows sharing
/// violation. Best-effort: a failure here must not fail the install, so the
/// caller logs and continues.
///
/// NOTE: because of that no-op, a user's staged installer is only ever written
/// by a full install/repair. Every later `--update` runs the ORIGINAL binary,
/// so an installer-protocol change can strand the whole installed base on a
/// binary that predates it (see `restage_from_checkout`, which repairs this
/// from the freshly-updated checkout).
pub fn copy_self_to_hermes_home() -> std::io::Result<()> {
let src = std::env::current_exe()?;
let dest = installer_dest();
Binary file not shown.

Before

Width:  |  Height:  |  Size: 361 KiB

After

Width:  |  Height:  |  Size: 94 KiB

+69 -4
View File
@@ -7,6 +7,7 @@ import {
appendUniquePathEntries,
buildDesktopBackendEnv,
buildDesktopBackendPath,
hermesManagedNodePathEntries,
normalizeHermesHomeRoot,
pathEnvKey,
POSIX_SANE_PATH_ENTRIES
@@ -22,8 +23,12 @@ test('desktop backend PATH adds Hermes-managed bins and missing POSIX sane entri
})
const entries = result.split(':')
assert.equal(entries[0], '/Users/test/.hermes/node/bin')
assert.equal(entries[1], '/Users/test/.hermes/hermes-agent/venv/bin')
// Both managed-Node layouts lead, POSIX-native shape first, then the venv.
assert.deepEqual(entries.slice(0, 3), [
'/Users/test/.hermes/node/bin',
'/Users/test/.hermes/node',
'/Users/test/.hermes/hermes-agent/venv/bin'
])
assert.ok(entries.includes('/opt/homebrew/bin'), 'Apple Silicon Homebrew bin is added')
assert.ok(entries.includes('/opt/homebrew/sbin'), 'Apple Silicon Homebrew sbin is added')
assert.ok(entries.includes('/usr/local/sbin'), 'missing standard sbin is added')
@@ -33,6 +38,56 @@ test('desktop backend PATH adds Hermes-managed bins and missing POSIX sane entri
}
})
test('managed Node dirs lead with the platform-native layout but always offer both', () => {
const posix = hermesManagedNodePathEntries('/Users/test/.hermes', {
platform: 'darwin',
pathModule: path.posix
})
const windows = hermesManagedNodePathEntries('C:\\Users\\test\\AppData\\Local\\hermes', {
platform: 'win32',
pathModule: path.win32
})
// install.sh uses node/bin; install.ps1 unpacks node.exe into node\ itself.
// Both shapes are always emitted so migrated installs keep resolving.
assert.deepEqual(posix, ['/Users/test/.hermes/node/bin', '/Users/test/.hermes/node'])
assert.deepEqual(windows, [
'C:\\Users\\test\\AppData\\Local\\hermes\\node',
'C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin'
])
})
test('managed Node dirs are empty without a Hermes home', () => {
assert.deepEqual(hermesManagedNodePathEntries(undefined, { platform: 'darwin', pathModule: path.posix }), [])
assert.deepEqual(hermesManagedNodePathEntries('', { platform: 'win32', pathModule: path.win32 }), [])
})
test('every managed Node dir outranks the inherited PATH on both platforms', () => {
for (const [platform, pathModule, home, inherited, delimiter] of [
['darwin', path.posix, '/Users/test/.hermes', '/usr/local/bin:/usr/bin', ':'],
['win32', path.win32, 'C:\\hermes', 'C:\\Program Files\\nodejs;C:\\Windows\\System32', ';']
] as const) {
const entries = buildDesktopBackendPath({
hermesHome: home,
venvRoot: null,
currentPath: inherited,
platform,
pathModule
}).split(delimiter)
const managed = hermesManagedNodePathEntries(home, { platform, pathModule })
const firstInherited = Math.min(...inherited.split(delimiter).map(entry => entries.indexOf(entry)))
for (const dir of managed) {
assert.ok(
entries.indexOf(dir) >= 0 && entries.indexOf(dir) < firstInherited,
`${dir} must precede the inherited PATH on ${platform}`
)
}
}
})
test('desktop backend PATH preserves first occurrence and avoids duplicates', () => {
const result = buildDesktopBackendPath({
hermesHome: '/Users/test/.hermes',
@@ -64,7 +119,11 @@ test('buildDesktopBackendEnv extends PYTHONPATH and backend PATH together', () =
})
assert.equal(env.PYTHONPATH, '/repo/hermes-agent:/existing/pythonpath')
assert.ok(env.PATH.startsWith('/Users/test/.hermes/node/bin:/Users/test/.hermes/hermes-agent/venv/bin:'))
assert.ok(
env.PATH.startsWith(
'/Users/test/.hermes/node/bin:/Users/test/.hermes/node:/Users/test/.hermes/hermes-agent/venv/bin:'
)
)
assert.ok(env.PATH.includes('/opt/homebrew/bin'))
})
@@ -115,7 +174,13 @@ test('Windows PATH casing and delimiter are preserved without POSIX sane entries
assert.equal(pathEnvKey({ Path: 'x' }, 'win32'), 'Path')
assert.equal(env.PATH, undefined)
assert.ok(env.Path.startsWith('C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin;'))
// Windows leads with the portable layout (install.ps1 unpacks node.exe
// straight into node\, no bin\), then the POSIX shape for migrated installs.
assert.ok(
env.Path.startsWith(
'C:\\Users\\test\\AppData\\Local\\hermes\\node;C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin;'
)
)
assert.ok(env.Path.includes('\\venv\\Scripts;'))
assert.ok(env.Path.includes(';C:\\Windows\\System32;C:\\Windows'))
assert.equal(env.Path.includes('/opt/homebrew/bin'), false)
+31 -2
View File
@@ -60,6 +60,34 @@ function appendUniquePathEntries(entries, { delimiter = path.delimiter } = {}) {
return ordered.join(delimiter)
}
/**
* Hermes-managed Node.js directories, in preferred lookup order.
*
* There are two on-disk layouts. `scripts/install.ps1` unpacks portable Node
* straight into `%LOCALAPPDATA%\hermes\node` (node.exe at the root, no `bin\`);
* `scripts/install.sh` and the node-bootstrap helper use the POSIX
* `$HERMES_HOME/node/bin`. Emit BOTH on every platform so mixed and migrated
* installs resolve, leading with the layout native to the current platform.
*
* This is the single source of truth for the ordering rule on the Node side —
* `main.ts` imports it rather than keeping its own copy. Mirrors
* `iter_hermes_node_dirs()` in hermes_constants.py, which the Electron main
* process cannot import.
*/
function hermesManagedNodePathEntries(
hermesHome,
{ platform = process.platform, pathModule = pathModuleForPlatform(platform) }: any = {}
) {
if (!hermesHome) {
return []
}
const root = pathModule.join(hermesHome, 'node')
const bin = pathModule.join(root, 'bin')
return platform === 'win32' ? [root, bin] : [bin, root]
}
function buildDesktopBackendPath({
hermesHome,
venvRoot,
@@ -68,11 +96,11 @@ function buildDesktopBackendPath({
pathModule = pathModuleForPlatform(platform)
}: any = {}) {
const delimiter = delimiterForPlatform(platform)
const hermesNodeBin = hermesHome ? pathModule.join(hermesHome, 'node', 'bin') : null
const hermesNodeDirs = hermesManagedNodePathEntries(hermesHome, { platform, pathModule })
const venvBin = venvRoot ? pathModule.join(venvRoot, platform === 'win32' ? 'Scripts' : 'bin') : null
const saneEntries = platform === 'win32' ? [] : POSIX_SANE_PATH_ENTRIES
return appendUniquePathEntries([hermesNodeBin, venvBin, currentPath, saneEntries], { delimiter })
return appendUniquePathEntries([hermesNodeDirs, venvBin, currentPath, saneEntries], { delimiter })
}
function normalizeHermesHomeRoot(hermesHome, { pathModule = pathModuleForPlatform(process.platform) }: any = {}) {
@@ -126,6 +154,7 @@ export {
buildDesktopBackendEnv,
buildDesktopBackendPath,
delimiterForPlatform,
hermesManagedNodePathEntries,
normalizeHermesHomeRoot,
pathEnvKey,
POSIX_SANE_PATH_ENTRIES
@@ -0,0 +1,109 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { decideBootstrapRepair } from './bootstrap-repair-guard'
test('first soft attempt with alive backend returns soft restart', () => {
const decision = decideBootstrapRepair({
attempt: 1,
primaryBackendAlive: true
})
assert.equal(decision.hardReinstall, false)
assert.equal(decision.attempt, 1)
assert.match(decision.reason, /still alive/)
assert.match(decision.reason, /1\/3/)
})
test('first attempt with dead backend still returns soft restart', () => {
const decision = decideBootstrapRepair({
attempt: 1,
primaryBackendAlive: false
})
assert.equal(decision.hardReinstall, false)
assert.match(decision.reason, /has exited/)
})
test('soft restart budget exhausts at maxSoftAttempts+1 and escalates', () => {
const decision = decideBootstrapRepair({
attempt: 4,
maxSoftAttempts: 3,
primaryBackendAlive: true
})
assert.equal(decision.hardReinstall, true)
assert.equal(decision.attempt, 4)
assert.match(decision.reason, /exceeds soft-restart budget/)
})
test('attempt exactly at maxSoftAttempts is still soft', () => {
const decision = decideBootstrapRepair({
attempt: 3,
maxSoftAttempts: 3,
primaryBackendAlive: true
})
assert.equal(decision.hardReinstall, false)
assert.equal(decision.attempt, 3)
})
test('custom maxSoftAttempts is honored', () => {
const soft = decideBootstrapRepair({
attempt: 5,
maxSoftAttempts: 10,
primaryBackendAlive: true
})
assert.equal(soft.hardReinstall, false)
const hard = decideBootstrapRepair({
attempt: 11,
maxSoftAttempts: 10,
primaryBackendAlive: false
})
assert.equal(hard.hardReinstall, true)
})
test('default maxSoftAttempts is 3', () => {
// Probe the default indirectly: attempt 4 with no override must escalate.
const decision = decideBootstrapRepair({
attempt: 4,
primaryBackendAlive: true
})
assert.equal(decision.hardReinstall, true)
})
test('fractional or zero attempts are clamped to 1', () => {
const zeroDecision = decideBootstrapRepair({
attempt: 0,
primaryBackendAlive: true
})
assert.equal(zeroDecision.attempt, 1)
assert.equal(zeroDecision.hardReinstall, false)
const fractionalDecision = decideBootstrapRepair({
attempt: 2.7,
primaryBackendAlive: true
})
assert.equal(fractionalDecision.attempt, 2)
assert.equal(fractionalDecision.hardReinstall, false)
})
test('alive=false on a high attempt number still escalates (defense in depth)', () => {
// A dead backend should normally be handled by the renderer before it
// reaches the repair path, but if it does reach us with a high attempt
// count we still escalate — never silently keep soft-restarting.
const decision = decideBootstrapRepair({
attempt: 5,
maxSoftAttempts: 3,
primaryBackendAlive: false
})
assert.equal(decision.hardReinstall, true)
})
@@ -0,0 +1,121 @@
/**
* Repair-loop guard for the desktop bootstrap.
*
* Why this exists
* ───────────────
* Hermes desktop can request a "repair" of its bundled backend when the
* renderer observes a transient backend failure (see issue #74874). The
* classic failure fingerprint:
*
* 1. Backend Python process hits a transient GIL stall (e.g. heavy
* import, MCP discovery, a long-running agent turn).
* 2. The renderer's WebSocket can't deliver the `gateway.ready` frame
* in time and treats the socket as dead.
* 3. Renderer calls `hermes:bootstrap:repair`.
* 4. Bootstrap unconditionally force-reinstalls the venv, restarting
* the backend — which stalls again for the same reason.
* 5. Renderer reports dead backend → another repair → infinite loop.
*
* The desktop should distinguish:
* - "the venv/install is genuinely broken" → hard reinstall is correct
* - "the runtime is healthy but temporarily stalled" → restart only,
* NOT a destructive reinstall that drops the venv
*
* What this module does
* ─────────────────────
* A pure decision helper. Given the current repair attempt count and a
* hint about whether the live backend process still looks alive, return
* whether the next repair should:
* - `hardReinstall: true` → run the installer, recreate the venv
* - `hardReinstall: false` → restart the existing backend, keep the venv
*
* Cap on soft restarts is bounded so an actually-corrupted install still
* eventually escalates to a hard reinstall after repeated stalls — the
* guard prevents the *unbounded* reinstall loop, not all reinstalls.
*
* The module is intentionally pure (no I/O, no logging, no global state)
* so it is unit-testable in isolation. Wiring into `main.ts` lives there.
*/
export type RepairDecision =
| {
/** Run the installer (recreate venv). Caller bypasses the active runtime. */
hardReinstall: true
/** Human-readable rationale for the desktop log. */
reason: string
/** 1-indexed repair attempt number for diagnostics. */
attempt: number
}
| {
/** Skip the installer; restart the existing backend only. */
hardReinstall: false
reason: string
attempt: number
}
export type RepairDecisionInput = {
/**
* 1-indexed count of how many repair attempts have happened in this
* failure episode. The first repair is `attempt === 1`; a successful
* boot resets the counter (see `main.ts`'s bootstrap completion path).
*/
attempt: number
/**
* Soft-restart budget before escalation to a hard reinstall. Defaults
* to 3: three "just restart" attempts, then a real reinstall. Bounded
* so a corrupt install still gets fixed; high enough that a GIL
* stall no longer loops the user into a 30-minute reinstall cycle.
*/
maxSoftAttempts?: number
/**
* Whether the live backend process (the one we are about to tear down
* to honour the repair request) still looks alive. A process whose
* `exitCode !== null` or `signalCode !== null` has actually exited;
* a process with both null is either still running or stalled — and a
* stall is exactly the case the soft-restart path is for.
*/
primaryBackendAlive: boolean
}
/**
* Decide the next repair action.
*
* Decision matrix:
* attempt ≤ maxSoftAttempts AND alive → soft restart (don't reinstall)
* attempt ≤ maxSoftAttempts AND dead → soft restart (process exited,
* but we don't yet trust that
* the install is corrupt; restart
* once to confirm)
* attempt > maxSoftAttempts → hard reinstall (give up on the
* current install)
*
* "Alive" being true does NOT force a soft restart on every call: the
* attempt counter still increments, so an actually-broken install that
* keeps respawning a child but never announces READY still escalates
* after `maxSoftAttempts` cycles.
*/
export function decideBootstrapRepair(input: RepairDecisionInput): RepairDecision {
const maxSoftAttempts = input.maxSoftAttempts ?? 3
const attempt = Math.max(1, Math.floor(input.attempt))
const alive = Boolean(input.primaryBackendAlive)
if (attempt > maxSoftAttempts) {
return {
hardReinstall: true,
attempt,
reason:
`repair attempt ${attempt} exceeds soft-restart budget ` + `(${maxSoftAttempts}); escalating to hard reinstall`
}
}
return {
hardReinstall: false,
attempt,
reason: alive
? `repair attempt ${attempt}/${maxSoftAttempts}: primary backend process ` +
`still alive (likely transient stall, see #74874); restarting only, ` +
`skipping installer`
: `repair attempt ${attempt}/${maxSoftAttempts}: primary backend process ` +
`has exited; restarting before escalating to reinstall`
}
}
@@ -133,16 +133,47 @@ test('profileRemoteOverride tolerates a missing/!object profiles map', () => {
assert.equal(profileRemoteOverride(null, 'coder'), null)
})
test('SSH remains separate from URL-shaped remote modes', () => {
test('SSH remains separate from URL-shaped remote modes and preserves an explicit remote profile', () => {
assert.equal(modeIsRemoteLike('ssh'), false)
const config = { profiles: { coder: { mode: 'ssh', host: 'alice@box:2222', keyPath: '/key' } } }
const config = {
profiles: { coder: { mode: 'ssh', host: 'alice@box:2222', keyPath: '/key', remoteProfile: 'default' } }
}
assert.equal(profileRemoteOverride(config, 'coder'), null)
assert.deepEqual(profileSshOverride(config, 'coder'), {
mode: 'ssh',
host: 'box',
user: 'alice',
port: 2222,
keyPath: '/key'
keyPath: '/key',
remoteProfile: 'default'
})
})
test('normalizeSshConfig rejects unsafe remote profile mappings', () => {
assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'writer_2' }), {
mode: 'ssh',
host: 'box',
remoteProfile: 'writer_2'
})
assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'bad profile' }), {
mode: 'ssh',
host: 'box'
})
assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: '' }), {
mode: 'ssh',
host: 'box'
})
assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'root' }), {
mode: 'ssh',
host: 'box'
})
assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'default' }), {
mode: 'ssh',
host: 'box',
remoteProfile: 'default'
})
})
@@ -350,6 +381,19 @@ test('normalizeRemoteBaseUrl rejects garbage', () => {
assert.throws(() => normalizeRemoteBaseUrl('not a url'), /not valid/)
})
test('normalizeRemoteBaseUrl auto-prepends http:// for scheme-less host:port input', () => {
assert.equal(normalizeRemoteBaseUrl('100.64.0.1:9119'), 'http://100.64.0.1:9119')
assert.equal(normalizeRemoteBaseUrl('mini.tailnet-1234.ts.net:9119'), 'http://mini.tailnet-1234.ts.net:9119')
assert.equal(normalizeRemoteBaseUrl('localhost:9119'), 'http://localhost:9119')
assert.equal(normalizeRemoteBaseUrl('gw.example.com'), 'http://gw.example.com')
assert.equal(normalizeRemoteBaseUrl('gw.example.com/hermes/'), 'http://gw.example.com/hermes')
})
test('normalizeRemoteBaseUrl still rejects explicit non-http(s) schemes after scheme-less handling', () => {
assert.throws(() => normalizeRemoteBaseUrl('ws://host:9119'), /http:\/\/ or https:\/\//)
assert.throws(() => normalizeRemoteBaseUrl('ftp://host:21'), /http:\/\/ or https:\/\//)
})
// --- buildGatewayWsUrl (token) ---
test('buildGatewayWsUrl uses wss for https and bakes the token', () => {
+24 -1
View File
@@ -45,14 +45,27 @@ const RT_COOKIE_VARIANTS = ['__Host-hermes_session_rt', '__Secure-hermes_session
// cookies above. `privy-token` is the access token (the required signal);
// variants cover the secured-prefix forms and the older `privy-session` name.
const PRIVY_SESSION_COOKIE_VARIANTS = ['__Host-privy-token', '__Secure-privy-token', 'privy-token', 'privy-session']
// Keep this aligned with hermes_cli.profiles.validate_profile_name(). `default`
// is the built-in root alias; these names cannot be created as profiles.
const RESERVED_REMOTE_PROFILES = new Set(['hermes', 'test', 'tmp', 'root', 'sudo'])
function normalizeRemoteBaseUrl(rawUrl) {
const value = String(rawUrl || '').trim()
let value = String(rawUrl || '').trim()
if (!value) {
throw new Error('Remote gateway URL is required.')
}
// Users routinely paste scheme-less "host:port" (a Tailscale IP, a LAN
// hostname). Without this, `new URL('100.64.0.1:9119')` either throws or —
// worse — parses `host:` as the protocol and produces a baffling
// "must be http:// or https://, got myhost:" error. Only a real
// `scheme://` prefix opts out, so explicit non-http schemes (ftp://,
// file://) still reach the protocol check below and get rejected.
if (!/^[a-z][a-z0-9+.-]*:\/\//i.test(value)) {
value = `http://${value}`
}
let parsed
try {
@@ -271,6 +284,16 @@ function normalizeSshConfig(entry) {
out.remoteHermesPath = remoteHermesPath
}
// A Desktop profile can be a local routing label rather than the profile
// name used by the remote Hermes installation. Preserve an explicit mapping
// when it is a valid Hermes profile identifier; otherwise fall back to the
// historical same-name behavior in the caller.
const remoteProfile = String(entry.remoteProfile || '').trim()
if (/^[a-z0-9][a-z0-9_-]{0,63}$/.test(remoteProfile) && !RESERVED_REMOTE_PROFILES.has(remoteProfile)) {
out.remoteProfile = remoteProfile
}
return out
}
+31 -1
View File
@@ -6,7 +6,7 @@ import path from 'node:path'
import { afterEach, test } from 'vitest'
import { gitFor, repoStatus, resolveRenamePath } from './git-review-ops'
import { gitFor, repoStatus, resolveRenamePath, REVIEW_FILE_CAP, reviewList } from './git-review-ops'
const tempDirs: string[] = []
@@ -87,3 +87,33 @@ test('repoStatus reports an untracked directory without recursively listing its
['generated/']
)
})
test('reviewList reports an untracked directory without recursively listing its contents', async () => {
const dir = makeRepo()
const nested = path.join(dir, 'browser-profile', 'Default', 'Cache')
fs.mkdirSync(nested, { recursive: true })
for (let i = 0; i < 20; i++) {
fs.writeFileSync(path.join(nested, `cache-${i}.bin`), 'generated\n')
}
const result = await reviewList(dir, 'uncommitted', null, 'git')
assert.deepEqual(
result.files.map(file => file.path),
['browser-profile/']
)
})
test('reviewList caps the file payload returned to the renderer', async () => {
const dir = makeRepo()
for (let i = 0; i < REVIEW_FILE_CAP + 10; i++) {
fs.writeFileSync(path.join(dir, `untracked-${String(i).padStart(4, '0')}.txt`), 'generated\n')
}
const result = await reviewList(dir, 'uncommitted', null, 'git')
assert.equal(result.files.length, REVIEW_FILE_CAP)
})
+21 -6
View File
@@ -14,6 +14,7 @@ import { resolveRequestedPathForIpc } from './hardening'
const COMMIT_CONTEXT_DIFF_MAX_CHARS = 120_000
const COMMIT_CONTEXT_UNTRACKED_MAX = 80
const REVIEW_FILE_CAP = 2_000
const UNTRACKED_LINE_COUNT_CONCURRENCY = 16
const UNTRACKED_LINE_COUNT_MAX_BYTES = 1024 * 1024
@@ -253,7 +254,7 @@ async function reviewList(repoPath, scope, baseRef, gitBin) {
const range = scope === 'branch' ? `${base}...HEAD` : base
const summary = await git.diffSummary([range])
const files = summary.files.map(file => ({
const files = summary.files.slice(0, REVIEW_FILE_CAP).map(file => ({
path: resolveRenamePath(file.file),
added: 'insertions' in file ? file.insertions : 0,
removed: 'deletions' in file ? file.deletions : 0,
@@ -262,12 +263,22 @@ async function reviewList(repoPath, scope, baseRef, gitBin) {
}))
// "Last turn" also surfaces files created since the baseline (untracked).
if (scope === 'lastTurn') {
const status = await git.status()
if (scope === 'lastTurn' && files.length < REVIEW_FILE_CAP) {
// Keep untracked directories compact. A recursive status can produce
// hundreds of thousands of rows for browser profiles, generated
// artifacts, or dependency trees before the response reaches the
// renderer.
const status = await git.status(['--untracked-files=normal'])
const knownPaths = new Set(files.map(file => file.path))
for (const path of status.not_added) {
if (!files.some(f => f.path === path)) {
if (files.length >= REVIEW_FILE_CAP) {
break
}
if (!knownPaths.has(path)) {
files.push({ path, added: 0, removed: 0, status: '?', staged: false })
knownPaths.add(path)
}
}
}
@@ -280,7 +291,10 @@ async function reviewList(repoPath, scope, baseRef, gitBin) {
// Default: uncommitted (staged + unstaged + untracked), one row per path.
const [status, staged, unstaged] = await Promise.all([
git.status(),
// `normal` reports an untracked directory as one row instead of walking
// every descendant. The result is also capped before per-file stat/read
// work and before crossing the Electron IPC boundary.
git.status(['--untracked-files=normal']),
git.diffSummary(['--cached']),
git.diffSummary([])
])
@@ -288,7 +302,7 @@ async function reviewList(repoPath, scope, baseRef, gitBin) {
const stagedCounts = countsByPath(staged)
const unstagedCounts = countsByPath(unstaged)
const files = status.files.map(file => {
const files = status.files.slice(0, REVIEW_FILE_CAP).map(file => {
const filePath = resolveRenamePath(file.path)
const sc = stagedCounts.get(filePath) || { added: 0, removed: 0 }
const uc = unstagedCounts.get(filePath) || { added: 0, removed: 0 }
@@ -696,6 +710,7 @@ export {
gitFor,
repoStatus,
resolveRenamePath,
REVIEW_FILE_CAP,
reviewCommit,
reviewCommitContext,
reviewCreatePr,
+168 -39
View File
@@ -35,7 +35,7 @@ import { classifyActiveRuntime } from './active-runtime-state'
import { stopBackendChild as stopBackendChildImpl } from './backend-child'
import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command'
import { createBackendConnectionState } from './backend-connection-state'
import { buildDesktopBackendEnv, normalizeHermesHomeRoot } from './backend-env'
import { buildDesktopBackendEnv, hermesManagedNodePathEntries, normalizeHermesHomeRoot } from './backend-env'
import { isReauthRequiredError, waitForHermesReady } from './backend-health'
import {
canImportHermesCli,
@@ -47,6 +47,7 @@ import {
import { waitForDashboardPortAnnouncement } from './backend-ready'
import { shouldLatchBackendStartFailure, shouldLatchRemoteReauthFailure } from './backend-start-failure'
import { detectRemoteDisplay, isWindowsBinaryPathInWsl, isWslEnvironment } from './bootstrap-platform'
import { decideBootstrapRepair } from './bootstrap-repair-guard'
import { runBootstrap } from './bootstrap-runner'
import { applyConnectionChange, resolveTerminalConnection } from './connection-apply'
import {
@@ -188,7 +189,7 @@ import { createStreamThrottle } from './stream-throttle'
import { nativeOverlayWidth as computeNativeOverlayWidth, macTitleBarOverlayHeight } from './titlebar-overlay-width'
import { resolveBehindCount, shouldCountCommits } from './update-count'
import { waitForUpdateClearance } from './update-gate'
import { readLiveUpdateMarker, writeUpdateMarker } from './update-marker'
import { readLiveUpdateMarker, updateHandoffConflict, writeUpdateMarker } from './update-marker'
import { runRebuildWithRetry } from './update-rebuild'
import {
buildRelaunchScript,
@@ -200,9 +201,14 @@ import {
sandboxPreflight
} from './update-relaunch'
import { isOfficialSshRemote, OFFICIAL_REPO_HTTPS_URL } from './update-remote'
import { spawnUpdaterProcess } from './updater-process'
import {
resolveStagedUpdaterBinary,
spawnUpdaterProcess,
stagedUpdaterSupportsPrewrittenMarker
} from './updater-process'
import { formatBlockerMessage, formatProbeFailedMessage, scanVenvBlockers } from './venv-blocker-scan'
import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace'
import { createWakeIndicatorWindowController } from './wake-indicator-window'
import {
computeWindowOptions,
debounce,
@@ -572,19 +578,10 @@ function resolveHermesHome() {
const HERMES_HOME = resolveHermesHome()
function hermesManagedNodePathEntries() {
// NOTE: keep this ordering in sync with iter_hermes_node_dirs() in
// hermes_constants.py — this Node main process cannot import the Python
// module, so the platform-ordering rule is mirrored here.
const root = path.join(HERMES_HOME, 'node')
const bin = path.join(root, 'bin')
const entries = IS_WINDOWS ? [root, bin] : [bin, root]
return entries.filter(directoryExists)
}
function pathWithHermesManagedNode(...entries) {
return [...hermesManagedNodePathEntries(), ...entries, process.env.PATH].filter(Boolean).join(path.delimiter)
const managed = hermesManagedNodePathEntries(HERMES_HOME).filter(directoryExists)
return [...managed, ...entries, process.env.PATH].filter(Boolean).join(path.delimiter)
}
// ACTIVE_HERMES_ROOT — the canonical mutable Hermes install. Same path
@@ -680,7 +677,15 @@ const WINDOW_BUTTON_POSITION = {
// (pure + unit-testable); computeNativeOverlayWidth() applies it per platform.
// It's only the pre-layout fallback — the renderer measures the exact overlay
// width live via the Window Controls Overlay API.
// The apple-touch PNG bakes in the macOS-style ~10% margin, which is correct
// for the dock but renders visibly smaller than neighboring taskbar icons on
// Windows, where icons are full-bleed. Windows prefers the full-bleed
// assets/icon.ico (shipped to resources/ via extraResources) and only falls
// back to the padded PNG if the ico is missing.
const APP_ICON_PATHS = [
...(IS_WINDOWS
? [path.join(process.resourcesPath ?? '', 'icon.ico'), path.join(APP_ROOT, 'assets', 'icon.ico')]
: []),
path.join(APP_ROOT, 'public', 'apple-touch-icon.png'),
path.join(APP_ROOT, 'dist', 'apple-touch-icon.png'),
path.join(unpackedPathFor(APP_ROOT), 'dist', 'apple-touch-icon.png')
@@ -1099,6 +1104,14 @@ let bootstrapAbortController = null
// repair can force the installer without destroying provenance about how the
// install was created. Cleared once the reinstall is under way.
let bootstrapRepairRequested = false
// Counter for in-flight repair attempts. Reset on a clean boot completion
// (see runBootstrap -> ensureRuntime resolve path). Each successive repair
// in the same failure episode increments this; once it crosses
// MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS the guard escalates from "soft restart"
// to "hard reinstall" so a transient backend stall (issue #74874) stops
// looping the user through a destructive venv reinstall.
let bootstrapRepairAttempt = 0
const MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS = 3
let connectionConfigCache = null
let connectionConfigCacheMtime = null
const hermesLog = []
@@ -2602,18 +2615,14 @@ let isQuittingForHandoff = false
let quitPromptOpen = false
let quitConfirmedWithActiveWork = false
// Resolve the staged updater binary. The Tauri installer copies itself to
// HERMES_HOME/hermes-setup.exe on a successful install (see
// apps/bootstrap-installer paths::copy_self_to_hermes_home). That binary owns
// ALL repo mutation — running `hermes update` + rebuilding the desktop — so
// the desktop never touches its own bits while running. Returns null when the
// updater isn't staged (e.g. a dev/source run that never went through the
// installer); callers degrade gracefully.
// Resolve the staged updater binary the desktop may hand an update to. On
// Windows that binary owns ALL repo mutation — running `hermes update` +
// rebuilding the desktop — so the desktop never touches its own bits while
// running. macOS/Linux stage the same binary but deliberately do not use it;
// see resolveStagedUpdaterBinary for the policy and for #74836. Returns null
// whenever no hand-off applies; callers degrade gracefully.
function resolveUpdaterBinary() {
const name = IS_WINDOWS ? 'hermes-setup.exe' : 'hermes-setup'
const candidate = path.join(HERMES_HOME, name)
return fileExists(candidate) ? candidate : null
return resolveStagedUpdaterBinary(HERMES_HOME, { fileExists, isWindows: IS_WINDOWS })
}
function repairMacUpdaterHelper(updater) {
@@ -2841,12 +2850,13 @@ async function applyUpdates(opts = {}) {
const updater = resolveUpdaterBinary()
if (!updater && !IS_WINDOWS) {
// macOS/Linux drag-install: no staged Tauri hermes-setup. Unlike Windows
// (where a venv-shim file lock forces the quit→hand-off→rebuild dance),
// there's no mandatory file locking here, so the desktop can drive the
// whole update itself: `hermes update` (backend) + `hermes desktop
// --build-only` (OS-aware GUI rebuild), then swap the running .app bundle
// with the freshly built one and relaunch.
// macOS/Linux: never hand off, staged hermes-setup or not — the resolver
// returns null there by policy. Unlike Windows (where a venv-shim file
// lock forces the quit→hand-off→rebuild dance), there's no mandatory file
// locking here, so the desktop can drive the whole update itself:
// `hermes update` (backend) + `hermes desktop --build-only` (OS-aware GUI
// rebuild), then swap the running .app bundle with the freshly built one
// and relaunch.
return await applyUpdatesPosixInApp(opts)
}
@@ -2884,6 +2894,19 @@ async function applyUpdates(opts = {}) {
return { ok: true, manual: true, command, hermesRoot: updateRoot }
}
const handoffConflict = updateHandoffConflict(HERMES_HOME)
if (handoffConflict) {
// A different updater already owns the marker — most often a previous
// "Update" click whose updater is still alive and parked mid-run.
// Spawning another here would overwrite its claim and let two updaters
// mutate the checkout at once (#75778); refuse instead.
rememberLog(`[updates] refusing hand-off: ${handoffConflict.message}`)
emitUpdateProgress({ stage: 'error', message: handoffConflict.message, percent: null })
return { ok: false, error: 'update-already-running', message: handoffConflict.message }
}
emitUpdateProgress({
stage: 'restart',
message:
@@ -2984,8 +3007,20 @@ async function applyUpdates(opts = {}) {
// the venv. By writing the marker ourselves the renderer's
// waitForUpdateToFinish() gate sees a live update and parks instead.
// The updater overwrites this with its own PID later; same format.
if (Number.isInteger(child.pid)) {
//
// SKIPPED for pre-#74782 staged updaters: those have no self-PID
// exclusion, so they read this very marker as a foreign live owner and
// abort with "Another Hermes update is already running (PID <itself>)" —
// an unbreakable loop, because the update that would replace the stale
// binary is the one being refused. Losing the anti-respawn hardening is
// strictly better than never updating again, and the updater still writes
// its own marker moments later.
if (Number.isInteger(child.pid) && stagedUpdaterSupportsPrewrittenMarker(updater)) {
writeUpdateMarker(HERMES_HOME, child.pid)
} else if (Number.isInteger(child.pid)) {
rememberLog(
`[updates] skipping marker pre-write: staged updater predates self-adopt (${updater}); it would refuse its own claim`
)
}
rememberLog(`[updates] launched updater: ${updater} ${updaterArgs.join(' ')}; exiting desktop to release venv shim`)
@@ -3017,6 +3052,24 @@ async function handOffWindowsBootstrapRecovery(reason) {
return false
}
const handoffConflict = updateHandoffConflict(HERMES_HOME)
if (handoffConflict) {
// Same hazard as applyUpdates (#75778): a live foreign updater already
// owns the marker. Spawning another here would overwrite its claim and
// race a second updater over the same install tree. The live updater
// is already working on this exact install and will restart us when
// it finishes, so treat this the same as a successful hand-off instead
// of clobbering it with our own.
rememberLog(`[bootstrap] refusing recovery hand-off: ${handoffConflict.message}`)
isQuittingForHandoff = true
setTimeout(() => {
app.quit()
}, UPDATE_HANDOFF_DWELL_MS)
return true
}
const updateRoot = resolveUpdateRoot()
const { branch: configuredBranch } = readDesktopUpdateConfig()
@@ -3054,9 +3107,15 @@ async function handOffWindowsBootstrapRecovery(reason) {
// Same marker pre-write as applyUpdates — see comment there. The recovery
// hand-off has the same window where the renderer can respawn a backend
// before the updater writes its own marker.
if (Number.isInteger(child.pid)) {
// before the updater writes its own marker, and the same stale-updater
// exclusion: a pre-#74782 binary would refuse its own pre-written claim and
// strand the very recovery meant to heal the install.
if (Number.isInteger(child.pid) && stagedUpdaterSupportsPrewrittenMarker(updater)) {
writeUpdateMarker(HERMES_HOME, child.pid)
} else if (Number.isInteger(child.pid)) {
rememberLog(
`[bootstrap] skipping marker pre-write: staged updater predates self-adopt (${updater}); it would refuse its own claim`
)
}
rememberLog(
@@ -4040,6 +4099,7 @@ async function ensureRuntime(backend) {
// The repair request has been honoured by reaching the installer; clear it
// so a later boot isn't forced through bootstrap again.
bootstrapRepairRequested = false
bootstrapRepairAttempt = 0
const bootstrapResult = await runBootstrap({
installStamp: backend.installStamp,
@@ -6949,6 +7009,7 @@ async function sanitizeDesktopConnectionConfig(config = readDesktopConnectionCon
sshPort: (ssh || savedSsh)?.port || null,
sshKeyPath: (ssh || savedSsh)?.keyPath || '',
sshRemoteHermesPath: (ssh || savedSsh)?.remoteHermesPath || '',
sshRemoteProfile: (ssh || savedSsh)?.remoteProfile || '',
// The env override only forces the global/primary connection; a per-profile
// scope is never overridden by HERMES_DESKTOP_REMOTE_URL.
envOverride
@@ -7081,7 +7142,8 @@ function buildSshBlock(input: any, existingBlock: any = {}) {
user: input.sshUser ?? existingBlock.user,
port: input.sshPort ?? existingBlock.port,
keyPath: input.sshKeyPath ?? existingBlock.keyPath,
remoteHermesPath: input.sshRemoteHermesPath ?? existingBlock.remoteHermesPath
remoteHermesPath: input.sshRemoteHermesPath ?? existingBlock.remoteHermesPath,
remoteProfile: input.sshRemoteProfile ?? existingBlock.remoteProfile
})
if (!merged) {
@@ -7370,7 +7432,7 @@ async function bootstrapSshConnectionInner(profile, sshConfig, reuseToken, sourc
const lifecycle = platform.os === 'Windows' ? connectWindowsRemote : remoteLifecycle.connect
result = await lifecycle({
ssh,
profile: connectionScopeKey(profile) || '',
profile: sshConfig.remoteProfile || connectionScopeKey(profile) || '',
remoteHermesPath: sshConfig.remoteHermesPath || '',
ownershipId: sshOwnershipKey(profile),
reuseToken: reuseToken || '',
@@ -8550,6 +8612,13 @@ async function startHermes() {
error: null
})
// A successful boot (including a soft restart that the repair-guard
// chose over a hard reinstall, see #74874) means any in-flight repair
// attempt counter has been honoured — reset it so the next genuine
// failure starts fresh from attempt 1 instead of inheriting the
// accumulated count of the resolved episode.
bootstrapRepairAttempt = 0
return {
baseUrl,
mode: 'local',
@@ -8804,6 +8873,17 @@ function createInstanceWindow() {
return win
}
// A macOS-only ambient wake cue. It is deliberately a gateway-less helper
// window: the active renderer owns voice state and sends only the visual phase.
const wakeIndicatorController = createWakeIndicatorWindowController({
devServer: DEV_SERVER,
isMac: IS_MAC,
loadWindowUrl,
preloadPath: PRELOAD_PATH,
rendererIndex: resolveRendererIndex,
wireWindow: window => wireCommonWindowHandlers(window, zoomWiringForWindowKind('wakeIndicator'))
})
// The pet overlay: a single transparent, frameless, always-on-top window that
// hosts ONLY the floating mascot. Shift-clicking the in-window pet "pops it out"
// here so it can leave the app's bounds and stay visible while Hermes is
@@ -9264,6 +9344,7 @@ function createWindow() {
const createdMainWindow = mainWindow
mainWindow.on('closed', () => {
closePetOverlay()
wakeIndicatorController.close()
if (mainWindow === createdMainWindow) {
mainWindow = null
@@ -9464,6 +9545,10 @@ ipcMain.handle('hermes:window:openInstance', async () => {
return { ok: true }
})
ipcMain.handle('hermes:wake-indicator:get', () => wakeIndicatorController.getState())
ipcMain.on('hermes:wake-indicator:set', (_event, state) => {
wakeIndicatorController.setState(state)
})
// --- Text size (zoom) -------------------------------------------------------
// The settings UI drives the same clamped zoom scale as the Ctrl/Cmd
@@ -9631,9 +9716,41 @@ ipcMain.handle('hermes:bootstrap:repair', async () => {
// transient backend errors on a perfectly healthy install, and deleting the
// marker in that case stranded the app in first-run setup with no way back
// (#72166). The explicit flag carries the intent instead.
rememberLog('[bootstrap] repair requested by renderer; forcing reinstall + clearing latched failure')
bootstrapRepairAttempt += 1
bootstrapRepairRequested = true
// Probe the live backend process so the guard can distinguish "venv is
// genuinely broken" (force reinstall) from "backend is just transiently
// stalled under GIL pressure" (#74874 — `event loop stalled` followed by
// `ws ready frame send failed`, then renderer keeps reporting dead).
const primaryProc = backendConnectionState.getProcess()
const primaryBackendAlive = Boolean(
primaryProc &&
(primaryProc as { exitCode?: number | null }).exitCode === null &&
(primaryProc as { signalCode?: string | null }).signalCode === null
)
const repairDecision = decideBootstrapRepair({
attempt: bootstrapRepairAttempt,
maxSoftAttempts: MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS,
primaryBackendAlive
})
rememberLog(
`[bootstrap] repair requested by renderer; forcing reinstall + clearing latched failure ` +
`(attempt=${repairDecision.attempt}/${MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS}, ` +
`primaryBackendAlive=${primaryBackendAlive}, ` +
`hardReinstall=${repairDecision.hardReinstall}): ${repairDecision.reason}`
)
// The guard may decide the install is healthy enough that a restart
// (without touching the venv) is the right answer. Translate that into
// the existing flag: if the guard said "soft restart", we skip the
// "bypass active runtime" path inside startHermes() and fall through
// to the normal restart branch, which just kills the current child
// and respawns it against the same venv. See #74874 — this is what
// breaks the infinite reinstall loop the user hit.
bootstrapRepairRequested = repairDecision.hardReinstall
bootstrapFailure = null
backendStartFailure = null
remoteReauthFailure = null
@@ -11711,6 +11828,17 @@ app.whenReady().then(() => {
// it without the renderer visiting Settings. A failed registration is logged
// here and surfaced in Settings via the IPC state (never silent).
applyQuickEntrySettings(readQuickEntrySettings())
if (IS_MAC) {
const reposition = () => wakeIndicatorController.reposition()
screen.on('display-added', reposition)
screen.on('display-metrics-changed', reposition)
screen.on('display-removed', reposition)
}
createWindow()
// Win/Linux cold start: the launching hermes:// URL is in our own argv.
@@ -11839,6 +11967,7 @@ app.on('before-quit', event => {
// The always-on-top overlay isn't a "real" app window; close it so a stray
// pet can't keep the process alive or float over a quit app.
closePetOverlay()
wakeIndicatorController.close()
// Same for the Quick Entry composer — and release its global accelerator so a
// quitting Hermes never keeps another app's chord hostage.
+1 -4
View File
@@ -207,10 +207,7 @@ test('parseTokenResponse cannot read a persisted set (the reload bug #73271)', (
})
test('parseStoredTokenSet rejects a non-normalized server response', () => {
assert.throws(
() => parseStoredTokenSet({ access_token: 'AT-server' }),
/missing accessToken/i
)
assert.throws(() => parseStoredTokenSet({ access_token: 'AT-server' }), /missing accessToken/i)
})
// --- refresh timing ---
+10
View File
@@ -8,6 +8,16 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
openSessionWindow: (sessionId, opts) => ipcRenderer.invoke('hermes:window:openSession', sessionId, opts),
openWindow: () => ipcRenderer.invoke('hermes:window:openInstance'),
claimAmbientCue: key => ipcRenderer.invoke('hermes:ambient:claim', key),
wakeIndicator: {
getState: () => ipcRenderer.invoke('hermes:wake-indicator:get'),
setState: state => ipcRenderer.send('hermes:wake-indicator:set', state),
onState: callback => {
const listener = (_event, state) => callback(state)
ipcRenderer.on('hermes:wake-indicator:state', listener)
return () => ipcRenderer.removeListener('hermes:wake-indicator:state', listener)
}
},
petOverlay: {
// Main renderer → main process: window lifecycle + drag. `request` is
// `{ bounds, screen }`; resolves with the screen bounds it actually used.
@@ -10,8 +10,7 @@ export interface PrimaryBackendStartupOptions<Backend, RuntimeBackend, Remote, C
}
export type PrimaryBackendStartupResult<RuntimeBackend, Connection> =
| { kind: 'local'; backend: RuntimeBackend }
| { kind: 'remote'; connection: Connection }
{ kind: 'local'; backend: RuntimeBackend } | { kind: 'remote'; connection: Connection }
export class FirstRunSetupResetError extends Error {
readonly firstRunSetupReset = true
+1 -6
View File
@@ -111,12 +111,7 @@ const ACCELERATOR_PUNCTUATION = new Set([
/** Why a shortcut string was rejected. The renderer maps these to copy. */
export type QuickEntryShortcutError =
| 'empty'
| 'invalid-key'
| 'invalid-modifier'
| 'no-key'
| 'no-modifier'
| 'reserved'
'empty' | 'invalid-key' | 'invalid-modifier' | 'no-key' | 'no-modifier' | 'reserved'
export type QuickEntryShortcutParse = { ok: false; reason: QuickEntryShortcutError } | { accelerator: string; ok: true }
@@ -2,6 +2,7 @@ import assert from 'node:assert/strict'
import { test } from 'vitest'
import { profileSshOverride } from './connection-config'
import {
buildSpawnCommand,
cleanupStale,
@@ -420,6 +421,51 @@ test('connect() spawns fresh when there is no lockfile, adopts the served token'
assert.equal(result.tokenFingerprint, fingerprintToken('the-served-token'))
})
test('managed SSH maps a local scope to a different non-default remote profile', async () => {
const localScope = 'work'
const sshConfig = profileSshOverride(
{
profiles: {
[localScope]: {
mode: 'ssh',
host: 'remote-box',
remoteProfile: 'writer_2'
}
}
},
localScope
)
assert.equal(sshConfig?.remoteProfile, 'writer_2')
const ssh = fakeSsh([
[/uname/, 'Linux\nx86_64'],
[/\[ -x/, 'OK'],
[/cat .*lock\.json/, ''],
[/grep -q ssh-session-token-file/, 'YES\n'],
[/python3 -c/, ''],
[/printf '%s\\n'/, ''],
[/setsid/, '778\n'],
[/kill -0 778/, 'ALIVE'],
[/cat .*\.log/, 'HERMES_BACKEND_READY port=52000\n']
])
await connect(
connectDeps(ssh, {
profile: sshConfig?.remoteProfile,
adoptServedToken: async () => 'mapped-profile-token'
})
)
const spawn = ssh.calls.find(command => /setsid|nohup/.test(command)) || ''
assert.match(spawn, /--profile\b/)
assert.ok(spawn.includes('writer_2'))
assert.match(spawn, /serve\s+--isolated/)
assert.match(spawn, /\.hermes\/desktop-ssh\/[0-9a-f]{32}\/[0-9a-f]{16}\.token/)
assert.ok(!spawn.includes(' work'), 'the local Desktop scope must not become the remote profile')
})
test('connect() reuses a healthy dashboard when fingerprint + probe pass', async () => {
const reuseToken = 'stored-token'
const lock = ownedLock({ tokenFingerprint: fingerprintToken(reuseToken) })
@@ -440,6 +486,36 @@ test('connect() reuses a healthy dashboard when fingerprint + probe pass', async
assert.ok(!ssh.calls.some(c => /setsid/.test(c)), 'reuse path must not spawn a new dashboard')
})
test('connect() respawns when the requested remote profile differs from the lockfile profile', async () => {
const reuseToken = 'stored-token'
const lock = ownedLock({ profile: 'desktop-work', tokenFingerprint: fingerprintToken(reuseToken) })
const ssh = fakeSsh([
[/uname/, 'Linux\nx86_64'],
[/\[ -x/, 'OK'],
[/cat .*lock\.json/, JSON.stringify(lock)],
[/kill -0 333/, 'ALIVE'],
[/print\("OWNED"/, 'OWNED\n'],
[/kill 333/, ''],
[/--version/, 'Hermes Agent v0.18.2\n'],
[/grep -q ssh-session-token-file/, 'YES\n'],
[/python3 -c/, ''],
[/setsid/, '890\n'],
[/kill -0 890/, 'ALIVE'],
[/cat .*\.log/, 'HERMES_DASHBOARD_READY port=52050\n']
])
const result = await connect(
connectDeps(ssh, { profile: 'default', reuseToken, adoptServedToken: async () => 'fresh' })
)
assert.equal(result.reused, false)
assert.ok(
ssh.calls.some(c => /setsid/.test(c)),
'profile mismatch must spawn a fresh dashboard'
)
})
test('connect() respawns when the lockfile hermesPath differs from the resolved path', async () => {
const reuseToken = 'stored-token'
const lock = ownedLock({ hermesPath: '/old/stale/hermes', tokenFingerprint: fingerprintToken(reuseToken) })
@@ -713,6 +713,7 @@ async function connect(deps) {
pidAlive &&
owned &&
lock.port > 0 &&
lock.profile === profile &&
Boolean(reuseToken) &&
lock.tokenFingerprint === fingerprintToken(reuseToken) &&
lock.hermesPath === hermesPath &&
@@ -28,6 +28,7 @@ test('sshConfigFingerprint covers scope and every connection field', () => {
port: 2222,
keyPath: '/other',
remoteHermesPath: '/other-hermes',
remoteProfile: 'default',
effectiveConfigFingerprint: 'changed-config'
})) {
assert.notEqual(base, sshConfigFingerprint('', { ...config, [field]: value }))
@@ -8,6 +8,7 @@ function sshConfigFingerprint(scope, config) {
config.port,
config.keyPath,
config.remoteHermesPath,
config.remoteProfile,
config.effectiveConfigFingerprint
]
@@ -24,6 +24,7 @@ import {
markerPath,
readLiveUpdateMarker,
UPDATE_MARKER_MAX_AGE_MS,
updateHandoffConflict,
writeUpdateMarker
} from './update-marker'
@@ -128,3 +129,51 @@ test('writeUpdateMarker + dead pid => self-heals on read', () => {
assert.equal(res, null, 'a dead-pid marker from writeUpdateMarker self-heals')
assert.ok(!fs.existsSync(markerPath(home)), 'marker file is pruned')
})
// ---------------------------------------------------------------------------
// updateHandoffConflict (#75778)
//
// A retried "Update" click must not spawn a second updater over a still-live
// one — writeUpdateMarker unconditionally overwrites the marker, so an
// unchecked hand-off clobbers the original updater's claim while it is still
// alive and mutating the checkout.
// ---------------------------------------------------------------------------
test('no marker => hand-off is not blocked', () => {
const home = tmpHome('conflict-none')
assert.equal(updateHandoffConflict(home, { kill: ALIVE }), null)
})
test('a different live updater already owns the marker => hand-off is blocked', () => {
const home = tmpHome('conflict-live')
const now = 1_000_000_000_000
writeMarker(home, 1010, Math.floor(now / 1000) - 6) // 6s old
const conflict = updateHandoffConflict(home, { kill: ALIVE, now: () => now })
assert.ok(conflict, 'a live foreign updater must block a new hand-off')
assert.equal(conflict.pid, 1010)
assert.match(conflict.message, /already running/)
assert.match(conflict.message, /PID 1010/)
assert.match(conflict.message, /6s/)
})
test('a dead-pid marker does not block a hand-off (self-heals)', () => {
const home = tmpHome('conflict-dead')
writeMarker(home, 999999, Math.floor(Date.now() / 1000))
assert.equal(updateHandoffConflict(home, { kill: DEAD }), null)
})
test('an expired marker does not block a hand-off (self-heals)', () => {
const home = tmpHome('conflict-expired')
const now = 1_000_000_000_000
writeMarker(home, 1010, Math.floor((now - UPDATE_MARKER_MAX_AGE_MS - 60_000) / 1000))
assert.equal(updateHandoffConflict(home, { kill: ALIVE, now: () => now }), null)
})
test('minutes-scale elapsed time is formatted as "Nm Ss"', () => {
const home = tmpHome('conflict-minutes')
const now = 1_000_000_000_000
writeMarker(home, 1010, Math.floor(now / 1000) - 125) // 2m 5s old
const conflict = updateHandoffConflict(home, { kill: ALIVE, now: () => now })
assert.ok(conflict)
assert.match(conflict.message, /2m 5s/)
})
+44
View File
@@ -136,3 +136,47 @@ export function writeUpdateMarker(hermesHome, pid, { now = Date.now } = {}) {
// updater will write its own when it reaches run_update.
}
}
/**
* Whether a NEW updater hand-off must be refused because a different,
* already-alive updater currently owns the marker (#75778).
*
* `writeUpdateMarker` unconditionally overwrites the marker file. Called
* before every hand-off with no conflict check, a user who clicks "Update"
* again while a prior updater is still parked mid-run (e.g. "waiting for
* Hermes to exit…") clobbers that still-running updater's claim: the
* retry's pre-write now names the NEW child, so the OLD process — alive
* and mutating the checkout — is no longer recorded as the owner. A second
* live updater can then run over the same tree unrecorded, the exact
* two-updaters-at-once hazard `UpdateMarkerGuard` in the Rust updater
* exists to prevent (apps/bootstrap-installer/src-tauri/src/update.rs).
*
* Returns the live foreign owner (with a ready-to-show message) when the
* hand-off must be refused, or `null` when it's safe to spawn — no marker,
* or the existing one is stale/dead and self-heals via
* `readLiveUpdateMarker`.
*/
export function updateHandoffConflict(
hermesHome,
opts: {
now?: () => number
maxAgeMs?: number
kill?: typeof process.kill
} = {}
) {
const owner = readLiveUpdateMarker(hermesHome, opts)
if (!owner) {
return null
}
const mins = Math.floor(owner.ageMs / 60_000)
const secs = Math.floor((owner.ageMs % 60_000) / 1000)
const elapsed = mins > 0 ? `${mins}m ${secs}s` : `${secs}s`
return {
pid: owner.pid,
ageMs: owner.ageMs,
message: `An update is already running (PID ${owner.pid}, started ${elapsed} ago). Wait for it to finish, then try again.`
}
}
+106 -1
View File
@@ -1,9 +1,68 @@
import assert from 'node:assert/strict'
import type { SpawnOptions } from 'node:child_process'
import path from 'node:path'
import { test } from 'vitest'
import { spawnUpdaterProcess } from './updater-process'
import {
MARKER_SELF_ADOPT_EPOCH_MS,
resolveStagedUpdaterBinary,
spawnUpdaterProcess,
stagedUpdaterSupportsPrewrittenMarker
} from './updater-process'
const DAY_MS = 24 * 60 * 60 * 1000
test('stagedUpdaterSupportsPrewrittenMarker rejects installers predating the self-adopt fix', () => {
// The real-world trap: an installer staged at first install months ago, never
// refreshed because copy_self_to_hermes_home no-ops during --update.
assert.equal(
stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', {
stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS - 60 * DAY_MS
}),
false
)
})
test('stagedUpdaterSupportsPrewrittenMarker accepts installers from the fix onward', () => {
assert.equal(
stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', {
stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS
}),
true
)
assert.equal(
stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', {
stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS + 30 * DAY_MS
}),
true
)
})
test('stagedUpdaterSupportsPrewrittenMarker treats an unreadable mtime as unsupported', () => {
// Bias toward the path that can always make progress: a skipped pre-write
// loses anti-respawn hardening, a wedged updater can never update again.
assert.equal(
stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', {
stagedMtimeMs: () => null
}),
false
)
})
test('resolveStagedUpdaterBinary still returns a stale staged updater on Windows', () => {
// Staleness gates only the marker PRE-WRITE, never the hand-off itself:
// the stale binary is the only updater these users have, and it works fine
// once it is allowed to write its own claim.
assert.equal(
resolveStagedUpdaterBinary('C:\\Hermes', {
fileExists: () => true,
isWindows: true,
stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS - 60 * DAY_MS
}),
path.join('C:\\Hermes', 'hermes-setup.exe')
)
})
test('spawnUpdaterProcess hides the updater console and detaches the child on Windows', () => {
const calls: Array<{ args: string[]; command: string; options: SpawnOptions }> = []
@@ -60,3 +119,49 @@ test('spawnUpdaterProcess preserves updater options off Windows', () => {
assert.deepEqual(capturedOptions, { detached: true, stdio: 'ignore' })
})
test('resolveStagedUpdaterBinary hands Windows the staged installer it finds', () => {
const home = 'C:\\Users\\hermes\\AppData\\Local\\hermes'
const staged = path.join(home, 'hermes-setup.exe')
const probed: string[] = []
const resolved = resolveStagedUpdaterBinary(home, {
fileExists: candidate => {
probed.push(candidate)
return candidate === staged
},
isWindows: true
})
assert.equal(resolved, staged)
assert.deepEqual(probed, [staged])
})
test('resolveStagedUpdaterBinary returns null off Windows even when hermes-setup is staged (#74836)', () => {
const home = '/Users/hermes/.hermes'
let probes = 0
const resolved = resolveStagedUpdaterBinary(home, {
// The installer stages hermes-setup on macOS/Linux too, so "it exists" is
// the normal case — and precisely the one that must not win.
fileExists: () => {
probes += 1
return true
},
isWindows: false
})
assert.equal(resolved, null)
assert.equal(probes, 0)
})
test('resolveStagedUpdaterBinary returns null on Windows when nothing is staged', () => {
const resolved = resolveStagedUpdaterBinary('C:\\Users\\hermes\\AppData\\Local\\hermes', {
fileExists: () => false,
isWindows: true
})
assert.equal(resolved, null)
})
+105
View File
@@ -1,4 +1,6 @@
import { spawn, type SpawnOptions } from 'node:child_process'
import { statSync } from 'node:fs'
import path from 'node:path'
import { hiddenWindowsChildOptions } from './windows-child-options'
@@ -7,6 +9,109 @@ export interface UpdaterChild {
unref: () => void
}
export interface ResolveStagedUpdaterBinaryDeps {
isWindows?: boolean
fileExists?: (candidate: string) => boolean
stagedMtimeMs?: (candidate: string) => number | null
}
/**
* Staged installers older than this have no self-PID exclusion in
* `UpdateMarkerGuard::acquire` and will refuse an update whose marker was
* pre-written on their behalf.
*
* The self-adopt fix landed in #74782 / 160586ff8 (2026-07-30 17:57 +0700).
* We compare against the start of 2026-07-31 UTC so the boundary is
* unambiguous for binaries staged that same day.
*/
export const MARKER_SELF_ADOPT_EPOCH_MS = Date.UTC(2026, 6, 31)
function stagedFileExists(candidate: string): boolean {
try {
return statSync(candidate).isFile()
} catch {
return false
}
}
function stagedFileMtimeMs(candidate: string): number | null {
try {
return statSync(candidate).mtimeMs
} catch {
return null
}
}
/**
* Decide which staged installer binary — if any — may be handed an update.
*
* The Tauri installer self-copies into HERMES_HOME on *every* platform
* (`hermes-setup.exe` on Windows, `hermes-setup` elsewhere — see
* apps/bootstrap-installer `paths::installer_dest` and
* `bootstrap::copy_self_to_hermes_home`), so finding that binary on macOS or
* Linux is expected, not leftover junk.
*
* Handing an update to it is nonetheless a Windows-only policy. Windows needs
* the quit -> hand-off -> rebuild dance because a venv shim file lock keeps the
* running desktop from rewriting its own bits; macOS and Linux have no such
* lock and update in place through applyUpdatesPosixInApp(). Off Windows the
* hand-off therefore buys nothing and costs a great deal: a staged binary older
* than the hand-off protocol holds the update marker, spawns `hermes update`,
* and that child refuses its own parent — wedging the in-app Update button for
* good, with no route (update, re-download, reinstall) to a newer binary
* (#74836). Returning null off Windows is what routes those platforms to the
* in-app updater.
*
* Null on Windows too when nothing is staged (a dev/source run, or a CLI
* install that never went through the installer); callers degrade gracefully.
*/
export function resolveStagedUpdaterBinary(
hermesHome: string,
deps: ResolveStagedUpdaterBinaryDeps = {}
): string | null {
const isWindows = deps.isWindows ?? process.platform === 'win32'
if (!isWindows) {
return null
}
const fileExists = deps.fileExists ?? stagedFileExists
const candidate = path.join(hermesHome, 'hermes-setup.exe')
return fileExists(candidate) ? candidate : null
}
/**
* True when the staged installer is new enough to survive a pre-written marker.
*
* `copy_self_to_hermes_home` deliberately no-ops during `--update`
* (apps/bootstrap-installer/src-tauri/src/paths.rs), so the binary staged by a
* user's ORIGINAL install orchestrates every later update — forever. Installers
* predating #74782 have no self-PID exclusion in `UpdateMarkerGuard::acquire`,
* so when the desktop pre-writes the marker naming that very updater, the
* updater reads its own claim as a foreign live owner and aborts with
* "Another Hermes update is already running (PID <itself>, started 1s ago)" —
* the observed infinite "Install didn't finish" loop. Skipping the pre-write
* for those binaries lets them acquire cleanly and run `hermes update`, which
* pulls the permanent fixes. See shouldPrewriteUpdateMarker.
*
* We cannot ask the binary its version without executing it, so use its mtime:
* the installer is written to HERMES_HOME at install/repair time, making mtime
* a faithful stamp of which installer generation produced it.
*
* Unreadable mtime counts as UNSUPPORTED — the pre-write is a best-effort
* hardening, while a wedged updater is unrecoverable, so we bias toward the
* path that can always make progress.
*/
export function stagedUpdaterSupportsPrewrittenMarker(
candidate: string,
deps: ResolveStagedUpdaterBinaryDeps = {}
): boolean {
const mtimeMs = (deps.stagedMtimeMs ?? stagedFileMtimeMs)(candidate)
return typeof mtimeMs === 'number' && Number.isFinite(mtimeMs) && mtimeMs >= MARKER_SELF_ADOPT_EPOCH_MS
}
export interface SpawnUpdaterProcessDeps {
isWindows?: boolean
spawnProcess?: (command: string, args: string[], options: SpawnOptions) => UpdaterChild
@@ -0,0 +1,174 @@
import { pathToFileURL } from 'node:url'
import { BrowserWindow, screen } from 'electron'
import {
normalizeWakeIndicatorState,
selectWakeIndicatorDisplay,
WAKE_INDICATOR_FADE_MS,
type WakeIndicatorState,
wakeIndicatorWindowBounds
} from './wake-indicator'
interface WakeIndicatorWindowOptions {
devServer?: string
isMac: boolean
loadWindowUrl: (window: BrowserWindow, url: string, label: string) => void
preloadPath: string
rendererIndex: () => string
wireWindow: (window: BrowserWindow) => void
}
export function createWakeIndicatorWindowController({
devServer,
isMac,
loadWindowUrl,
preloadPath,
rendererIndex,
wireWindow
}: WakeIndicatorWindowOptions) {
let hideTimer: NodeJS.Timeout | null = null
let state: WakeIndicatorState = 'hidden'
let window: BrowserWindow | null = null
const url = () => {
if (devServer) {
return `${devServer.endsWith('/') ? devServer.slice(0, -1) : devServer}/?win=wake#/`
}
return `${pathToFileURL(rendererIndex()).toString()}?win=wake#/`
}
const selectedDisplay = () => selectWakeIndicatorDisplay(screen.getAllDisplays(), screen.getPrimaryDisplay())
const reposition = () => {
if (!window || window.isDestroyed()) {
return
}
window.setBounds(wakeIndicatorWindowBounds(selectedDisplay()))
}
const sendState = () => {
if (!window || window.isDestroyed()) {
return
}
window.webContents.send('hermes:wake-indicator:state', state)
}
const spawn = () => {
const next = new BrowserWindow({
...wakeIndicatorWindowBounds(selectedDisplay()),
alwaysOnTop: true,
backgroundColor: '#00000000',
focusable: false,
frame: false,
fullscreenable: false,
hasShadow: false,
hiddenInMissionControl: true,
maximizable: false,
minimizable: false,
movable: false,
resizable: false,
show: false,
skipTaskbar: false,
transparent: true,
type: 'panel',
webPreferences: {
backgroundThrottling: false,
contextIsolation: true,
devTools: true,
nodeIntegration: false,
preload: preloadPath,
sandbox: true
}
})
next.setAlwaysOnTop(true, 'floating')
next.setHiddenInMissionControl?.(true)
next.setIgnoreMouseEvents(true, { forward: true })
try {
next.setVisibleOnAllWorkspaces(true, {
skipTransformProcessType: true,
visibleOnFullScreen: true
})
} catch {
// Best effort on older Electron/macOS combinations.
}
wireWindow(next)
next.webContents.on('did-finish-load', sendState)
next.once('ready-to-show', () => {
if (!next.isDestroyed() && state !== 'hidden') {
next.showInactive()
}
})
next.on('closed', () => {
if (window === next) {
window = null
}
})
loadWindowUrl(next, url(), 'Wake indicator')
return next
}
const setState = (value: unknown) => {
if (!isMac) {
return
}
state = normalizeWakeIndicatorState(value)
if (hideTimer) {
clearTimeout(hideTimer)
hideTimer = null
}
if (state === 'hidden') {
sendState()
hideTimer = setTimeout(() => {
hideTimer = null
if (state === 'hidden' && window && !window.isDestroyed()) {
window.hide()
}
}, WAKE_INDICATOR_FADE_MS)
return
}
if (!window || window.isDestroyed()) {
window = spawn()
} else {
reposition()
sendState()
window.showInactive()
}
}
const close = () => {
if (hideTimer) {
clearTimeout(hideTimer)
hideTimer = null
}
if (window && !window.isDestroyed()) {
window.close()
}
window = null
state = 'hidden'
}
return {
close,
getState: () => state,
reposition,
setState
}
}
@@ -0,0 +1,143 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { normalizeWakeIndicatorState, selectWakeIndicatorDisplay, wakeIndicatorWindowBounds } from './wake-indicator'
import { createWakeIndicatorWindowController } from './wake-indicator-window'
const electronMock = vi.hoisted(() => {
type Listener = () => void
class FakeBrowserWindow {
private destroyed = false
private readonly listeners = new Map<string, Listener[]>()
readonly close = vi.fn(() => {
this.destroyed = true
this.emit('closed')
})
readonly hide = vi.fn()
readonly setAlwaysOnTop = vi.fn()
readonly setBounds = vi.fn()
readonly setHiddenInMissionControl = vi.fn()
readonly setIgnoreMouseEvents = vi.fn()
readonly setVisibleOnAllWorkspaces = vi.fn()
readonly showInactive = vi.fn()
readonly webContents = {
on: vi.fn(),
send: vi.fn()
}
constructor(readonly options: unknown) {
windows.push(this)
}
isDestroyed() {
return this.destroyed
}
on(event: string, listener: Listener) {
const listeners = this.listeners.get(event) ?? []
listeners.push(listener)
this.listeners.set(event, listeners)
return this
}
once(event: string, listener: Listener) {
return this.on(event, listener)
}
private emit(event: string) {
for (const listener of this.listeners.get(event) ?? []) {
listener()
}
}
}
const windows: FakeBrowserWindow[] = []
const display = {
bounds: { height: 982, width: 1512, x: 0, y: 0 },
id: 'internal',
internal: true
}
return {
BrowserWindow: FakeBrowserWindow,
screen: {
getAllDisplays: () => [display],
getPrimaryDisplay: () => display
},
windows
}
})
vi.mock('electron', () => ({
BrowserWindow: electronMock.BrowserWindow,
screen: electronMock.screen
}))
beforeEach(() => {
electronMock.windows.length = 0
})
describe('wake indicator window', () => {
it('centers the helper window at the top of the selected display', () => {
expect(
wakeIndicatorWindowBounds({
bounds: { height: 982, width: 1512, x: -120, y: 40 }
})
).toEqual({
height: 52,
width: 176,
x: 548,
y: 40
})
})
it('prefers the internal display and falls back to the primary display', () => {
const primary = {
bounds: { height: 1080, width: 1920, x: 0, y: 0 },
id: 'primary',
internal: false
}
const internal = {
bounds: { height: 982, width: 1512, x: 1920, y: 0 },
id: 'internal',
internal: true
}
expect(selectWakeIndicatorDisplay([primary, internal], primary)).toBe(internal)
expect(selectWakeIndicatorDisplay([primary], primary)).toBe(primary)
})
it('rejects unknown renderer states', () => {
expect(normalizeWakeIndicatorState('capturing')).toBe('capturing')
expect(normalizeWakeIndicatorState('other')).toBe('hidden')
expect(normalizeWakeIndicatorState(null)).toBe('hidden')
})
})
describe('wake indicator window controller', () => {
it('closes an active helper window and resets its state', () => {
const controller = createWakeIndicatorWindowController({
isMac: true,
loadWindowUrl: vi.fn(),
preloadPath: '/tmp/preload.cjs',
rendererIndex: () => '/tmp/index.html',
wireWindow: vi.fn()
})
controller.setState('detected')
expect(electronMock.windows).toHaveLength(1)
expect(controller.getState()).toBe('detected')
const [window] = electronMock.windows
controller.close()
expect(window.close).toHaveBeenCalledOnce()
expect(controller.getState()).toBe('hidden')
})
})
+34
View File
@@ -0,0 +1,34 @@
export const WAKE_INDICATOR_WINDOW_WIDTH = 176
export const WAKE_INDICATOR_WINDOW_HEIGHT = 52
export const WAKE_INDICATOR_FADE_MS = 500
export const WAKE_INDICATOR_STATES = ['hidden', 'detected', 'capturing'] as const
export type WakeIndicatorState = (typeof WAKE_INDICATOR_STATES)[number]
interface DisplayLike {
bounds: {
height: number
width: number
x: number
y: number
}
internal?: boolean
}
export function normalizeWakeIndicatorState(value: unknown): WakeIndicatorState {
return WAKE_INDICATOR_STATES.includes(value as WakeIndicatorState) ? (value as WakeIndicatorState) : 'hidden'
}
export function selectWakeIndicatorDisplay<T extends DisplayLike>(displays: T[], primary: T): T {
return displays.find(display => display.internal === true) ?? primary
}
export function wakeIndicatorWindowBounds(display: DisplayLike) {
return {
height: WAKE_INDICATOR_WINDOW_HEIGHT,
width: WAKE_INDICATOR_WINDOW_WIDTH,
x: Math.round(display.bounds.x + (display.bounds.width - WAKE_INDICATOR_WINDOW_WIDTH) / 2),
y: Math.round(display.bounds.y)
}
}
@@ -1,4 +1,5 @@
import assert from 'node:assert/strict'
import crypto from 'node:crypto'
import { test } from 'vitest'
@@ -9,6 +10,7 @@ import {
helperCommand,
powerShellCommand,
psLiteral,
reusableWindowsLock,
validLock
} from './windows-remote-lifecycle'
@@ -114,6 +116,31 @@ test('Windows lock validation is scoped and exact', () => {
assert.equal(validLock({ ...lock, port: -1 }, ownershipId), false)
})
test('Windows SSH reuse requires the requested remote profile to match the lock', () => {
const token = 'stored-token'
const lock = {
schemaVersion: 2,
protocolVersion: 1,
ownershipId,
spawnNonce: '0123456789abcdef',
pid: 10,
creationTimeNs: '1784219690452757504',
port: 1234,
profile: 'default',
tokenFingerprint: crypto.createHash('sha256').update(token).digest('hex').slice(0, 32),
hermesPath: 'C:\\h\\hermes.exe',
hermesHome: 'C:\\h'
}
const state = { alive: true, owned: true }
const runtime = { hermesPath: lock.hermesPath, hermesHome: lock.hermesHome }
assert.equal(reusableWindowsLock(lock, state, 'default', token, runtime), true)
assert.equal(reusableWindowsLock(lock, state, 'desktop-work', token, runtime), false)
assert.equal(reusableWindowsLock({ ...lock, profile: '' }, state, '', token, runtime), true)
})
test('Windows integrated terminal uses encoded PowerShell and preserves cwd as literal data', () => {
const command = buildWindowsInteractiveCommand("C:\\Users\\O'Brien\\repo")
const script = Buffer.from(command.split(' ').pop()!, 'base64').toString('utf16le')
@@ -149,6 +149,19 @@ function validLock(lock, ownershipId) {
)
}
function reusableWindowsLock(lock, state, profile, reuseToken, runtime) {
return Boolean(
state.alive &&
state.owned &&
lock.port > 0 &&
lock.profile === profile &&
reuseToken &&
lock.tokenFingerprint === fingerprintToken(reuseToken) &&
lock.hermesPath === runtime.hermesPath &&
lock.hermesHome === runtime.hermesHome
)
}
function assertCurrent(signal) {
if (signal?.aborted) {
const error: any = new Error('SSH bootstrap was cancelled.')
@@ -299,14 +312,7 @@ async function connectWindowsRemote(deps) {
throw error
}
const reusable =
state.alive &&
state.owned &&
lock.port > 0 &&
Boolean(reuseToken) &&
lock.tokenFingerprint === fingerprintToken(reuseToken) &&
lock.hermesPath === runtime.hermesPath &&
lock.hermesHome === runtime.hermesHome
const reusable = reusableWindowsLock(lock, state, profile, reuseToken, runtime)
if (reusable) {
const localPort = await pickLocalPort()
@@ -451,5 +457,6 @@ export {
powerShellCommand,
probeWindowsRemote,
psLiteral,
reusableWindowsLock,
validLock
}
+6 -2
View File
@@ -162,8 +162,8 @@ test('installZoomReassertOnWindowEvents skips destroyed windows', () => {
assert.equal(calls, 0)
})
// Zoom-wiring contract: chat windows keep global UI zoom, the pet overlay
// opts out. Tested via the extracted config — no source-text regex.
// Zoom-wiring contract: chat windows keep global UI zoom while fixed-size
// helper windows opt out. Tested via the extracted config — no source-text regex.
test('chat windows opt into zoom', () => {
assert.deepEqual(zoomWiringForWindowKind('chat'), { zoom: true })
})
@@ -172,6 +172,10 @@ test('pet overlay opts out of zoom', () => {
assert.deepEqual(zoomWiringForWindowKind('petOverlay'), { zoom: false })
})
test('wake indicator opts out of zoom', () => {
assert.deepEqual(zoomWiringForWindowKind('wakeIndicator'), { zoom: false })
})
test('unknown window kinds default to chat (zoom enabled)', () => {
assert.deepEqual(zoomWiringForWindowKind('unknown'), { zoom: true })
assert.deepEqual(zoomWiringForWindowKind(undefined), { zoom: true })
+2 -1
View File
@@ -108,7 +108,8 @@ export function installZoomReassertOnWindowEvents(win, reassert, platform = proc
export const ZOOM_WINDOW_CONFIG = {
chat: { zoom: true },
petOverlay: { zoom: false },
quickEntry: { zoom: false }
quickEntry: { zoom: false },
wakeIndicator: { zoom: false }
} as const
export function zoomWiringForWindowKind(kind) {
+93 -102
View File
@@ -8,13 +8,13 @@
"type": "module",
"main": "dist/electron-main.mjs",
"engines": {
"node": "^20.19.0 || >=22.12.0"
"node": ">=22.22.0"
},
"scripts": {
"clean": "npm run clean:e2e && npm run clean:renderer && npm run clean:electron",
"clean:e2e":"tsc --build tsconfig.e2e.json --clean",
"clean:renderer":"tsc --build tsconfig.json --clean ",
"clean:electron":"tsc --build tsconfig.electron.json --clean",
"clean:e2e": "tsc --build tsconfig.e2e.json --clean",
"clean:renderer": "tsc --build tsconfig.json --clean ",
"clean:electron": "tsc --build tsconfig.electron.json --clean",
"dev": "concurrently -k \"npm:dev:renderer\" \"npm:dev:electron\"",
"dev:fake-boot": "cross-env HERMES_DESKTOP_BOOT_FAKE=1 HERMES_DESKTOP_BOOT_FAKE_STEP_MS=650 npm run dev",
"dev:mock": "node scripts/dev-mock.mjs",
@@ -64,110 +64,101 @@
"test:e2e:update-snapshots": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list --update-snapshots"
},
"dependencies": {
"@assistant-ui/react": "^0.14.23",
"@assistant-ui/react-streamdown": "^0.3.4",
"@audiowave/react": "^0.6.2",
"@chenglou/pretext": "^0.0.6",
"@codemirror/commands": "^6.10.4",
"@codemirror/language": "^6.12.4",
"@codemirror/language-data": "^6.5.2",
"@codemirror/state": "^6.7.0",
"@codemirror/view": "^6.43.3",
"@dnd-kit/core": "^6.3.1",
"@dnd-kit/sortable": "^10.0.0",
"@dnd-kit/utilities": "^3.2.2",
"@assistant-ui/core": "0.2.23",
"@assistant-ui/react": "0.14.24",
"@assistant-ui/react-streamdown": "0.3.5",
"@audiowave/react": "0.6.2",
"@chenglou/pretext": "0.0.6",
"@codemirror/commands": "6.10.4",
"@codemirror/language": "6.12.4",
"@codemirror/language-data": "6.5.2",
"@codemirror/state": "6.7.1",
"@codemirror/view": "6.43.6",
"@dnd-kit/core": "6.3.1",
"@dnd-kit/sortable": "10.0.0",
"@dnd-kit/utilities": "3.2.2",
"@hermes/shared": "file:../shared",
"@icons-pack/react-simple-icons": "=13.11.1",
"@lezer/highlight": "^1.2.3",
"@nanostores/react": "^1.1.0",
"@nous-research/ui": "^0.13.0",
"@radix-ui/react-slot": "^1.2.4",
"@streamdown/code": "^1.1.1",
"@tabler/icons-react": "^3.41.1",
"@tailwindcss/typography": "^0.5.19",
"@tailwindcss/vite": "^4.2.4",
"@tanstack/react-query": "^5.100.6",
"@tanstack/react-virtual": "^3.13.24",
"@vscode/codicons": "^0.0.45",
"@xterm/addon-fit": "^0.11.0",
"@xterm/addon-serialize": "^0.14.0",
"@xterm/addon-unicode11": "^0.9.0",
"@xterm/addon-web-links": "^0.12.0",
"@xterm/addon-webgl": "^0.19.0",
"@xterm/xterm": "^6.0.0",
"class-variance-authority": "^0.7.1",
"clsx": "^2.1.1",
"cmdk": "^1.1.1",
"d3-force": "^3.0.0",
"dnd-core": "^14.0.1",
"dompurify": "^3.4.11",
"emojibase-data": "^16.0.3",
"fflate": "^0.8.3",
"frimousse": "^0.3.0",
"hast-util-from-html-isomorphic": "^2.0.0",
"hast-util-to-text": "^4.0.2",
"ignore": "^7.0.5",
"katex": "^0.16.45",
"mermaid": "^11.15.0",
"motion": "^12.38.0",
"nanostores": "^1.3.0",
"@icons-pack/react-simple-icons": "13.11.1",
"@lezer/highlight": "1.2.3",
"@nanostores/react": "1.1.0",
"@nous-research/ui": "0.18.2",
"@streamdown/code": "1.1.1",
"@tabler/icons-react": "3.44.0",
"@tailwindcss/typography": "0.5.20",
"@tailwindcss/vite": "4.3.3",
"@tanstack/react-query": "5.101.2",
"@tanstack/react-virtual": "3.14.6",
"@vscode/codicons": "0.0.45",
"@xterm/addon-fit": "0.11.0",
"@xterm/addon-serialize": "0.14.0",
"@xterm/addon-unicode11": "0.9.0",
"@xterm/addon-web-links": "0.12.0",
"@xterm/addon-webgl": "0.19.0",
"@xterm/xterm": "6.0.0",
"class-variance-authority": "0.7.1",
"clsx": "2.1.1",
"cmdk": "1.1.1",
"d3-force": "3.0.0",
"dnd-core": "14.0.1",
"dompurify": "3.4.12",
"emojibase-data": "16.0.3",
"fflate": "0.8.3",
"frimousse": "0.3.0",
"hast-util-from-html-isomorphic": "2.0.0",
"hast-util-to-text": "4.0.2",
"ignore": "7.0.6",
"katex": "0.16.47",
"mermaid": "11.16.0",
"motion": "12.42.2",
"nanostores": "1.4.0",
"node-pty": "1.1.0",
"radix-ui": "^1.6.5",
"react": "^19.2.5",
"react-arborist": "^3.5.0",
"react-dnd-html5-backend": "^14.0.3",
"react-dom": "^19.2.5",
"react-router-dom": "^7.17.0",
"react-shiki": "^0.9.3",
"remark-math": "^6.0.0",
"remend": "^1.3.0",
"shiki": "^4.0.2",
"simple-git": "^3.36.0",
"streamdown": "^2.5.0",
"tailwind-merge": "^3.5.0",
"tailwindcss": "^4.2.4",
"tw-shimmer": "^0.4.11",
"unicode-animations": "^1.0.3",
"unified": "^11.0.5",
"unist-util-visit-parents": "^6.0.2",
"use-stick-to-bottom": "^1.1.6",
"vfile": "^6.0.3",
"web-haptics": "^0.0.6"
"radix-ui": "1.6.7",
"react": "19.2.7",
"react-arborist": "3.13.2",
"react-dnd-html5-backend": "14.1.0",
"react-dom": "19.2.7",
"react-router": "8.3.0",
"react-shiki": "0.9.3",
"remark-math": "6.0.0",
"remend": "1.3.0",
"shiki": "4.3.1",
"simple-git": "3.36.0",
"streamdown": "2.5.0",
"tailwind-merge": "3.6.0",
"tailwindcss": "4.3.3",
"tw-shimmer": "0.4.12",
"unicode-animations": "1.0.3",
"unified": "11.0.5",
"unist-util-visit-parents": "6.0.2",
"use-stick-to-bottom": "1.1.6",
"vfile": "6.0.3",
"web-haptics": "0.0.6"
},
"devDependencies": {
"@electron/rebuild": "^4.0.6",
"@eslint/js": "^9.39.4",
"@playwright/test": "=1.58.2",
"@testing-library/dom": "^10.4.0",
"@testing-library/react": "^16.3.2",
"@types/d3-force": "^3.0.10",
"@types/hast": "^3.0.4",
"@types/node": "^22.20.0",
"@types/react": "^19.2.14",
"@types/react-dom": "^19.2.3",
"@typescript-eslint/eslint-plugin": "^8.59.1",
"@typescript-eslint/parser": "^8.59.1",
"@vitejs/plugin-react": "^6.0.1",
"@electron/rebuild": "4.2.0",
"@playwright/test": "1.58.2",
"@testing-library/dom": "10.4.1",
"@testing-library/react": "16.3.2",
"@types/d3-force": "3.0.10",
"@types/hast": "3.0.5",
"@types/node": "22.20.1",
"@types/react": "19.2.17",
"@types/react-dom": "19.2.3",
"@vitejs/plugin-react": "6.0.3",
"bippy": "0.5.43",
"concurrently": "^10.0.3",
"cross-env": "^10.1.0",
"concurrently": "10.0.4",
"cross-env": "10.1.0",
"electron": "40.10.2",
"electron-builder": "^26.8.1",
"esbuild": "^0.28.1",
"eslint": "^9.39.4",
"eslint-plugin-perfectionist": "^5.9.0",
"eslint-plugin-react": "^7.37.5",
"eslint-plugin-react-hooks": "^7.1.1",
"eslint-plugin-unused-imports": "^4.4.1",
"globals": "^16.5.0",
"jsdom": "^29.1.1",
"prettier": "^3.8.3",
"rcedit": "^5.0.2",
"tsx": "^4.22.4",
"typescript": "^6.0.3",
"vite": "^8.0.10",
"vitest": "^4.1.5",
"wait-on": "^9.0.5"
"electron-builder": "26.15.3",
"esbuild": "0.28.1",
"jsdom": "29.1.1",
"prettier": "3.9.5",
"rcedit": "5.0.2",
"tsx": "4.23.1",
"typescript": "6.0.3",
"vite": "8.2.0",
"vitest": "4.1.10",
"wait-on": "9.0.10"
},
"build": {
"electronVersion": "40.10.2",
+1 -1
View File
@@ -1,6 +1,6 @@
import type * as React from 'react'
import { memo, useCallback, useEffect, useMemo, useState } from 'react'
import { useNavigate } from 'react-router-dom'
import { useNavigate } from 'react-router'
import { ZoomableImage } from '@/components/chat/zoomable-image'
import { PageLoader } from '@/components/page-loader'
+13 -1
View File
@@ -1,12 +1,14 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const closeFocusedSessionTab = vi.fn(() => false)
const closeFocusedToolTab = vi.fn(() => false)
const nextSessionTileForWorkspace = vi.fn<() => null | string>(() => null)
const closeSessionTile = vi.fn()
const requestFreshSession = vi.fn()
vi.mock('@/components/pane-shell/tree/store', () => ({
closeFocusedSessionTab: () => closeFocusedSessionTab()
closeFocusedSessionTab: () => closeFocusedSessionTab(),
closeFocusedToolTab: () => closeFocusedToolTab()
}))
vi.mock('@/store/session-states', () => ({
@@ -51,6 +53,7 @@ beforeEach(() => {
$activeSessionId.set(null)
$workspaceIsPage.set(false)
closeFocusedSessionTab.mockReturnValue(false)
closeFocusedToolTab.mockReturnValue(false)
nextSessionTileForWorkspace.mockReturnValue(null)
vi.clearAllMocks()
})
@@ -135,4 +138,13 @@ describe('closeWorkspaceTab', () => {
expect(closeActiveTab(vi.fn())).toBe(true)
expect(requestFreshSession).toHaveBeenCalledTimes(1)
})
it('a focused tool panel (terminal / logs) claims ⌘W before main empties', () => {
loadedMainOnly()
closeFocusedToolTab.mockReturnValue(true)
expect(closeActiveTab(vi.fn())).toBe(true)
// The logs/terminal tab closed — main keeps its loaded chat.
expect(requestFreshSession).not.toHaveBeenCalled()
})
})
+10 -3
View File
@@ -1,7 +1,7 @@
import { mainChatOccupied } from '@/app/open-session'
import { closeActiveTerminal } from '@/app/right-sidebar/terminal/terminals'
import { $workspaceIsPage } from '@/app/routes'
import { closeFocusedSessionTab } from '@/components/pane-shell/tree/store'
import { closeFocusedSessionTab, closeFocusedToolTab } from '@/components/pane-shell/tree/store'
import { isFocusWithin } from '@/lib/keybinds/combo'
import { $previewTabs, closeActiveRightRailTab } from '@/store/preview'
import { requestFreshSession } from '@/store/profile'
@@ -56,12 +56,13 @@ export function closeWorkspaceTab(loadSessionIntoWorkspace?: (storedSessionId: s
* 1. a focused terminal → its active terminal tab,
* 2. right-rail tabs (live preview and/or file peeks),
* 3. the FOCUSED chat zone → its active tab (a session tile stacked into it).
* 4. the workspace tab itself — see `closeWorkspaceTab`.
* 4. a focused TOOL PANEL zone (terminal / logs) → its active tab.
* 5. the workspace tab itself — see `closeWorkspaceTab`.
* Returns false when nothing closes, so ⌘W is a no-op — it never closes the
* window. Shared by the keyboard path (Win/Linux) and the macOS
* menu-accelerator IPC.
*
* Steps 3-4 follow the same focused zone ⌘1…⌘9 indexes, so a second chat zone
* Steps 3-5 follow the same focused zone ⌘1…⌘9 indexes, so a second chat zone
* with its own tab strip closes ITS tab instead of main's.
*/
export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: string) => void): boolean {
@@ -84,5 +85,11 @@ export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: stri
return true
}
// A tool panel zone hosts no chat strip, so the chat rung skips it — but its
// tabs close like any other. Without this ⌘W was dead over terminal / logs.
if (closeFocusedToolTab()) {
return true
}
return closeWorkspaceTab(loadSessionIntoWorkspace)
}
@@ -0,0 +1,150 @@
import { cleanup, fireEvent, render, screen } from '@testing-library/react'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { I18nProvider } from '@/i18n'
import { ComposerDirectiveActions } from './directive-actions'
import { refChipElement } from './rich-editor'
const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] }
const openSession = vi.fn()
vi.mock('@/app/open-session', () => ({ openSession: (...args: unknown[]) => openSession(...args) }))
/** A live contenteditable holding real chips, with the watcher bound to it —
* the same pair both composers mount. */
function mountEditor(chips: { kind: string; value: string }[]) {
const editor = document.createElement('div')
editor.contentEditable = 'true'
editor.append(...chips.map(chip => refChipElement(chip.kind, `\`${chip.value}\``)))
document.body.append(editor)
render(
<I18nProvider configClient={null} initialLocale="en">
<ComposerDirectiveActions editorRef={{ current: editor }} />
</I18nProvider>
)
return editor
}
function chips(editor: HTMLElement, kind: string) {
return Array.from(editor.querySelectorAll(`[data-ref-kind="${kind}"]`))
}
function hover(node: Element) {
fireEvent.pointerOver(node, { bubbles: true })
}
/** The reference the visible action pill points at, or null when there is none. */
function pillValue() {
return document.querySelector('[data-slot="composer-directive-action"]')?.getAttribute('data-value') ?? null
}
afterEach(() => {
cleanup()
document.body.replaceChildren()
delete desktopWindow.hermesDesktop
openSession.mockReset()
vi.useRealTimers()
})
describe('ComposerDirectiveActions', () => {
it('offers an action for a hovered actionable chip', () => {
const editor = mountEditor([{ kind: 'url', value: 'https://example.com/docs' }])
expect(pillValue()).toBeNull()
hover(chips(editor, 'url')[0]!)
expect(pillValue()).toBe('https://example.com/docs')
})
it('opens a url externally rather than navigating the app', () => {
const openExternal = vi.fn().mockResolvedValue(undefined)
desktopWindow.hermesDesktop = { openExternal } as unknown as Window['hermesDesktop']
const editor = mountEditor([{ kind: 'url', value: 'https://example.com/docs' }])
hover(chips(editor, 'url')[0]!)
fireEvent.click(screen.getByRole('button'))
expect(openExternal).toHaveBeenCalledWith('https://example.com/docs')
expect(pillValue()).toBeNull()
})
it('runs the kind-specific action — a session chip opens the session', async () => {
const editor = mountEditor([{ kind: 'session', value: 'default/20260722_204335_d62c16' }])
hover(chips(editor, 'session')[0]!)
fireEvent.click(screen.getByRole('button'))
// openSessionRef lazy-imports the navigator, so the call lands a tick later.
await vi.waitFor(() =>
expect(openSession).toHaveBeenCalledWith('20260722_204335_d62c16', expect.any(Function), 'tab')
)
})
it('leaves kinds with no action alone', () => {
const editor = mountEditor([{ kind: 'file', value: 'src/main.tsx' }])
hover(chips(editor, 'file')[0]!)
expect(pillValue()).toBeNull()
})
it('follows the pointer from one chip to the next', () => {
const editor = mountEditor([
{ kind: 'url', value: 'https://one.example' },
{ kind: 'url', value: 'https://two.example' }
])
const [first, second] = chips(editor, 'url')
hover(first!)
expect(pillValue()).toBe('https://one.example')
hover(second!)
expect(pillValue()).toBe('https://two.example')
})
it('keeps the pill up while the pointer crosses onto it', () => {
vi.useFakeTimers()
const editor = mountEditor([{ kind: 'url', value: 'https://example.com' }])
const chip = chips(editor, 'url')[0]!
hover(chip)
fireEvent.pointerOut(chip, { relatedTarget: document.body })
fireEvent.mouseEnter(screen.getByRole('button').parentElement!)
vi.advanceTimersByTime(500)
expect(pillValue()).toBe('https://example.com')
})
it('binds to the document so a late-attached editor still gets the affordance', () => {
// The edit composer's editor isn't reliably in the DOM when the effect
// first runs; a document listener that reads the editor lazily works
// regardless — this is the whole reason it binds to document, not editor.
const editor = document.createElement('div')
editor.contentEditable = 'true'
editor.append(refChipElement('url', '`https://late.example`'))
render(
<I18nProvider configClient={null} initialLocale="en">
<ComposerDirectiveActions editorRef={{ current: editor }} />
</I18nProvider>
)
// Editor attached AFTER mount.
document.body.append(editor)
hover(editor.querySelector('[data-ref-kind="url"]')!)
expect(pillValue()).toBe('https://late.example')
})
})
@@ -0,0 +1,154 @@
/**
* Hover actions for directive chips in a composer.
*
* A directive chip (`@url:`, `@session:`, …) reads as the thing it points at
* and is coloured like one, but a composer is an editor — a click inside the
* contenteditable only places the caret, so there's no way to *act* on the
* reference. Instead, hovering a chip whose kind has an action floats a small
* pill above it that runs it.
*
* The kind → action table (`DIRECTIVE_ACTIONS`) lives in `directive-text`, so
* it is shared with the sent-message chip: one entry lights up both surfaces.
*/
import { type RefObject, useCallback, useEffect, useRef, useState } from 'react'
import { createPortal } from 'react-dom'
import { DIRECTIVE_ACTIONS, type DirectiveAction } from '@/components/assistant-ui/directive-text'
import { composerFloatingPill } from '@/components/chat/composer-dock'
import { Codicon } from '@/components/ui/codicon'
import { useI18n } from '@/i18n'
import { cn } from '@/lib/utils'
/** Moving between the chip and the pill crosses a gap where neither is hovered.
* Short enough that it still reads as instant on the way out. */
const HIDE_DELAY_MS = 120
/** The actionable directive chip under `target` that also belongs to `editor`,
* if there is one. */
function actionableChipAt(target: EventTarget | null, editor: HTMLElement): HTMLElement | null {
const chip = target instanceof Element ? target.closest<HTMLElement>('[data-ref-kind]') : null
const kind = chip?.dataset.refKind
return chip && kind && chip.dataset.refId && editor.contains(chip) && DIRECTIVE_ACTIONS[kind] ? chip : null
}
interface Anchor {
action: DirectiveAction
chip: HTMLElement
left: number
top: number
value: string
}
function anchorFor(chip: HTMLElement): Anchor | null {
const value = chip.dataset.refId
const action = chip.dataset.refKind ? DIRECTIVE_ACTIONS[chip.dataset.refKind] : undefined
if (!value || !action || !chip.isConnected) {
return null
}
const rect = chip.getBoundingClientRect()
return { action, chip, left: rect.left, top: rect.top, value }
}
/**
* Renders the action pill for whichever actionable chip in `editorRef` is
* hovered.
*
* Listeners bind to `document`, not the editor, so mount timing can't strand
* them: the edit composer's contenteditable isn't reliably attached when this
* effect first runs, and a document listener that reads the editor lazily works
* regardless. Each instance filters to its own editor, so the docked and edit
* composers never show two pills for one chip.
*/
export function ComposerDirectiveActions({ editorRef }: { editorRef: RefObject<HTMLElement | null> }) {
const { t } = useI18n()
const [anchor, setAnchor] = useState<Anchor | null>(null)
const hideTimerRef = useRef<number | undefined>(undefined)
const cancelHide = useCallback(() => {
window.clearTimeout(hideTimerRef.current)
}, [])
const hideSoon = useCallback(() => {
cancelHide()
hideTimerRef.current = window.setTimeout(() => setAnchor(null), HIDE_DELAY_MS)
}, [cancelHide])
useEffect(() => {
const onPointerOver = (event: PointerEvent) => {
const editor = editorRef.current
const chip = editor && actionableChipAt(event.target, editor)
if (!chip) {
return
}
cancelHide()
setAnchor(current => (current?.chip === chip ? current : anchorFor(chip)))
}
const onPointerOut = (event: PointerEvent) => {
const editor = editorRef.current
const chip = editor && actionableChipAt(event.target, editor)
// A move within the same chip (its icon → its label) is not a leave.
if (chip && editor && chip === actionableChipAt(event.relatedTarget, editor)) {
return
}
hideSoon()
}
// The chip can move or vanish under a parked pointer: the editor scrolls,
// the window resizes, or the user deletes the reference the pill points at.
const reanchor = () => setAnchor(current => (current ? anchorFor(current.chip) : null))
document.addEventListener('pointerover', onPointerOver)
document.addEventListener('pointerout', onPointerOut)
window.addEventListener('scroll', reanchor, true)
window.addEventListener('resize', reanchor)
return () => {
document.removeEventListener('pointerover', onPointerOver)
document.removeEventListener('pointerout', onPointerOut)
window.removeEventListener('scroll', reanchor, true)
window.removeEventListener('resize', reanchor)
window.clearTimeout(hideTimerRef.current)
}
}, [cancelHide, editorRef, hideSoon])
if (!anchor) {
return null
}
return createPortal(
<div
className="fixed z-(--z-over-modal) -translate-y-full pb-1"
data-slot="composer-directive-action"
data-value={anchor.value}
onMouseEnter={cancelHide}
onMouseLeave={hideSoon}
style={{ left: anchor.left, top: anchor.top }}
>
<button
className={cn(composerFloatingPill, 'shadow-nous')}
onClick={() => {
anchor.action.run(anchor.value)
setAnchor(null)
}}
// Never let the press reach the editor: mousedown inside a
// contenteditable moves the caret and can collapse a selection the user
// still wants, and the edit composer treats a blur as "cancel".
onMouseDown={event => event.preventDefault()}
type="button"
>
<Codicon className="shrink-0 opacity-70" name={anchor.action.icon} size="0.75rem" />
{anchor.action.label(t)}
</button>
</div>,
document.body
)
}
@@ -1,7 +1,9 @@
import { describe, expect, it } from 'vitest'
import {
beginComposerComposition,
composerPlainText,
deleteChipBeforeCaret,
normalizeComposerEditorDom,
renderComposerContents,
RICH_INPUT_SLOT
@@ -136,4 +138,133 @@ describe('an emptied composer shows its placeholder again', () => {
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true)
})
// Chromium leaves zero-length text nodes behind whenever an edit lands next
// to a contenteditable=false chip. They render as nothing, so an editor
// holding only those is empty to the user — counting them as contents left
// the placeholder hidden under a composer that looked blank.
it('advertises emptiness for an editor holding only zero-length text nodes', () => {
const el = editor()
el.append(document.createTextNode(''), document.createTextNode(''))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true)
})
it('does not advertise emptiness while real text sits beside that litter', () => {
const el = editor()
el.append(document.createTextNode(''), document.createTextNode('one'), document.createTextNode(''))
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
// Input events are skipped for the duration of an IME composition, so nothing
// else clears the marker until it ends — the hint would sit behind the
// hiragana the user is composing (#75960).
it('hides the placeholder before IME preedit text starts', () => {
const el = emptied()
beginComposerComposition(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false)
})
it('brings the placeholder back when composition ends with nothing committed', () => {
const el = emptied()
beginComposerComposition(el)
normalizeComposerEditorDom(el)
expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true)
})
})
/** A directive chip, as `refChipElement` builds it. */
function chip(): HTMLSpanElement {
const el = document.createElement('span')
el.contentEditable = 'false'
el.dataset.refText = '@folder:`apps/desktop/`'
el.append(document.createTextNode('apps/desktop/'))
return el
}
function caretAt(node: Node, offset: number) {
const range = document.createRange()
range.setStart(node, offset)
range.collapse(true)
const selection = window.getSelection()
selection?.removeAllRanges()
selection?.addRange(range)
}
/** Committing a completion empties the typed token's text node instead of
* removing it, and `Range.insertNode` splits the line around the caret — so a
* freshly-chipped directive sits between zero-length text nodes. Backspace has
* to see past them or the chip can't be deleted at all. */
describe('backspace deletes a chip surrounded by Chromium litter', () => {
it('deletes the chip when the caret sits in a zero-length text node after it', () => {
const el = editor()
el.append(document.createTextNode(''), chip(), document.createTextNode(''))
caretAt(el.childNodes[2] as Node, 0)
expect(deleteChipBeforeCaret(el)).toBe(true)
expect(el.querySelector('[data-ref-text]')).toBeNull()
})
it('deletes the chip when the caret is past a zero-length text node at editor level', () => {
const el = editor()
el.append(chip(), document.createTextNode(''))
caretAt(el, 2)
expect(deleteChipBeforeCaret(el)).toBe(true)
expect(el.querySelector('[data-ref-text]')).toBeNull()
})
it('still swallows the auto-inserted trailing space through that litter', () => {
const el = editor()
el.append(chip(), document.createTextNode(''), document.createTextNode(' '))
caretAt(el, 2)
expect(deleteChipBeforeCaret(el)).toBe(true)
expect(composerPlainText(el)).toBe('')
})
it('keeps real following text when it deletes the chip', () => {
const el = editor()
el.append(chip(), document.createTextNode(''), document.createTextNode(' and this'))
caretAt(el, 2)
expect(deleteChipBeforeCaret(el)).toBe(true)
expect(composerPlainText(el)).toBe('and this')
})
it('leaves plain text to the native backspace', () => {
const el = editor()
el.append(document.createTextNode('hello'))
caretAt(el.firstChild as Node, 5)
expect(deleteChipBeforeCaret(el)).toBe(false)
})
it('sweeps the litter out of the editor when it normalizes', () => {
const el = editor()
el.append(document.createTextNode(''), chip(), document.createTextNode(''))
normalizeComposerEditorDom(el)
expect(Array.from(el.childNodes).map(node => node.nodeName)).toEqual(['SPAN'])
})
})
@@ -4,6 +4,7 @@ import { useCallback, useEffect, useRef, useState } from 'react'
import { useI18n } from '@/i18n'
import { chatMessageText, collectUnspokenTurnSpeech } from '@/lib/chat-messages'
import { triggerHaptic } from '@/lib/haptics'
import { clearWakeIndicator, syncWakeIndicatorWithVoice } from '@/lib/wake-indicator'
import { $voiceConversationStartRequest, takeVoiceConversationStart } from '@/store/composer'
import { resetBrowseState } from '@/store/composer-input-history'
import { $gateway } from '@/store/gateway'
@@ -62,6 +63,7 @@ export function useComposerVoice({
const { $messages } = useComposerScope()
const [voiceConversationActive, setVoiceConversationActive] = useState(false)
const lastSpokenIdRef = useRef<string | null>(null)
const ownsWakeIndicatorRef = useRef(false)
const voiceStartRequest = useStore($voiceConversationStartRequest)
const { dictate, voiceActivityState, voiceStatus } = useVoiceRecorder({
@@ -150,6 +152,26 @@ export function useComposerVoice({
beforeMicOpen: () => wakePauseBarrierRef.current ?? undefined
})
// eslint-disable-next-line no-restricted-syntax -- ownership token used only by unmount cleanup
useEffect(() => {
if (target !== 'main') {
return
}
if (syncWakeIndicatorWithVoice(voiceConversationActive, conversation.status)) {
ownsWakeIndicatorRef.current = voiceConversationActive
}
}, [conversation.status, target, voiceConversationActive])
useEffect(
() => () => {
if (ownsWakeIndicatorRef.current) {
clearWakeIndicator()
}
},
[]
)
// The `composer.voice` hotkey (Ctrl+B) toggles the conversation. Starting
// with STT unconfigured lets the conversation surface its own "configure
// speech-to-text" notice rather than silently no-opping.
@@ -287,8 +287,7 @@ export function useVoiceConversation({
// start — don't auto-start the next sentence, the user chose to stop.
const stoppedByUser =
stoppedDuringSetup ||
(speechStartSequenceRef.current > 0 &&
$voicePlayback.get().sequence > speechStartSequenceRef.current)
(speechStartSequenceRef.current > 0 && $voicePlayback.get().sequence > speechStartSequenceRef.current)
speechStartSequenceRef.current = 0
+9 -3
View File
@@ -32,6 +32,7 @@ import {
import { ContextMenu } from './context-menu'
import { COMPOSER_AREAS, runComposerMiddleware } from './contrib'
import { ComposerControls } from './controls'
import { ComposerDirectiveActions } from './directive-actions'
import { COMPOSER_DROP_ACTIVE_CLASS, COMPOSER_DROP_FADE_CLASS } from './drop-affordance'
import { markActiveComposer } from './focus'
import { HelpHint } from './help-hint'
@@ -57,7 +58,7 @@ import { ActionBadges } from './micro-actions'
import { chipTypedPathOnSpace, pathifyRefs } from './path-refs'
import { QueuePanel } from './queue-panel'
import {
COMPOSER_PLACEHOLDER_CLASS,
beginComposerComposition,
composerPlainText,
deleteChipBeforeCaret,
deleteSelectionInEditor,
@@ -947,7 +948,6 @@ export function ChatBar({
autoCorrect="off"
className={cn(
'min-h-[1.625rem] min-h-(--composer-input-min-height) max-h-(--composer-input-max-height) cursor-text overflow-y-auto whitespace-pre-wrap break-words [overflow-wrap:anywhere] bg-transparent pb-1 pr-1 pt-1 leading-normal text-foreground outline-none disabled:cursor-not-allowed',
COMPOSER_PLACEHOLDER_CLASS,
'**:data-ref-text:cursor-default',
stacked && 'pl-3',
stacked ? 'w-full' : 'min-w-(--composer-input-inline-min-width) flex-1'
@@ -969,8 +969,13 @@ export function ChatBar({
// until an unrelated edit forces a sync (#39614).
flushEditorToDraft(event.currentTarget)
}}
onCompositionStart={() => {
onCompositionStart={event => {
composingRef.current = true
// Input events are skipped for the rest of the composition, so
// nothing else would clear the empty marker until it ends — and the
// hint would sit behind the preedit text the whole time (#75960).
beginComposerComposition(event.currentTarget)
}}
onDragOver={handleInputDragOver}
onDrop={handleInputDrop}
@@ -985,6 +990,7 @@ export function ChatBar({
spellCheck={false}
suppressContentEditableWarning
/>
<ComposerDirectiveActions editorRef={editorRef} />
{/* assistant-ui requires ComposerPrimitive.Input somewhere in the tree
so the composer-state binding (text + IME + paste + form-submit hookup)
wires up. We render the real input UI ourselves above via the
@@ -1,5 +1,6 @@
import { memo, useState } from 'react'
import { composerFloatingPill } from '@/components/chat/composer-dock'
import { Codicon } from '@/components/ui/codicon'
import { useSessionSlice } from '@/lib/use-session-slice'
import { cn } from '@/lib/utils'
@@ -7,11 +8,9 @@ import { $composerActionsBySession, type ComposerAction } from '@/store/composer
import { notifyError } from '@/store/notifications'
/**
* Floating pill — the treatment the thread's jump/approval button uses for a
* control that sits over scrolling content: full radius, hairline border, the
* shared composer fill behind a blur so thread text never bleeds through.
* Sized against the composer's own control height so a row of pills lines up
* with the chrome it floats above.
* Floating pill — the shared treatment for a control that sits over the
* composer (`composerFloatingPill`), plus this strip's own width cap and
* disabled state.
*
* NEVER `pointer-events-none`, not even when disabled. The pop-out drag region
* is an `absolute` sibling behind these pills, so a pill that stops taking
@@ -19,10 +18,8 @@ import { notifyError } from '@/store/notifications'
* becomes a grab handle that floats the composer.
*/
const PILL = cn(
'inline-flex h-(--composer-control-size) max-w-56 shrink-0 cursor-pointer items-center gap-1.5 rounded-full px-2.5',
'border border-border/65 bg-(--composer-fill) backdrop-blur-[0.75rem] [-webkit-backdrop-filter:blur(0.75rem)]',
'text-xs font-normal text-(--ui-text-secondary) transition-colors',
'hover:bg-(--chrome-action-hover) hover:text-foreground',
composerFloatingPill,
'max-w-56',
'disabled:cursor-default disabled:opacity-50 disabled:hover:bg-(--composer-fill)',
'focus-visible:outline-none focus-visible:ring-[0.1875rem] focus-visible:ring-ring/50'
)
@@ -21,28 +21,67 @@ import { slashCommandMatches, type SlashCommandScanOptions } from './slash-refs'
export const RICH_INPUT_SLOT = 'composer-rich-input'
/** Paints `data-placeholder` while the editor is empty.
/** Chromium's litter: editing beside a `contenteditable=false` chip splits the
* line and leaves zero-length text nodes behind. They render as nothing and
* serialize as nothing, so no reader of the editor should count them. */
function isEmptyTextNode(node: ChildNode | null): boolean {
return node?.nodeType === Node.TEXT_NODE && !node.textContent
}
/** The node before `node`, stepping over that litter. */
function meaningfulPreviousSibling(node: ChildNode | null): ChildNode | null {
let prev = node?.previousSibling ?? null
while (isEmptyTextNode(prev)) {
prev = prev?.previousSibling ?? null
}
return prev
}
/** The node after `node`, stepping over that litter. */
function meaningfulNextSibling(node: ChildNode | null): ChildNode | null {
let next = node?.nextSibling ?? null
while (isEmptyTextNode(next)) {
next = next?.nextSibling ?? null
}
return next
}
/** Keep the `data-empty` marker the placeholder paints on in step with the
* editor root's contents.
*
* `:empty` can't be the whole test: a cleared editor keeps a scaffolding <br>
* so the contenteditable doesn't collapse, and that break makes `:empty`
* false. Nor can CSS infer it on its own — a text node is invisible to
* selectors, so `one<br>` and a lone `<br>` are the same shape, and
* `:has(> br:only-child)` would paint the placeholder straight over the
* user's text. The code that empties the editor is what knows, so it marks it.
* `:has(> br:only-child)` would paint the placeholder over the user's text.
* The code that empties the editor is what knows, so it marks it.
*
* @see markEditorEmptiness */
export const COMPOSER_PLACEHOLDER_CLASS =
'[&:is(:empty,[data-empty])]:before:content-[attr(data-placeholder)] [&:is(:empty,[data-empty])]:before:text-muted-foreground/60'
/** Keep that marker in step with the editor root's contents. */
* Zero-length text nodes don't count as contents. Chromium leaves them behind
* whenever an edit lands next to a `contenteditable=false` chip, and counting
* them left an editor the user had emptied looking occupied. */
export function markEditorEmptiness(editor: HTMLElement) {
if (editor.childNodes.length === 0) {
if (Array.from(editor.childNodes).every(isEmptyTextNode)) {
editor.dataset.empty = ''
} else {
delete editor.dataset.empty
}
}
/** Drop the marker as IME composition starts, before any preedit text lands.
*
* Input events during composition are deliberately skipped (they carry
* uncommitted preedit text), so nothing else clears the marker until
* `compositionend` — and the hint would otherwise sit behind the hiragana the
* user is composing. `normalizeComposerEditorDom` restores it if composition
* ends with nothing committed. */
export function beginComposerComposition(editor: HTMLElement) {
delete editor.dataset.empty
}
/** @see referenceRe — the shared pattern every surface recognises a reference
* with. Module-level `/g` regexes carry `lastIndex`, so call sites reset it. */
export const REF_RE = referenceRe()
@@ -407,7 +446,14 @@ export function replaceBeforeCaret(editor: HTMLElement, length: number, fragment
/** Backspace at a collapsed caret immediately after a chip: delete the chip AND
* the single trailing space we auto-insert after it, atomically — so removing a
* directive never strands an orphaned space (the contenteditable-driven cleanup
* was unreliable). Returns whether it ran. */
* was unreliable). Returns whether it ran.
*
* "Immediately after" has to be read through Chromium's litter. Committing a
* completion empties the typed token's text node rather than removing it, and
* `Range.insertNode` splits around the caret, so the chip routinely sits
* between zero-length text nodes. Reading those as content made the caret look
* like it was after plain text; the delete declined and Chromium's own
* backspace bounced between the leftovers instead of removing the chip. */
export function deleteChipBeforeCaret(editor: HTMLElement): boolean {
const hit = composerSelectionRange(editor)
@@ -419,16 +465,20 @@ export function deleteChipBeforeCaret(editor: HTMLElement): boolean {
let chip: ChildNode | null = null
if (startContainer === editor) {
chip = startOffset > 0 ? editor.childNodes[startOffset - 1] : null
chip = startOffset > 0 ? (editor.childNodes[startOffset - 1] ?? null) : null
if (isEmptyTextNode(chip)) {
chip = meaningfulPreviousSibling(chip)
}
} else if (startContainer.nodeType === Node.TEXT_NODE && startOffset === 0) {
chip = startContainer.previousSibling
chip = meaningfulPreviousSibling(startContainer as ChildNode)
}
if (chip?.nodeType !== Node.ELEMENT_NODE || !(chip as HTMLElement).dataset.refText) {
return false
}
const after = chip.nextSibling
const after = meaningfulNextSibling(chip)
chip.remove()
// Drop the auto-inserted trailing space; keep any real following text.
@@ -666,6 +716,14 @@ function isBlankNode(node: ChildNode | null): boolean {
* rendering emits (we use text nodes + <br> + chips). Real <br> line breaks
* (Shift+Enter, which sit after actual text) are preserved. */
export function normalizeComposerEditorDom(editor: HTMLElement) {
// Chromium's zero-length text nodes first: every check below reads siblings,
// and litter between them makes a chip look like it has text either side.
for (const child of Array.from(editor.childNodes)) {
if (isEmptyTextNode(child)) {
child.remove()
}
}
// A trailing block wrapper holding only a break/whitespace is the phantom
// "new line" Chromium adds after a chip on backspace — drop it.
const tailBlock = editor.lastChild as HTMLElement | null
@@ -0,0 +1,93 @@
import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
import { atom } from 'nanostores'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { $notifications, clearNotifications } from '@/store/notifications'
vi.mock('@/store/coding-status', () => ({
registerRepoStatusCwd: () => undefined,
repoStatusForCwd: () =>
atom({
added: 12,
ahead: 0,
behind: 0,
branch: 'bb/hitbox',
defaultBranch: 'main',
detached: false,
removed: 3,
untracked: 0
}),
repoWorktreesForCwd: () => atom([])
}))
const { CodingStatusRow } = await import('./coding-row')
describe('CodingStatusRow', () => {
afterEach(() => {
cleanup()
})
it('opens the review pane from the branch and the diff counts, never the bar itself', () => {
const onOpen = vi.fn()
const { container } = render(<CodingStatusRow onOpen={onOpen} repoPath="/repo" />)
const bar = container.querySelector<HTMLElement>('.coding-status-bar')
expect(bar).not.toBeNull()
fireEvent.click(bar!)
expect(onOpen).not.toHaveBeenCalled()
fireEvent.click(screen.getByText('bb/hitbox'))
expect(onOpen).toHaveBeenCalledTimes(1)
fireEvent.click(screen.getByText('12'))
expect(onOpen).toHaveBeenCalledTimes(2)
})
it('wraps the click targets without adding a layout box', () => {
const { container } = render(<CodingStatusRow onOpen={() => undefined} repoPath="/repo" />)
// `display: contents` is what keeps the branch label and the counts direct
// flex children of the row — the hit areas cost nothing visually.
expect(screen.getByText('bb/hitbox').parentElement?.classList.contains('contents')).toBe(true)
expect(screen.getByText('12').closest('button')?.classList.contains('contents')).toBe(true)
// The glyph button fills the row's existing 3.5 leading slot exactly.
expect(container.querySelector('button[class~="size-3.5"]')).not.toBeNull()
})
it('parks the copy glyph against the end of the path, not the end of the row', () => {
render(<CodingStatusRow onOpen={() => undefined} repoPath="/Users/someone/www/repo" />)
const path = screen.getByText('~/www/repo')
// The path sizes to its content and the glyph is its immediate sibling, so
// the pair reads as one unit. `flex-1` belongs to the wrapper (which holds
// the row's slack open) — on the label it stretched the text and pushed the
// glyph out to the kebab.
expect(path.classList.contains('flex-1')).toBe(false)
expect(path.parentElement?.classList.contains('flex-1')).toBe(true)
expect(path.nextElementSibling?.tagName).toBe('BUTTON')
})
it('copies the absolute cwd inline — checkmark feedback, no toast', async () => {
const writeText = vi.fn().mockResolvedValue(undefined)
Object.defineProperty(navigator, 'clipboard', { configurable: true, value: { writeText } })
clearNotifications()
render(<CodingStatusRow onOpen={() => undefined} repoPath="/Users/someone/www/repo" />)
// Painted tildified, copied raw.
expect(screen.getByText('~/www/repo')).toBeTruthy()
const copy = screen.getByRole('button', { name: 'Copy Path' })
fireEvent.click(copy)
await waitFor(() => expect(writeText).toHaveBeenCalledWith('/Users/someone/www/repo'))
// Confirmation is the button turning into a checkmark, not a notification.
await waitFor(() => expect(screen.getByRole('button', { name: 'Copied' })).toBeTruthy())
expect($notifications.get()).toHaveLength(0)
})
})

Some files were not shown because too many files have changed in this diff Show More