diff --git a/.github/actions/detect-changes/action.yml b/.github/actions/detect-changes/action.yml index 145d742d5b..cecbd4515c 100644 --- a/.github/actions/detect-changes/action.yml +++ b/.github/actions/detect-changes/action.yml @@ -15,6 +15,9 @@ outputs: python: description: Run Python tests / ruff / ty / windows-footguns. value: ${{ steps.classify.outputs.python }} + python_prod: + description: Python changes outside tests/ — gates product jobs (Desktop E2E, Docker). + value: ${{ steps.classify.outputs.python_prod }} frontend: description: Run the TypeScript testing matrix + desktop build. value: ${{ steps.classify.outputs.frontend }} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3441d7bb6d..beb9c41f6a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -42,6 +42,7 @@ jobs: timeout-minutes: 10 outputs: python: ${{ steps.classify.outputs.python }} + python_prod: ${{ steps.classify.outputs.python_prod }} frontend: ${{ steps.classify.outputs.frontend }} site: ${{ steps.classify.outputs.site }} scan: ${{ steps.classify.outputs.scan }} @@ -89,7 +90,19 @@ jobs: e2e-desktop: name: Desktop E2E needs: detect - if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' + # python_prod (not python): the Playwright suite exercises the built app + # + `hermes serve` backend, which never import anything under tests/. + # Tests-only PRs (~17% of commits) skip this 5-minute job — the longest + # single job in the workflow — while still running the full pytest lanes. + # + # ⛔ TEMPORARILY DISABLED (Aug 2, 2026, Teknium) — the suite is red on + # every PR and on main itself since the Aug 1 night engines/npm churn + # (#76499 → #76562 → #76575): the mock-backend Electron window never + # gets a title, so boot/chat/setup/interim specs all fail identically + # regardless of the PR's diff (verified on #76573 and the docs-only + # #76582). Tracking issue: #76627 (assigned: Ari). To re-enable, + # delete the `false &&` below — nothing else changed. + if: ${{ false && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true') }} uses: ./.github/workflows/e2e-desktop.yml docs-site: @@ -137,8 +150,10 @@ jobs: needs: detect # Trusted main pushes run docker.yml directly so its container-publish # environment secrets never cross this reusable-workflow call. PR runs - # remain build/test-only and secret-free. - if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true') + # remain build/test-only and secret-free. Gated on python_prod (not + # python): the image copies installed code, never tests/ — tests-only + # PRs skip the build. + if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true') uses: ./.github/workflows/docker.yml supply-chain: diff --git a/.github/workflows/contributor-check.yml b/.github/workflows/contributor-check.yml index 014e1e2ff9..791e630558 100644 --- a/.github/workflows/contributor-check.yml +++ b/.github/workflows/contributor-check.yml @@ -67,6 +67,8 @@ jobs: echo -e "$MISSING" echo "" echo "Add a mapping file (do NOT edit AUTHOR_MAP in release.py):" + echo " python3 scripts/audit_pr_attribution.py --fix # auto-resolve + create files" + echo "or manually:" echo -e "$MISSING" | while read -r line; do email=$(echo "$line" | sed 's/^ *//' | cut -d' ' -f1) [ -z "$email" ] && continue @@ -78,7 +80,7 @@ jobs: # Emit review_status for unmapped emails DETAIL=$(echo -e "$MISSING" | sed '/^$/d; s/^ //') - HOW_TO_FIX=$'Add mappings to scripts/release.py AUTHOR_MAP:\n```\n"": "",\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n' + HOW_TO_FIX=$'Run from the PR branch:\n```\npython3 scripts/audit_pr_attribution.py --fix\ngit add contributors && git commit -m "chore: map contributor emails" && git push\n```\nOr map one email manually (do NOT edit AUTHOR_MAP in release.py):\n```\npython3 scripts/add_contributor.py \n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n' REVIEW_STATUS=$(jq -nc \ --arg detail "$DETAIL" \ --arg how_to_fix "$HOW_TO_FIX" \ diff --git a/.github/workflows/deploy-site.yml b/.github/workflows/deploy-site.yml index 3ac2c4741f..588d7707ea 100644 --- a/.github/workflows/deploy-site.yml +++ b/.github/workflows/deploy-site.yml @@ -65,10 +65,14 @@ jobs: - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 22 + node-version: 26 cache: npm cache-dependency-path: website/package-lock.json + - name: grab npm 12 + run: | + npm i -g npm@12 + - uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0 with: python-version: '3.11' diff --git a/.github/workflows/docs-site-checks.yml b/.github/workflows/docs-site-checks.yml index 41acf1790f..cf775f89e0 100644 --- a/.github/workflows/docs-site-checks.yml +++ b/.github/workflows/docs-site-checks.yml @@ -15,10 +15,14 @@ jobs: - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 22 + node-version: 26 cache: npm cache-dependency-path: website/package-lock.json + - name: grab npm 12 + run: | + npm i -g npm@12 + - name: Install website dependencies uses: ./.github/actions/retry with: diff --git a/.github/workflows/e2e-desktop.yml b/.github/workflows/e2e-desktop.yml index bdaaa6d6cb..b951c60ead 100644 --- a/.github/workflows/e2e-desktop.yml +++ b/.github/workflows/e2e-desktop.yml @@ -39,8 +39,13 @@ jobs: # ── Node ─────────────────────────────────────────────────────────── - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 22 + node-version: 26 cache: npm + + - name: grab npm 12 + run: | + npm i -g npm@12 + # Full npm ci (not --ignore-scripts): electron's postinstall # downloads the binary we launch, and node-pty's native build is # needed for the terminal pane. @@ -56,7 +61,7 @@ jobs: # fetching a manifest from raw.githubusercontent.com on EVERY job — # a transient fetch failure fails the whole job (2026-07-28 slice-5 # incident). Pinned, the binary downloads directly; no manifest hop. - version: "0.9.28" + version: '0.9.28' enable-cache: true cache-dependency-glob: | pyproject.toml @@ -106,11 +111,11 @@ jobs: npx playwright test --reporter=list fi env: - CI: "true" + CI: 'true' # Ensure no real API keys leak into the test env. - OPENROUTER_API_KEY: "" - OPENAI_API_KEY: "" - NOUS_API_KEY: "" + OPENROUTER_API_KEY: '' + OPENAI_API_KEY: '' + NOUS_API_KEY: '' # ── Save updated baselines to cache (main only) ─────────────────── - name: Save updated baselines to cache diff --git a/.github/workflows/js-autofix.yml b/.github/workflows/js-autofix.yml index 2dfbe7e0b1..d0fa3513e8 100644 --- a/.github/workflows/js-autofix.yml +++ b/.github/workflows/js-autofix.yml @@ -67,9 +67,13 @@ jobs: - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 22 + node-version: 26 cache: npm + - name: grab npm 12 + run: | + npm i -g npm@12 + # --ignore-scripts: eslint only needs TS sources + eslint packages. - uses: ./.github/actions/retry with: diff --git a/.github/workflows/js-tests.yml b/.github/workflows/js-tests.yml index 25786b1c95..9119e0c7a9 100644 --- a/.github/workflows/js-tests.yml +++ b/.github/workflows/js-tests.yml @@ -15,8 +15,13 @@ jobs: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 22 + node-version: 26 cache: npm + + - name: grab npm 12 + run: | + npm i -g npm@12 + - uses: ./.github/actions/retry with: command: npm ci --ignore-scripts @@ -61,8 +66,13 @@ jobs: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 22 + node-version: 26 cache: npm + + - name: grab npm 12 + run: | + npm i -g npm@12 + - uses: ./.github/actions/retry with: command: npm ci diff --git a/.github/workflows/osv-scanner.yml b/.github/workflows/osv-scanner.yml index 455ede33dd..c3aaa50a7b 100644 --- a/.github/workflows/osv-scanner.yml +++ b/.github/workflows/osv-scanner.yml @@ -43,11 +43,13 @@ jobs: uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8 with: # Scan explicit lockfiles rather than recursing, so we only look at - # the three sources of truth and skip vendored / test / worktree dirs. + # the five sources of truth and skip vendored / test / worktree dirs. scan-args: |- --lockfile=uv.lock --lockfile=package-lock.json --lockfile=website/package-lock.json + --lockfile=plugins/platforms/photon/sidecar/package-lock.json + --lockfile=scripts/whatsapp-bridge/package-lock.json # The upstream reusable workflow uploads this exact file under its # fixed artifact name, which the wrapper downloads below. results-file-name: osv-results.sarif diff --git a/.npmrc b/.npmrc new file mode 100644 index 0000000000..d25bd185bb --- /dev/null +++ b/.npmrc @@ -0,0 +1,51 @@ +# needed to prevent bad npm that has min-release-age but not exclude +engine-strict=true + +min-release-age=14 +# allow assistant-ui packages & a couple specific deps since they update a LOT. +# remove this when we stabilize (or we haven't updated in 2 wks) +min-release-age-exclude[]=@assistant-ui/* +min-release-age-exclude[]=assistant-cloud +min-release-age-exclude[]=assistant-stream +min-release-age-exclude[]=@radix-ui/* +min-release-age-exclude[]=radix-ui +min-release-age-exclude[]=safe-content-frame + +# react-router 8.3.0 includes fixes for vulns. remove this when 8.3.0 is > 2wks old. +min-release-age-exclude[]=react-router + +# eslint 10.8.0 includes fixes for vulns. remove this when 10.8.0 is > 2wks old. +min-release-age-exclude[]=eslint +min-release-age-exclude[]=@eslint/* + +# tar 7.5.21 includes fixes for vulns. remove this when 7.5.21 is > 2wks old +min-release-age-exclude[]=tar + +# concurrently 10.0.4 includes fixes for vulns. remove this when 10.0.4 is > 2wks old +min-release-age-exclude[]=concurrently + +# fast-uri 3.1.4 includes fixes for vulns. remove this when 3.1.4 is > 2wks old +min-release-age-exclude[]=fast-uri + +# minimatch 10.2.6 includes fixes for vulns. remove this when 10.2.6 is > 2wks old +min-release-age-exclude[]=minimatch + + +# brace-expansion 5.0.8 includes fixes for vulns. remove this when 5.0.8 is > 2wks old +min-release-age-exclude[]=brace-expansion + +# vite 8.2.0 is the first release depending on rolldown >= 1.2.1, which fixes +# a rolldown panic that breaks `npm run build` in apps/desktop +# (rolldown/rolldown#10337 — a regression in 1.1.5, the version vite 8.1.5 +# pins as ~1.1.5). @oxc-project/types is here because rolldown 1.2.1 pins it +# as `=0.142.0` — an exact pin, so no older release satisfies it and the age +# gate would fail the whole install with ETARGET. +# remove these once vite 8.2.0 is > 2wks old. +min-release-age-exclude[]=vite +min-release-age-exclude[]=rolldown +min-release-age-exclude[]=@rolldown/* +min-release-age-exclude[]=@oxc-project/types + +# ink needs +min-release-age-exclude[]=lightningcss +min-release-age-exclude[]=postcss diff --git a/.nvmrc b/.nvmrc new file mode 100644 index 0000000000..6f4247a625 --- /dev/null +++ b/.nvmrc @@ -0,0 +1 @@ +26 diff --git a/.python-version b/.python-version new file mode 100644 index 0000000000..2c0733315e --- /dev/null +++ b/.python-version @@ -0,0 +1 @@ +3.11 diff --git a/Dockerfile b/Dockerfile index 6d6e9ac2d9..2de6192715 100644 --- a/Dockerfile +++ b/Dockerfile @@ -41,14 +41,14 @@ RUN apt-get -o Acquire::Retries=3 update && \ make install FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source -# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x -# which reached EOL in April 2026 — we copy node + npm + corepack from the -# upstream node:22 image instead so we can stay on a supported LTS without -# waiting for Debian 14 (forky, ~mid-2027). Bookworm-based slim image used -# so the produced binary links against glibc 2.36, which runs cleanly on -# our Debian 13 (trixie, glibc 2.41) runtime. Bumping to a new Node major -# is a one-line ARG change; see #4977. -FROM node:22-bookworm-slim@sha256:7af03b14a13c8cdd38e45058fd957bf00a72bbe17feac43b1c15a689c029c732 AS node_source +# Node 26 source stage. Debian trixie's bundled nodejs is pinned to 20.x +# which reached EOL in April 2026 — we copy node + npm from the upstream +# node:26 image instead (Hermes pins its toolchain to Node 26 everywhere). +# Bookworm-based slim image used so the produced binary links +# against glibc 2.36, which runs cleanly on our Debian 13 (trixie, glibc +# 2.41) runtime. Bumping to a new Node major is a one-line ARG change; see +# #4977. +FROM node:26-bookworm-slim@sha256:9e6f9357d371591e32ab6f2d8a26d63bdd0d17c29eee3f4f3e7e454d9634bf73 AS node_source FROM debian:13.4 # Disable Python stdout buffering to ensure logs are printed immediately. @@ -70,7 +70,7 @@ ENV PLAYWRIGHT_BROWSERS_PATH=/opt/hermes/.playwright # hermes process, the dashboard, and per-profile gateways. RUN apt-get -o Acquire::Retries=3 update && \ apt-get -o Acquire::Retries=3 install -y --no-install-recommends \ - ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev procps git openssh-client docker-cli xz-utils && \ + ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev libatomic1 procps git openssh-client docker-cli xz-utils && \ rm -rf /var/lib/apt/lists/* # Prefer the fixed SQLite over Debian's vulnerable libsqlite3.so.0. Keep the @@ -151,17 +151,20 @@ RUN useradd -u 10000 -m -d /opt/data hermes COPY --chmod=0755 --from=uv_source /usr/local/bin/uv /usr/local/bin/uvx /usr/local/bin/ -# Node 22 LTS: copy the node binary plus the bundled npm + corepack JS -# installs from the upstream image. npm and npx are recreated as symlinks -# because they're symlinks in the source image (and need to live on PATH). +# Node 26: copy the node binary plus the bundled npm JS install from the +# upstream image. npm and npx are recreated as symlinks because they're +# symlinks in the source image (and need to live on PATH). +# +# No corepack: Node unbundled it upstream, so node:26 ships only npm in +# /usr/local/lib/node_modules. Nothing here needs it — no package.json +# declares a `packageManager`, and no build step shells out to yarn or pnpm. +# # See node_source stage at the top of the file for the version-bump # rationale (#4977). COPY --chmod=0755 --from=node_source /usr/local/bin/node /usr/local/bin/ COPY --from=node_source /usr/local/lib/node_modules/npm /usr/local/lib/node_modules/npm -COPY --from=node_source /usr/local/lib/node_modules/corepack /usr/local/lib/node_modules/corepack RUN ln -sf /usr/local/lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm && \ - ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx && \ - ln -sf /usr/local/lib/node_modules/corepack/dist/corepack.js /usr/local/bin/corepack + ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx WORKDIR /opt/hermes @@ -400,6 +403,8 @@ ENV HERMES_LAZY_INSTALL_TARGET=/opt/data/lazy-packages # Recursion is impossible because the shim exec's the venv binary by # absolute path (/opt/hermes/.venv/bin/hermes). See the shim source for # the opt-out env var (HERMES_DOCKER_EXEC_AS_ROOT=1). +COPY --chmod=0755 docker/hermes-exec-shim.sh /opt/hermes/bin/hermes +COPY --chmod=0755 docker/entrypoint-dispatch.sh /opt/hermes/docker/entrypoint-dispatch.sh # Pre-s6 entrypoint.sh did `source .venv/bin/activate` which exported # the venv bin onto PATH; Architecture B's main-wrapper.sh does the @@ -416,27 +421,37 @@ ENV PATH="/opt/hermes/bin:/opt/hermes/.venv/bin:/opt/data/.local/bin:${PATH}" RUN mkdir -p /opt/data VOLUME [ "/opt/data" ] -# s6-overlay's /init is PID 1. It sets up the supervision tree, runs -# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services -# declared in /etc/s6-overlay/s6-rc.d/, then exec's its remaining -# argv as the container's "main program" with stdin/stdout/stderr -# inherited (this is what makes interactive --tui work). When the -# main program exits, /init begins stage 3 shutdown and the container -# exits with the program's exit code. Replaces tini — see Phase 2 of -# docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md. +# The image ENTRYPOINT is a tiny dispatcher rather than `/init` directly. +# When the image really owns PID 1 (normal Docker / Podman), the dispatcher +# execs `/init` and preserves the full s6 supervision tree. When a platform +# wraps the image entrypoint under its own PID-1 init (Fly Machines, +# `docker run --init`, some schedulers), `/init` would abort with +# `can only run as pid 1`; in that case the dispatcher falls back to +# `stage2-hook.sh` + `main-wrapper.sh` directly so foreground commands still +# work. See #38349. +# +# On the PID-1 path, s6-overlay's /init sets up the supervision tree, runs +# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services declared in +# /etc/s6-overlay/s6-rc.d/, then exec's its remaining argv as the container's +# "main program" with stdin/stdout/stderr inherited (this is what makes +# interactive --tui work). When the main program exits, /init begins stage 3 +# shutdown and the container exits with the program's exit code. Replaces +# tini — see Phase 2 of docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md. # # We use the ENTRYPOINT+CMD split rather than CMD alone so the # wrapper is prepended to user-supplied args automatically: # -# docker run → /init main-wrapper.sh (CMD default) -# docker run chat -q "hi" → /init main-wrapper.sh chat -q hi -# docker run sleep infinity → /init main-wrapper.sh sleep infinity -# docker run --tui → /init main-wrapper.sh --tui +# docker run → entrypoint-dispatch.sh (CMD default) +# docker run chat -q "hi" → entrypoint-dispatch.sh chat -q hi +# docker run sleep infinity → entrypoint-dispatch.sh sleep infinity +# docker run --tui → entrypoint-dispatch.sh --tui # # main-wrapper.sh handles arg routing (bare-exec vs. hermes # subcommand vs. no-args), drops to the hermes user via s6-setuidgid, # and exec's the final program so its exit code becomes the container -# exit code. Without the wrapper-as-ENTRYPOINT, leading-dash args -# like `--version` would be intercepted by /init's POSIX shell. -ENTRYPOINT [ "/init", "/opt/hermes/docker/main-wrapper.sh" ] +# exit code. The dispatcher preserves that contract across both the +# supervised PID-1 path and the non-PID-1 fallback path. Without the +# wrapper-as-ENTRYPOINT, leading-dash args like `--version` would be +# intercepted by /init's POSIX shell. +ENTRYPOINT [ "/opt/hermes/docker/entrypoint-dispatch.sh" ] CMD [ ] diff --git a/acp_adapter/entry.py b/acp_adapter/entry.py index fb9ed95450..7b549006bc 100644 --- a/acp_adapter/entry.py +++ b/acp_adapter/entry.py @@ -247,16 +247,22 @@ def main(argv: list[str] | None = None) -> None: import acp from .server import HermesACPAgent - # MCP tool discovery from config.yaml — run before asyncio.run() so - # it's safe to use blocking waits. (ACP also registers per-session - # MCP servers dynamically via asyncio.to_thread inside the event - # loop; that path is unaffected.) Moved from model_tools.py module - # scope to avoid freezing the gateway's loop on lazy import (#16856). + # MCP tool discovery from config.yaml — fire-and-forget in a + # background daemon thread so the ACP server becomes responsive + # immediately while MCP servers connect. Previously this blocked + # asyncio.run() for 2-5 s. (ACP also registers per-session MCP + # servers dynamically via asyncio.to_thread inside the event loop; + # that path is unaffected.) Moved from model_tools.py module scope + # to avoid freezing the gateway's loop on lazy import (#16856). # Metadata-only hosts can opt out of unrelated global MCP startup. if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1": try: - from tools.mcp_tool import discover_mcp_tools - discover_mcp_tools() + from hermes_cli.mcp_startup import start_background_mcp_discovery + + start_background_mcp_discovery( + logger=logger, + thread_name="acp-mcp-discovery", + ) except Exception: logger.debug("MCP tool discovery failed at ACP startup", exc_info=True) diff --git a/acp_adapter/server.py b/acp_adapter/server.py index 1b0046fe32..44d1df5ecc 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -78,6 +78,7 @@ from agent.context_compressor import ( COMPRESSED_SUMMARY_METADATA_KEY, ContextCompressor, ) +from agent.interrupt_compat import request_hard_interrupt from tools.approval import ( reset_hermes_interactive_context, set_hermes_interactive_context, @@ -1037,6 +1038,102 @@ class HermesACPAgent(acp.Agent): exc_info=True, ) + def _schedule_mcp_late_refresh(self, state: SessionState) -> None: + """Refresh the agent's tool snapshot when background MCP discovery lands late. + + ACP entry.py starts MCP tool discovery in a background daemon thread so a + slow/dead configured server can't block ``asyncio.run()``. ``_make_agent`` + briefly joins that thread (``wait_for_mcp_discovery``, bounded ~1.5s) so + already-spawning fast servers land in the snapshot — but a server slower + than the bound lands *after* the agent is built, leaving its tools absent + for the whole session. + + This schedules an off-critical-path daemon that waits for discovery to + finish (bounded 30s), then rebuilds the snapshot via the shared + ``refresh_agent_mcp_tools`` helper — the same rebuild ``/reload-mcp`` + performs, but automatic. Mirrors the TUI late-refresh (PR #48403). + + Cache safety: the rebuild only runs while the session is still + pre-first-turn (no API call made yet → nothing cached to invalidate). + Once the user has sent a message we leave the snapshot frozen rather + than break the cached prompt prefix mid-conversation; servers that land + later are picked up cache-safely by the between-turns prologue refresh + (``agent/turn_context.py``) at the next turn boundary. The marginal + value of this pre-first-turn daemon is therefore freshness in the + window [session created → first message] — e.g. the "Available tools" + listing a client may request before the first prompt. + No-op when discovery already finished, when the join times out, when the + registry was unchanged, or when the session was closed while waiting. + """ + try: + from hermes_cli.mcp_startup import mcp_discovery_in_flight + except Exception: + return + if not mcp_discovery_in_flight(): + return + + import threading + + agent = state.agent + session_id = state.session_id + + def _wait_then_refresh() -> None: + try: + from hermes_cli.mcp_startup import join_mcp_discovery + + if not join_mcp_discovery(timeout=30.0): + return + + # Session may have been closed while we waited. In-memory-only + # lookup on purpose: ``get_session()`` falls through to a DB + # restore that builds a whole new AIAgent as a side effect just + # to decide "no-op" here (the TUI equivalent also checks its + # in-memory dict only). + with self.session_manager._lock: + current = self.session_manager._sessions.get(session_id) + if current is None or current.agent is not agent: + return + + # Cache safety: never rebuild the tool list once the conversation + # has started — that would invalidate the cached prompt prefix. + # Serialized with turn start: ``prompt()`` flips ``is_running`` + # under ``runtime_lock`` before dispatching, so holding it here + # (and bailing when a turn is already running) closes the window + # where the guard passes but the first prompt starts before the + # refresh publishes — which would swap ``tools=`` mid-turn and + # break the just-created cache prefix. + with current.runtime_lock: + if current.is_running: + return + if ( + int(getattr(agent, "_user_turn_count", 0) or 0) > 0 + or int(getattr(agent, "_api_call_count", 0) or 0) > 0 + ): + return + + from tools.mcp_tool import refresh_agent_mcp_tools + + added = refresh_agent_mcp_tools(agent, quiet_mode=True) + if added: + logger.info( + "Session %s: late MCP refresh added %d tools: %s", + session_id, + len(added), + ", ".join(sorted(added)), + ) + except Exception: + logger.debug( + "Session %s: late MCP refresh failed", + session_id, + exc_info=True, + ) + + threading.Thread( + target=_wait_then_refresh, + name=f"acp-mcp-late-refresh-{session_id}", + daemon=True, + ).start() + # ---- ACP lifecycle ------------------------------------------------------ async def initialize( @@ -1343,6 +1440,7 @@ class HermesACPAgent(acp.Agent): ) -> NewSessionResponse: state = self.session_manager.create_session(cwd=cwd) await self._register_session_mcp_servers(state, mcp_servers) + self._schedule_mcp_late_refresh(state) logger.info("New session %s (cwd=%s)", state.session_id, cwd) self._schedule_available_commands_update(state.session_id) self._schedule_usage_update(state) @@ -1367,6 +1465,7 @@ class HermesACPAgent(acp.Agent): logger.warning("load_session: session %s not found", session_id) return None await self._register_session_mcp_servers(state, mcp_servers) + self._schedule_mcp_late_refresh(state) logger.info("Loaded session %s", session_id) # Per ACP spec, `session/load` must stream the prior conversation back # to the client via `session/update` notifications BEFORE responding, @@ -1414,6 +1513,7 @@ class HermesACPAgent(acp.Agent): logger.warning("resume_session: session %s not found, creating new", session_id) state = self.session_manager.create_session(cwd=cwd) await self._register_session_mcp_servers(state, mcp_servers) + self._schedule_mcp_late_refresh(state) logger.info("Resumed session %s", state.session_id) # See `load_session` above for the spec rationale — replay must # complete before the response so clients receive the full transcript @@ -1448,8 +1548,8 @@ class HermesACPAgent(acp.Agent): # redirectable work. state.cancel_event.set() try: - if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"): - state.agent.interrupt() + if getattr(state, "agent", None): + request_hard_interrupt(state.agent) except Exception: logger.debug( "Failed to interrupt ACP session %s", @@ -1768,8 +1868,11 @@ class HermesACPAgent(acp.Agent): # while the tools are rooted at the client's project, so the # model emits absolute paths under ~/.hermes/workspace and the # edit silently lands outside the editor's workspace. + # cron_session="" explicitly marks this as a non-cron context, + # masking any leaked process-global HERMES_CRON_SESSION (#37968). session_tokens = set_session_vars( - session_key=session_id, cwd=state.cwd, + session_key=session_id, session_id=session_id, cwd=state.cwd, + cron_session="", ) except Exception: session_tokens = None diff --git a/acp_adapter/session.py b/acp_adapter/session.py index 6f1e17a07f..6e16016a6b 100644 --- a/acp_adapter/session.py +++ b/acp_adapter/session.py @@ -648,6 +648,30 @@ class SessionManager: logger.debug("ACP session falling back to default provider resolution", exc_info=True) _register_task_cwd(session_id, cwd) + + # Bounded wait for background MCP discovery so already-spawning fast + # servers land in the agent's tool snapshot. ACP entry.py fires + # discovery in a background daemon thread (start_background_mcp_discovery); + # the agent snapshots tools once at build (run_agent/agent_init) and + # never re-reads the registry, so without this join a reachable-but- + # slow configured server would be invisible for the whole session. + # ``ensure_mcp_discovery_before_agent_build`` also (re)starts discovery + # when the entry.py spawn never ran or exited with zero connected + # servers (the retry-after-zero-connected allowance), making this + # construction site self-sufficient. Bounded by + # ``mcp_discovery_timeout`` (config.yaml, default ~1.5s) so a dead + # server can't block — servers that miss the bound are picked up by + # the automatic late-refresh (see HermesACPAgent._schedule_mcp_late_refresh). + try: + from hermes_cli.mcp_startup import ensure_mcp_discovery_before_agent_build + + ensure_mcp_discovery_before_agent_build( + logger=logger, + thread_name="acp-mcp-discovery", + ) + except Exception: + logger.debug("ACP: bounded MCP discovery wait failed", exc_info=True) + agent = AIAgent(**kwargs) # Codex app-server sessions are spawned lazily on the first turn. Stamp # the ACP workspace onto the agent so the Codex runtime starts from the diff --git a/agent/agent_init.py b/agent/agent_init.py index c24c5c65ff..db8f026f63 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -33,6 +33,7 @@ from urllib.parse import parse_qs, urlparse, urlunparse from agent.context_compressor import ContextCompressor from agent.iteration_budget import IterationBudget from agent.memory_manager import StreamingContextScrubber +from agent.session_activity import ActivityProvenance from agent.model_metadata import ( MINIMUM_CONTEXT_LENGTH, fetch_model_metadata, @@ -765,6 +766,9 @@ def init_agent( # Interrupt mechanism for breaking out of tool loops agent._interrupt_requested = False agent._interrupt_message = None # Optional message that triggered interrupt + # Explicit hard cancellation is separate from redirect/message state. A + # thread-safe Event makes the cause atomic for auxiliary stream pollers. + agent._hard_interrupt_requested = threading.Event() agent._execution_thread_id: int | None = None # Set at run_conversation() start agent._interrupt_thread_signal_pending = False agent._client_lock = threading.RLock() @@ -834,18 +838,34 @@ def init_agent( agent._use_prompt_caching, agent._use_native_cache_layout = ( agent._anthropic_prompt_cache_policy() ) + agent._cache_disabled = False # Anthropic supports "5m" (default) and "1h" cache TTL tiers. Read from # config.yaml under prompt_caching.cache_ttl; unknown values keep "5m". # 1h tier costs 2x on write vs 1.25x for 5m, but amortizes across long # sessions with >5-minute pauses between turns (#14971). + # + # Setting cache_ttl to a falsy value (false / null / "off" / "disabled" / + # "no" / "none") disables prompt caching entirely. This is useful for + # OAuth subscription users where cache writes bill against "extra usage" + # or for third-party proxies that inject their own cache_control markers + # (#13477). The disable propagates through anthropic_prompt_cache_policy() + # and restore_primary_runtime() so it survives /model switches and + # fallback re-derivation (#33555). agent._cache_ttl = "5m" try: from hermes_cli.config import load_config_readonly as _load_pc_cfg + from agent.agent_runtime_helpers import cache_ttl_means_disabled + _pc_cfg = _load_pc_cfg().get("prompt_caching", {}) or {} _ttl = _pc_cfg.get("cache_ttl", "5m") if _ttl in {"5m", "1h"}: agent._cache_ttl = _ttl + elif cache_ttl_means_disabled(_ttl): + agent._use_prompt_caching = False + agent._use_native_cache_layout = False + agent._cache_ttl = None + agent._cache_disabled = True except Exception: pass @@ -864,6 +884,11 @@ def init_agent( # notifications to show progress. agent._last_activity_ts: float = time.time() agent._last_activity_desc: str = "initializing" + # Default / unmigrated paths and _touch_activity stamp unknown; named + # provenances are stamped by compression writers (heartbeat / timeout / cooldown). + agent._last_activity_provenance = ActivityProvenance.UNKNOWN + # Rate-limit durable SessionDB activity stamps from _touch_activity (#72016). + agent._session_activity_last_persist_mono: float = 0.0 agent._current_tool: str | None = None agent._api_call_count: int = 0 # Opt-out flag for the between-turns MCP tool refresh (build_turn_context). @@ -1221,16 +1246,20 @@ def init_agent( _fb_entries = [fallback_model] _fb_resolved = False for _fb in _fb_entries: - _fb_explicit_key = (_fb.get("api_key") or "").strip() or None - if not _fb_explicit_key: - _fb_key_env = (_fb.get("key_env") or _fb.get("api_key_env") or "").strip() - if _fb_key_env: - _fb_explicit_key = os.getenv(_fb_key_env, "").strip() or None - _fb_client, _fb_model = resolve_provider_client( - _fb["provider"], model=_fb["model"], raw_codex=True, - explicit_base_url=_fb.get("base_url"), - explicit_api_key=_fb_explicit_key, - ) + try: + from hermes_cli.fallback_config import resolve_entry_api_key + _fb_explicit_key = resolve_entry_api_key(_fb) + _fb_client, _fb_model = resolve_provider_client( + _fb["provider"], model=_fb["model"], raw_codex=True, + explicit_base_url=_fb.get("base_url"), + explicit_api_key=_fb_explicit_key, + ) + except Exception as _fb_exc: + logger.debug( + "Init-time fallback entry %s failed: %s", + _fb.get("provider"), _fb_exc, + ) + continue if _fb_client is not None: agent.provider = _fb["provider"] agent.model = _fb_model or _fb["model"] @@ -1465,7 +1494,17 @@ def init_agent( set_current_session_id(agent.session_id) except Exception: - os.environ["HERMES_SESSION_ID"] = agent.session_id + # Preserve the root-agent legacy fallback, but never let delegated + # construction publish a child ID process-wide even if the ContextVar + # bridge itself failed to import. + try: + from agent.delegation_context import is_delegated_child_context + + delegated_child = is_delegated_child_context() + except Exception: + delegated_child = False + if not delegated_child: + os.environ["HERMES_SESSION_ID"] = agent.session_id # Session logs go into ~/.hermes/sessions/ alongside gateway sessions hermes_home = get_hermes_home() diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 1b592c2069..579314018c 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -151,7 +151,7 @@ def convert_to_trajectory_format(agent, messages: List[Dict[str, Any]], user_que except json.JSONDecodeError: # This shouldn't happen since we validate and retry during conversation, # but if it does, log warning and use empty dict - logger.warning(f"Unexpected invalid JSON in trajectory conversion: {tool_call['function']['arguments'][:100]}") + logger.warning("Unexpected invalid JSON in trajectory conversion: %s", tool_call['function']['arguments'][:100]) arguments = {} tool_call_json = { @@ -1210,7 +1210,7 @@ def recover_with_credential_pool( refreshed_id, ) return False, has_retried_429 - _ra().logger.info(f"Credential auth failure — refreshed pool entry {getattr(refreshed, 'id', '?')}") + _ra().logger.info("Credential auth failure — refreshed pool entry %s", getattr(refreshed, 'id', '?')) agent._swap_credential(refreshed) return True, has_retried_429 # Refresh failed — rotate to next credential instead of giving up. @@ -1475,6 +1475,12 @@ def restore_primary_runtime(agent) -> bool: "use_native_cache_layout", agent.api_mode == "anthropic_messages" and agent.provider == "anthropic", ) + # If the operator has disabled caching via config (cache_ttl is + # falsy → _cache_disabled flag is set), the disable must survive + # runtime snapshot restoration (#33555). + if getattr(agent, "_cache_disabled", False): + agent._use_prompt_caching = False + agent._use_native_cache_layout = False # ── Rebuild client for the primary provider ── if agent.provider == "moa": @@ -1829,11 +1835,148 @@ def dump_api_request_debug( return dump_file except Exception as dump_error: if agent.verbose_logging: - logger.warning(f"Failed to dump API request debug payload: {dump_error}") + logger.warning("Failed to dump API request debug payload: %s", dump_error) return None +def _direct_native_anthropic_tool_cache_capability( + agent, + *, + provider: Optional[str] = None, + base_url: Optional[str] = None, + api_mode: Optional[str] = None, + model: Optional[str] = None, +) -> bool: + """Return whether this resolved destination accepts native tool markers.""" + eff_base_url = base_url if base_url is not None else (agent.base_url or "") + eff_api_mode = api_mode if api_mode is not None else (agent.api_mode or "") + return ( + eff_api_mode == "anthropic_messages" + and base_url_hostname(eff_base_url) == "api.anthropic.com" + ) + + +def cache_ttl_means_disabled(ttl: Any) -> bool: + """Return True when a ``prompt_caching.cache_ttl`` value means caching off. + + Single source of truth for the disable-synonym detection shared by + ``agent_init`` (live-agent ``_cache_disabled`` flag) and the stub policy + paths below. Keeping one predicate prevents the two sites from drifting + (a synonym added in only one place would recreate #76085). + + Unknown values (e.g. ``"2h"``, integers) are NOT a disable — callers keep + caching enabled with the default TTL, matching ``agent_init``. + """ + if ttl in ("5m", "1h"): + return False + if ttl is False or ttl is None: + return True + return str(ttl).lower() in ("off", "false", "disabled", "no", "none") + + +def prompt_caching_disabled_from_config() -> bool: + """Return True when ``prompt_caching.cache_ttl`` is configured as off. + + Same disable detection as ``agent_init`` (via ``cache_ttl_means_disabled``) + so stub-based policy paths (MoA slot decoration, auxiliary fallback + replan) honor the same config contract without holding a live + ``AIAgent`` (#76085 / #33555). + """ + try: + from hermes_cli.config import load_config_readonly + + pc_cfg = load_config_readonly().get("prompt_caching", {}) or {} + ttl = pc_cfg.get("cache_ttl", "5m") + except Exception: + return False + return cache_ttl_means_disabled(ttl) + + +def blank_cache_policy_stub(cache_disabled: Optional[bool] = None): + """Build the destination-identity-blank stub for ``anthropic_prompt_cache_policy``. + + Single sanctioned constructor for that stub. Callers that resolve cache + policy against a destination identified out-of-band (not a live + ``AIAgent``) must go through here so ``_cache_disabled`` is never left + off a hand-rolled ``SimpleNamespace`` (#76085). + + When ``cache_disabled`` is omitted, falls back to the global config so + stub paths without an agent snapshot still honor an operator disable. + """ + from types import SimpleNamespace + + if cache_disabled is None: + cache_disabled = prompt_caching_disabled_from_config() + return SimpleNamespace( + provider="", + base_url="", + api_mode="", + model="", + _cache_disabled=bool(cache_disabled), + ) + + +def plan_cache_sections_for_destination( + messages: list, + tools: Optional[list], + *, + provider: str, + base_url: str, + api_mode: str, + model: str, + cache_disabled: Optional[bool] = None, +) -> Tuple[list, list]: + """Plan request-local cache sections for one resolved destination. + + Shared core of the synchronous acting-aggregator (MoA) and auxiliary + fallback senders: resolve the cache policy for the destination's real + provider/base_url/api_mode/model, then either return stripped canonical + copies (non-caching route) or a :func:`build_prompt_cache_plan` layout + (caching route, with the direct-native tool marker when the destination + is api.anthropic.com on the Messages wire). + + Never mutates ``messages`` or ``tools`` — both return values are + request-local copies. + + ``cache_disabled`` threads the operator's ``prompt_caching.cache_ttl`` + disable into the blank policy stub. When omitted, the live config is + consulted so MoA/auxiliary paths cannot re-enable markers after the + user turned caching off (#76085). + """ + from agent.prompt_caching import ( + build_prompt_cache_plan, + strip_anthropic_cache_control, + strip_anthropic_tool_cache_control, + ) + + stub = blank_cache_policy_stub(cache_disabled) + should_cache, native_layout = anthropic_prompt_cache_policy( + stub, + provider=provider, + base_url=base_url, + api_mode=api_mode, + model=model, + ) + if not should_cache: + canonical_messages = copy.deepcopy(messages or []) + strip_anthropic_cache_control(canonical_messages) + return canonical_messages, strip_anthropic_tool_cache_control(tools) + plan = build_prompt_cache_plan( + messages, + tools, + native_anthropic=native_layout, + direct_native_tool_cache=_direct_native_anthropic_tool_cache_capability( + stub, + provider=provider, + base_url=base_url, + api_mode=api_mode, + model=model, + ), + ) + return plan.messages, plan.tools + + def anthropic_prompt_cache_policy( agent, *, @@ -1860,13 +2003,25 @@ def anthropic_prompt_cache_policy( gateway implements the Anthropic cache_control contract (MiniMax, Zhipu GLM, LiteLLM's Anthropic proxy mode all do). - Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct - Alibaba (DashScope) also honour Anthropic-style ``cache_control`` - markers on OpenAI-wire chat completions. Upstream pi-mono #3392 / - pi #3393 documented this for opencode-go Qwen. Without markers - these providers serve zero cache hits, re-billing the full prompt - on every turn. + Qwen models on OpenCode and direct Alibaba (DashScope), plus DeepSeek + models on OpenCode, also honour Anthropic-style ``cache_control`` markers + on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi #3393 + documented this for opencode-go Qwen; #24617 reports the same gateway + contract for DeepSeek. Without markers these providers serve zero cache + hits, re-billing the full prompt on every turn. + + If the operator has set ``prompt_caching.cache_ttl`` to a falsy value + (``false``, ``null``, ``"off"``, etc.) in config.yaml, prompt caching + is fully disabled — this early return ensures the disable survives + ``/model`` switches, fallback re-derivation, and runtime snapshot + restoration (#33555). We check ``"_cache_disabled"`` (set by + init_agent when the disable is detected) rather than ``_cache_ttl`` + directly, because ``_cache_ttl`` is not yet set when the policy runs + during the initial ``init_agent`` call. """ + if getattr(agent, "_cache_disabled", False): + return (False, False) + eff_provider = (provider if provider is not None else agent.provider) or "" eff_base_url = base_url if base_url is not None else (agent.base_url or "") eff_api_mode = api_mode if api_mode is not None else (agent.api_mode or "") @@ -1980,16 +2135,22 @@ def anthropic_prompt_cache_policy( if is_minimax_provider or is_minimax_host: return True, True - # Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire - # transport that accepts Anthropic-style cache_control markers and - # rewards them with real cache hits. Without this branch - # qwen3.6-plus on opencode-go reports 0% cached tokens and burns - # through the subscription on every turn. + # Qwen on OpenCode (Zen/Go) and native DashScope, plus DeepSeek on + # OpenCode only: OpenAI-wire transports that accept Anthropic-style + # cache_control markers and reward them with real cache hits. Keep direct + # Alibaba specific to Qwen; its catalog does not establish the same + # contract for DeepSeek. model_is_qwen = "qwen" in model_lower + model_is_deepseek = "deepseek" in model_lower + provider_is_opencode = provider_lower in { + "opencode", "opencode-zen", "opencode-go", + } provider_is_alibaba_family = provider_lower in { "opencode", "opencode-zen", "opencode-go", "alibaba", } - if provider_is_alibaba_family and model_is_qwen: + if (provider_is_alibaba_family and model_is_qwen) or ( + provider_is_opencode and model_is_deepseek + ): # Envelope layout (native_anthropic=False): markers on inner # content parts, not top-level tool messages. Matches # pi-mono's "alibaba" cacheControlFormat. @@ -2082,6 +2243,31 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo # restore, request-scoped); auxiliary_client builds its own clients and keeps # SDK retries because it is NOT wrapped by the conversation loop. client_kwargs.setdefault("max_retries", 0) + # Defense-in-depth: guarantee Copilot requests carry the integration + # headers regardless of which build path we came through. The primary + # header wiring lives in `_apply_client_headers_for_base_url`, but two + # rebuild paths (`primary_recovery`, `restore_primary` in this module) + # reconstruct the client purely from a `_primary_runtime` snapshot and do + # NOT re-run that wiring. If the snapshot's client_kwargs ever lacks + # `default_headers` (older snapshot, header-less resolver result), the + # client goes out WITHOUT `Copilot-Integration-Id: vscode-chat`; the + # Copilot server then routes it to the "copilot-language-server" integrator + # whose model allowlist omits enterprise-only models (claude-opus-4.8) → + # HTTP 400 model_not_available_for_integrator on every turn. This chokepoint + # is the single place every primary OpenAI client passes through, so filling + # missing Copilot headers here closes the whole class. We only ADD missing + # keys — never override headers a caller deliberately set. + try: + if base_url_host_matches(str(client_kwargs.get("base_url", "")), "githubcopilot.com"): + from hermes_cli.models import copilot_default_headers + existing = dict(client_kwargs.get("default_headers") or {}) + existing_lower = {k.lower() for k in existing} + for hk, hv in copilot_default_headers().items(): + if hk.lower() not in existing_lower: + existing[hk] = hv + client_kwargs["default_headers"] = existing + except Exception: + _ra().logger.debug("Copilot default-header guard skipped", exc_info=True) # Uses the module-level `OpenAI` name, resolved lazily on first # access via __getattr__ below. Tests patch via `run_agent.OpenAI`. client = _ra().OpenAI(**client_kwargs) @@ -3773,6 +3959,9 @@ __all__ = [ "restore_primary_runtime", "extract_reasoning", "dump_api_request_debug", + "prompt_caching_disabled_from_config", + "blank_cache_policy_stub", + "plan_cache_sections_for_destination", "anthropic_prompt_cache_policy", "create_openai_client", "switch_model", diff --git a/agent/anthropic_adapter.py b/agent/anthropic_adapter.py index 0d59d94c9c..7457119921 100644 --- a/agent/anthropic_adapter.py +++ b/agent/anthropic_adapter.py @@ -24,6 +24,18 @@ from urllib.parse import urlparse from hermes_constants import get_hermes_home from typing import Any, Dict, List, Optional, Tuple from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars +from agent.secret_scope import get_secret as _get_secret + + +def _getenv(name: str, default: str = "") -> str: + """Profile-scoped replacement for os.getenv on credential reads. + + Routes through the secret scope (Workstream A): identical to os.getenv + when multiplexing is off, scope-aware (and fail-closed on an unscoped + read) when on. Mirrors the same wrapper in hermes_cli/runtime_provider.py. + """ + val = _get_secret(name, default) + return val if val is not None else default # NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls # ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.) @@ -1358,7 +1370,7 @@ def resolve_anthropic_token() -> Optional[str]: creds = read_claude_code_credentials() # 1. Hermes-managed OAuth/setup token env var - token = os.getenv("ANTHROPIC_TOKEN", "").strip() + token = _getenv("ANTHROPIC_TOKEN").strip() if token: preferred = _prefer_refreshable_claude_code_token(token, creds) if preferred: @@ -1366,7 +1378,7 @@ def resolve_anthropic_token() -> Optional[str]: return token # 2. CLAUDE_CODE_OAUTH_TOKEN (used by Claude Code for setup-tokens) - cc_token = os.getenv("CLAUDE_CODE_OAUTH_TOKEN", "").strip() + cc_token = _getenv("CLAUDE_CODE_OAUTH_TOKEN").strip() if cc_token: preferred = _prefer_refreshable_claude_code_token(cc_token, creds) if preferred: @@ -1385,7 +1397,7 @@ def resolve_anthropic_token() -> Optional[str]: # 5. Regular API key, or a legacy OAuth token saved in ANTHROPIC_API_KEY. # This remains as a compatibility fallback for pre-migration Hermes configs. - api_key = os.getenv("ANTHROPIC_API_KEY", "").strip() + api_key = _getenv("ANTHROPIC_API_KEY").strip() if api_key: return api_key @@ -1428,7 +1440,7 @@ def run_oauth_setup_token() -> Optional[str]: # Check env vars that may have been set for env_var in ("CLAUDE_CODE_OAUTH_TOKEN", "ANTHROPIC_TOKEN"): - val = os.getenv(env_var, "").strip() + val = _getenv(env_var).strip() if val: return val diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index e16c52ae3b..44d7d69683 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -14,6 +14,10 @@ Resolution order for text tasks (auto mode): 6. Direct API-key providers (z.ai/GLM, Kimi/Moonshot, MiniMax, MiniMax-CN) 7. None +OpenRouter fallback cost guard: ``auxiliary.free_only: true`` restricts the +step-2 fallback to ``:free`` SKUs; ``auxiliary.openrouter_model`` overrides +the default. A one-time WARNING is logged for non-``:free`` models. + Resolution order for vision/multimodal tasks (auto mode): 1. Selected main provider, if it is one of the supported vision backends below 2. OpenRouter @@ -42,6 +46,7 @@ Payment / credit exhaustion fallback: import contextlib import contextvars +import copy import functools import hashlib import inspect @@ -54,7 +59,7 @@ import time import uuid from pathlib import Path # noqa: F401 — used by test mocks from types import SimpleNamespace -from typing import Any, Callable, Dict, List, Optional, Tuple, TYPE_CHECKING +from typing import Any, Callable, Dict, List, NamedTuple, Optional, Tuple, TYPE_CHECKING from urllib.parse import urlparse, parse_qs, urlunparse # NOTE: `from openai import OpenAI` is deliberately NOT at module top — the @@ -223,30 +228,134 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any: # part-way, compression falls back to a static "summary unavailable" marker # and the real handoff is lost (#23975). A thread-local flag lets such a # task mark its in-flight LLM call as interrupt-protected; the Codex -# Responses stream's cancellation check honors it. TIMEOUTS still fire +# Responses stream's cancellation check honors it. An explicit host cancel +# (CLI Ctrl+C or /stop) may install a cancel check that overrides protection; +# ordinary incoming-message interrupts remain protected. TIMEOUTS still fire # (a hung call must die), and all OTHER aux tasks (vision, web_extract, # title_generation, …) remain freely interruptible. _aux_interrupt_protection = threading.local() +class AuxiliaryExplicitCancellation(BaseException): + """Frozen signal that an auxiliary attempt was explicitly hard-cancelled. + + This deliberately follows ``asyncio.CancelledError`` and inherits directly + from ``BaseException``: provider retry/fallback code catches ``Exception`` + broadly and must never reinterpret an explicit host stop as a transport + failure. ``cause`` is immutable class data so downstream compression code + does not re-query a mutable host Event after the transport has unwound. + """ + + cause = "explicit_host_cancel" + + def __init__(self) -> None: + super().__init__("auxiliary request explicitly cancelled by host") + + def _aux_interrupt_protected() -> bool: return bool(getattr(_aux_interrupt_protection, "active", False)) +def _aux_interrupt_cancel_requested() -> bool: + """Return whether an explicit host cancel overrides aux protection.""" + event = getattr(_aux_interrupt_protection, "cancel_event", None) + if event is not None: + try: + return bool(event.is_set()) + except Exception: + logger.debug("aux interrupt cancel event check failed", exc_info=True) + return False + check = getattr(_aux_interrupt_protection, "cancel_check", None) + if not callable(check): + return False + try: + return bool(check()) + except Exception: + logger.debug("aux interrupt cancel check failed", exc_info=True) + return False + + @contextlib.contextmanager -def aux_interrupt_protection(active: bool = True): +def aux_interrupt_protection( + active: bool = True, + cancel_check=None, + cancel_event=None, +): """Mark the current thread's auxiliary LLM call as interrupt-protected. Used by atomic aux tasks (compression) so a mid-flight gateway interrupt doesn't abort the call and trigger a degraded fallback. Re-entrant-safe: - restores the previous value on exit. + restores the previous value on exit. ``cancel_check`` lets the host retain + an explicit hard-cancel path; ``cancel_event`` is preferred when the host + already owns an Event. Nested protection scopes inherit both values. """ prev = getattr(_aux_interrupt_protection, "active", False) + prev_cancel_check = getattr(_aux_interrupt_protection, "cancel_check", None) + prev_cancel_event = getattr(_aux_interrupt_protection, "cancel_event", None) _aux_interrupt_protection.active = active + if callable(cancel_check): + _aux_interrupt_protection.cancel_check = cancel_check + if cancel_event is not None and callable(getattr(cancel_event, "is_set", None)): + _aux_interrupt_protection.cancel_event = cancel_event try: yield finally: _aux_interrupt_protection.active = prev + _aux_interrupt_protection.cancel_check = prev_cancel_check + _aux_interrupt_protection.cancel_event = prev_cancel_event + + +def _capture_aux_cancel_check() -> Optional[Callable[[], Any]]: + """Capture the current explicit-cancel source on the owning request thread.""" + event = getattr(_aux_interrupt_protection, "cancel_event", None) + is_set = getattr(event, "is_set", None) + if callable(is_set): + return is_set + check = getattr(_aux_interrupt_protection, "cancel_check", None) + if callable(check): + # Preserve callable identity so attempt-local decision objects retain + # methods such as begin_timeout_cleanup() when captured by adapters. + return check + return None + + +def _captured_aux_cancel_requested(cancel_check: Callable[[], Any]) -> bool: + """Read a request-thread cancellation source without leaking its failures.""" + try: + return bool(cancel_check()) + except Exception: + logger.debug("captured aux cancel check failed", exc_info=True) + return False + + +class _AuxiliaryCancellationDecision: + """Atomically choose explicit cancellation or provider timeout per attempt.""" + + def __init__(self, source_cancel_check: Callable[[], Any]) -> None: + self._source_cancel_check = source_cancel_check + self._lock = threading.Lock() + self._outcome = "active" + + def __call__(self) -> bool: + with self._lock: + if self._outcome == "cancelled": + return True + if self._outcome == "timed_out": + return False + if _captured_aux_cancel_requested(self._source_cancel_check): + self._outcome = "cancelled" + return True + return False + + def begin_timeout_cleanup(self) -> bool: + """Return whether timeout won and destructive cleanup is permitted.""" + with self._lock: + if self._outcome == "active": + if _captured_aux_cancel_requested(self._source_cancel_check): + self._outcome = "cancelled" + else: + self._outcome = "timed_out" + return self._outcome == "timed_out" # ── Forward-progress hook for streamed auxiliary calls ─────────────────── @@ -293,6 +402,75 @@ def aux_progress_hook(hook): _aux_progress.hook = prev +def _run_protected_sync_provider_call( + callback: Callable[[dict[str, Any]], Any], + kwargs: dict[str, Any], +) -> Any: + """Run one protected provider callback in an attempt-isolated daemon. + + A hard cancel must release the compression-owning thread promptly, but + auxiliary clients are process-shared and cannot safely be closed or evicted + to wake one request. Only protected calls with a captured hard-cancel source + use this seam. Their provider callback (including stream aggregation) runs + in a daemon worker while the owner polls cancellation. On cancel the owner + unwinds immediately; the worker is left to finish under the provider timeout + already present in ``kwargs``. It owns no transcript or compressor commit + state and never holds the session lock. + + Ordinary auxiliary calls, and protected calls without a cancellation source, + retain the historical direct synchronous path with no extra thread. + """ + source_cancel_check = _capture_aux_cancel_check() + if not _aux_interrupt_protected() or not callable(source_cancel_check): + return callback(kwargs) + + # Freeze one linearized outcome for this isolated attempt. The host Event is + # reused and cleared on a later turn, while the Codex timeout Timer may race + # owner polling. Both paths must decide under the same attempt-local lock. + cancel_check = _AuxiliaryCancellationDecision(source_cancel_check) + + if cancel_check(): + raise AuxiliaryExplicitCancellation() + + progress_hook = getattr(_aux_progress, "hook", None) + provider_context = contextvars.copy_context() + done = threading.Event() + outcome: dict[str, Any] = {} + + def _provider_worker() -> None: + try: + with aux_progress_hook(progress_hook), aux_interrupt_protection( + cancel_check=cancel_check + ): + outcome["result"] = callback(kwargs) + except BaseException as exc: + outcome["exception"] = exc + finally: + done.set() + + threading.Thread( + target=provider_context.run, + args=(_provider_worker,), + name="hermes-protected-aux-provider", + daemon=True, + ).start() + + while True: + # Cancellation is checked before and after every completion wait so it + # wins whenever result publication and the host Event become visible in + # the same polling interval. + if _captured_aux_cancel_requested(cancel_check): + raise AuxiliaryExplicitCancellation() + if not done.wait(0.02): + continue + if _captured_aux_cancel_requested(cancel_check): + raise AuxiliaryExplicitCancellation() + exception = outcome.get("exception") + if exception is not None: + raise exception + return outcome.get("result") + + def _safe_isinstance(obj: Any, maybe_type: Any) -> bool: """Return False instead of raising when a patched symbol is not a type.""" try: @@ -955,6 +1133,29 @@ def _nous_min_key_ttl_seconds() -> int: return 1800 +def _scoped_key_env(name: str) -> str: + """Read a provider API key env var through the profile secret scope. + + Auxiliary-client resolution runs both inside agent turns (secret scope + installed — its verdict is authoritative under multiplex, so a scoped + miss must NOT borrow another profile's process-env key) and on unscoped + startup/CLI probe paths, which keep the legacy ``os.environ`` read via + the ``UnscopedSecretError`` fallback (Slack pattern, #59739). + """ + if not name: + return "" + try: + from agent.secret_scope import UnscopedSecretError, get_secret + + try: + return (get_secret(name) or "").strip() + except UnscopedSecretError: + pass + except Exception: + pass + return (os.getenv(name) or "").strip() + + # ── Codex Responses → chat.completions adapter ───────────────────────────── # All auxiliary consumers call client.chat.completions.create(**kwargs) and # read response.choices[0].message.content. This adapter translates those @@ -1150,12 +1351,53 @@ class _CodexCompletionsAdapter: deadline = time.monotonic() + float(total_timeout) if total_timeout else None timed_out = threading.Event() timeout_timer: Optional[threading.Timer] = None + # A protected provider call may outlive its owning compression attempt: + # the owner returns promptly on hard cancellation while this adapter is + # still blocked in the SDK stream on its isolated worker. Timer threads + # do not inherit this worker's thread-local protection state, so freeze + # the hard-cancel source here, before creating the timer. + protected_cancel_check = ( + _capture_aux_cancel_check() if _aux_interrupt_protected() else None + ) + attempt_stream_lock = threading.Lock() + attempt_stream: List[Any] = [] def _timeout_message() -> str: return f"Codex auxiliary Responses stream exceeded {float(total_timeout):.1f}s total timeout" def _close_client_on_timeout() -> None: + begin_timeout_cleanup = getattr( + protected_cancel_check, "begin_timeout_cleanup", None + ) + if callable(begin_timeout_cleanup): + timeout_won = bool(begin_timeout_cleanup()) + else: + timeout_won = not ( + callable(protected_cancel_check) + and _captured_aux_cancel_requested(protected_cancel_check) + ) + # Publish transport timeout only after the attempt-local decision is + # fixed, so owner polling cannot observe completion in between. timed_out.set() + if not timeout_won: + # The request owner already hard-cancelled this attempt. The + # OpenAI client is process-shared, so closing/evicting it here + # would disrupt unrelated sessions. Wake only this attempt's + # event stream when responses.create() returned one in time; + # otherwise rely on the bounded SDK/provider timeout. + with attempt_stream_lock: + stream = attempt_stream[0] if attempt_stream else None + close_stream = getattr(stream, "close", None) + if callable(close_stream): + try: + close_stream() + except Exception: + logger.debug( + "Codex auxiliary: cancelled attempt stream close " + "during timeout failed", + exc_info=True, + ) + return close = getattr(self._client, "close", None) if callable(close): try: @@ -1182,11 +1424,14 @@ class _CodexCompletionsAdapter: from tools.interrupt import is_interrupted # Honor interrupt protection for atomic aux tasks (compression): # a mid-flight gateway interrupt must NOT abort the summary call - # and trigger a degraded fallback marker (#23975). Timeouts above - # still fire; other aux tasks remain interruptible. + # and trigger a degraded fallback marker (#23975). Explicit host + # cancellation has its own frozen exception; timeouts above still + # fire and other aux tasks remain interruptible. + if _aux_interrupt_cancel_requested(): + raise AuxiliaryExplicitCancellation() if is_interrupted() and not _aux_interrupt_protected(): raise InterruptedError("Codex auxiliary Responses stream interrupted") - except InterruptedError: + except (InterruptedError, AuxiliaryExplicitCancellation): raise except Exception: # Interrupt state is a best-effort UX hook; never make it a @@ -1225,12 +1470,39 @@ class _CodexCompletionsAdapter: _check_cancelled() event_stream = self._client.responses.create(**stream_kwargs) + with attempt_stream_lock: + attempt_stream.append(event_stream) + # The timer can fire while responses.create() is blocked. If the + # cancelled attempt had no stream to close at that instant, close it + # now that it is safely attempt-owned; never touch the shared client. + if ( + timed_out.is_set() + and callable(protected_cancel_check) + and _captured_aux_cancel_requested(protected_cancel_check) + ): + close_fn = getattr(event_stream, "close", None) + if callable(close_fn): + try: + close_fn() + except Exception: + logger.debug( + "Codex auxiliary: late cancelled attempt stream close failed", + exc_info=True, + ) try: - final = _consume_codex_event_stream( - event_stream, - model=resp_kwargs.get("model"), - on_event=_on_each_event, - ) + # Some Codex-compatible hosts accept ``stream=True`` but return + # a completed Responses object instead of an SSE iterator. Do + # not hand that object to the event consumer: typed Responses + # (and compatibility shims such as SimpleNamespace) are not + # event streams and may not be iterable at all. + if hasattr(event_stream, "output"): + final = event_stream + else: + final = _consume_codex_event_stream( + event_stream, + model=str(resp_kwargs.get("model") or model), + on_event=_on_each_event, + ) finally: close_fn = getattr(event_stream, "close", None) if callable(close_fn): @@ -1238,6 +1510,8 @@ class _CodexCompletionsAdapter: close_fn() except Exception: pass + with attempt_stream_lock: + attempt_stream.clear() if final is None: raise RuntimeError("Codex auxiliary Responses stream did not return a final response") @@ -2150,8 +2424,63 @@ def _resolve_api_key_provider() -> Tuple[Optional[OpenAI], Optional[str]]: # ── Provider resolution helpers ───────────────────────────────────────────── +_paid_lane_warned: set = set() + + +def _is_free_model(model: Optional[str]) -> bool: + """True when ``model`` is an OpenRouter free SKU (``:free`` suffix).""" + return bool(model) and str(model).strip().endswith(":free") + + +def _aux_openrouter_settings() -> Tuple[bool, str]: + """Read free_only and openrouter_model from config in one pass. + + Returns (free_only, model) — defaults (False, _OPENROUTER_MODEL) on any + config-read failure. + """ + try: + from hermes_cli.config import cfg_get, load_config_readonly + + cfg = load_config_readonly() + free_only = bool(cfg_get(cfg, "auxiliary", "free_only", default=False)) + val = cfg_get(cfg, "auxiliary", "openrouter_model") + model = val.strip() if isinstance(val, str) and val.strip() else _OPENROUTER_MODEL + return free_only, model + except Exception: + return False, _OPENROUTER_MODEL + + +def _warn_paid_lane_once(model: str) -> None: + """Log a WARNING the first time a non-:free OpenRouter model is engaged.""" + if model in _paid_lane_warned: + return + _paid_lane_warned.add(model) + logger.warning( + "Auxiliary client: PAID lane engaged for auxiliary task — OpenRouter " + "fallback model %r is not a :free SKU and may incur real spend. Set " + "auxiliary.free_only: true to restrict auxiliary fallbacks to free " + "models, or auxiliary.openrouter_model to a :free model.", + model, + ) + def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Optional[OpenAI], Optional[str]]: + free_only, cfg_model = _aux_openrouter_settings() + or_model = model or cfg_model + if free_only and not _is_free_model(or_model): + logger.warning( + "Auxiliary client: auxiliary.free_only is enabled but the " + "OpenRouter fallback model %r is not a :free SKU — skipping the " + "OpenRouter fallback. Set auxiliary.openrouter_model to a :free " + "model (e.g. nvidia/nemotron-3-ultra-550b-a55b:free) or disable " + "auxiliary.free_only.", + or_model, + ) + _mark_provider_unhealthy("openrouter", ttl=60) + return None, None + if not _is_free_model(or_model): + _warn_paid_lane_once(or_model) + pool_present, entry = _select_pool_entry("openrouter") if pool_present: or_key = explicit_api_key or _pool_runtime_api_key(entry) @@ -2159,18 +2488,18 @@ def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Op base_url = _pool_runtime_base_url(entry, OPENROUTER_BASE_URL) or OPENROUTER_BASE_URL logger.debug("Auxiliary client: OpenRouter via pool") return _create_openai_client(api_key=or_key, base_url=base_url, - default_headers=build_or_headers()), model or _OPENROUTER_MODEL + default_headers=build_or_headers()), or_model # Pool exists but is exhausted (no usable runtime key) — fall through to # the OPENROUTER_API_KEY env-var path rather than failing outright. logger.debug("Auxiliary client: OpenRouter pool exhausted, trying OPENROUTER_API_KEY") - or_key = explicit_api_key or os.getenv("OPENROUTER_API_KEY") + or_key = explicit_api_key or _scoped_key_env("OPENROUTER_API_KEY") if not or_key: _mark_provider_unhealthy("openrouter", ttl=60) return None, None logger.debug("Auxiliary client: OpenRouter") return _create_openai_client(api_key=or_key, base_url=OPENROUTER_BASE_URL, - default_headers=build_or_headers()), model or _OPENROUTER_MODEL + default_headers=build_or_headers()), or_model def _describe_openrouter_unavailable() -> str: @@ -2181,7 +2510,7 @@ def _describe_openrouter_unavailable() -> str: return "OpenRouter credential pool has no usable entries (credentials may be exhausted)" if not _pool_runtime_api_key(entry): return "OpenRouter credential pool entry is missing a runtime API key" - if not str(os.getenv("OPENROUTER_API_KEY") or "").strip(): + if not _scoped_key_env("OPENROUTER_API_KEY"): return "OPENROUTER_API_KEY not set" return "no usable OpenRouter credentials found" @@ -2607,14 +2936,17 @@ def _relay_sync_completion( ) -> Any: callback = create or (lambda request: client.chat.completions.create(**request)) route = _relay_auxiliary_metadata(provider=provider, api_mode=api_mode) + # Protected compression calls isolate only the provider callback and stream + # aggregation. The owning thread remains free to unwind its lease/DB + # transaction on hard cancel without touching the process-shared client. if route is None: - return callback(kwargs) + return _run_protected_sync_provider_call(callback, kwargs) provider_name, fallback_model, metadata = route from agent import relay_llm return relay_llm.execute_current( kwargs, - callback, + lambda request: _run_protected_sync_provider_call(callback, request), name=provider_name, model_name=str(kwargs.get("model") or fallback_model), metadata=metadata, @@ -2667,6 +2999,7 @@ def _relay_sync_stream( model_name=str(kwargs.get("model") or fallback_model), finalizer=dict, metadata=metadata, + completed_response_predicate=lambda value: hasattr(value, "choices"), ) _RUNTIME_MAIN_COMPAT_SNAPSHOT: Tuple[Any, ...] = ("", "", "", "", "", "") _RUNTIME_MAIN_COMPAT_LOCK = threading.Lock() @@ -2812,7 +3145,7 @@ def _resolve_custom_runtime() -> Tuple[Optional[str], Optional[str], Optional[st if not isinstance(runtime, dict): openai_base = os.getenv("OPENAI_BASE_URL", "").strip().rstrip("/") - openai_key = os.getenv("OPENAI_API_KEY", "").strip() + openai_key = _scoped_key_env("OPENAI_API_KEY") if not openai_base: return None, None, None runtime = { @@ -4174,6 +4507,27 @@ def _auth_refresh_provider_for_route( return normalized +def _fallback_chain_entry(task: Optional[str], fb_label: str) -> Optional[Dict[str, Any]]: + """Resolve the configured ``fallback_chain`` entry a label points at. + + Labels minted by :func:`_try_configured_fallback_chain` carry the entry + index in our own stable format (``fallback_chain[]()``). + Returns ``None`` when the label is not a configured-chain candidate or + the index no longer resolves to a dict entry. + """ + if not task or not fb_label: + return None + m = re.match(r"fallback_chain\[(\d+)\]", fb_label) + if not m: + return None + try: + chain = _get_auxiliary_task_config(task).get("fallback_chain") + entry = chain[int(m.group(1))] if isinstance(chain, list) else None + except Exception: + return None + return entry if isinstance(entry, dict) else None + + def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[float]: """Resolve a per-entry ``timeout`` for a configured fallback candidate. @@ -4185,29 +4539,113 @@ def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[floa primary's 30s deadline every turn (#62452). Entries in ``auxiliary..fallback_chain`` may declare their own - ``timeout`` (seconds). This helper reads it by parsing the entry index - out of the label minted by :func:`_try_configured_fallback_chain` - (``fallback_chain[]()`` — our own stable format). Returns - ``None`` when the label is not a configured-chain candidate, the entry - has no ``timeout``, or the value is invalid — callers then keep the - task-level timeout, preserving existing behavior. + ``timeout`` (seconds). Returns ``None`` when the label is not a + configured-chain candidate, the entry has no ``timeout``, or the value + is invalid — callers then keep the task-level timeout, preserving + existing behavior. """ - if not task or not fb_label: - return None - m = re.match(r"fallback_chain\[(\d+)\]", fb_label) - if not m: - return None - try: - chain = _get_auxiliary_task_config(task).get("fallback_chain") - entry = chain[int(m.group(1))] if isinstance(chain, list) else None - raw = entry.get("timeout") if isinstance(entry, dict) else None - except Exception: - return None + entry = _fallback_chain_entry(task, fb_label) + raw = entry.get("timeout") if entry else None if isinstance(raw, (int, float)) and not isinstance(raw, bool) and raw > 0: return float(raw) return None +def _fallback_provider_from_label(label: str) -> str: + """Recover the provider identifier from a fallback display label.""" + match = re.match(r"(?:fallback_chain\[\d+\]|main-agent)\(([^)]+)\)$", label or "") + return match.group(1).strip() if match else str(label or "").strip() + + +class _FallbackDestination(NamedTuple): + provider: str + base_url: str + api_mode: Optional[str] + model: Optional[str] + + +def _complete_fallback_destination( + provider: str, + base_url: str, + api_mode: Optional[str], + model: Optional[str], +) -> _FallbackDestination: + if not api_mode: + if _endpoint_speaks_anthropic_messages(base_url): + api_mode = "anthropic_messages" + else: + try: + from hermes_cli.runtime_provider import resolve_runtime_provider + + runtime = resolve_runtime_provider( + requested=provider, + explicit_base_url=base_url or None, + target_model=model or "", + ) + api_mode = str(runtime.get("api_mode") or "").strip() or None + except Exception: + pass + return _FallbackDestination(provider, base_url, api_mode, model) + + +def _fallback_destination_from_entry( + entry: Dict[str, Any], + fb_client: Any, + fb_model: Optional[str], +) -> _FallbackDestination: + provider = str(entry.get("provider") or "").strip() + base_url = str( + entry.get("base_url") or getattr(fb_client, "base_url", "") or "" + ).strip() + api_mode = str( + entry.get("api_mode") or entry.get("transport") or "" + ).strip() or None + model = fb_model or str(entry.get("model") or "").strip() or None + return _complete_fallback_destination(provider, base_url, api_mode, model) + + +def _fallback_destination( + task: Optional[str], + fb_client: Any, + fb_model: Optional[str], + fb_label: str, +) -> _FallbackDestination: + """Return the resolved route identity used by a fallback request.""" + attached = getattr(fb_client, "_hermes_fallback_destination", None) + if isinstance(attached, _FallbackDestination): + return attached + + provider = _fallback_provider_from_label(fb_label) + base_url = str(getattr(fb_client, "base_url", "") or "") + api_mode = None + model = fb_model + + entry = _fallback_chain_entry(task, fb_label) + if entry is not None: + return _fallback_destination_from_entry(entry, fb_client, fb_model) + + return _complete_fallback_destination(provider, base_url, api_mode, model) + + +def _replan_synchronous_cache_sections( + messages: list, + tools: Optional[list], + *, + destination: _FallbackDestination, +) -> tuple[list, list]: + """Strip source decoration and plan one synchronous destination locally.""" + from agent.agent_runtime_helpers import plan_cache_sections_for_destination + + return plan_cache_sections_for_destination( + messages, + tools, + provider=destination.provider, + base_url=destination.base_url, + api_mode=destination.api_mode or "", + model=destination.model or "", + ) + + def _call_fallback_candidate_sync( fb_client: Any, fb_model: Optional[str], @@ -4250,36 +4688,70 @@ def _call_fallback_candidate_sync( task or "call", fb_label, fb_timeout, effective_timeout, ) effective_timeout = fb_timeout - fb_base = str(getattr(fb_client, "base_url", "") or "") + destination = _fallback_destination(task, fb_client, fb_model, fb_label) + fallback_messages, fallback_tools = _replan_synchronous_cache_sections( + messages, + tools, + destination=destination, + ) fb_kwargs = _build_call_kwargs( - fb_label, fb_model, messages, + destination.provider, destination.model, fallback_messages, temperature=temperature, max_tokens=max_tokens, - tools=tools, timeout=effective_timeout, + tools=fallback_tools, timeout=effective_timeout, extra_body=effective_extra_body, reasoning_config=reasoning_config, - base_url=fb_base, task=task) + base_url=destination.base_url, task=task) try: return _validate_llm_response( - _relay_sync_completion(fb_client, fb_kwargs, provider=fb_label), task) + _relay_sync_completion( + fb_client, + fb_kwargs, + provider=destination.provider, + api_mode=destination.api_mode, + ), + task, + ) except Exception as fb_err: if not _is_auth_error(fb_err): raise - fb_provider = _auth_refresh_provider_for_route(fb_label, fb_base) + fb_provider = _auth_refresh_provider_for_route( + destination.provider, destination.base_url + ) if fb_provider not in {"auto", "", None} and _refresh_provider_credentials(fb_provider): - retry_client, retry_model = _get_cached_client(fb_provider, fb_model) + retry_client, retry_model = _get_cached_client( + fb_provider, + destination.model, + base_url=destination.base_url or None, + api_mode=destination.api_mode, + ) if retry_client is not None: + retry_destination = _FallbackDestination( + fb_provider, + destination.base_url + or str(getattr(retry_client, "base_url", "") or ""), + destination.api_mode, + retry_model or destination.model, + ) + retry_messages, retry_tools = _replan_synchronous_cache_sections( + messages, + tools, + destination=retry_destination, + ) retry_kwargs = _build_call_kwargs( - fb_provider, retry_model or fb_model, messages, + retry_destination.provider, + retry_destination.model, + retry_messages, temperature=temperature, max_tokens=max_tokens, - tools=tools, timeout=effective_timeout, + tools=retry_tools, timeout=effective_timeout, extra_body=effective_extra_body, reasoning_config=reasoning_config, - base_url=str(getattr(retry_client, "base_url", "") or fb_base), task=task) + base_url=retry_destination.base_url, task=task) try: return _validate_llm_response( _relay_sync_completion( retry_client, retry_kwargs, - provider=fb_provider, + provider=retry_destination.provider, + api_mode=retry_destination.api_mode, ), task, ) @@ -4322,43 +4794,71 @@ async def _call_fallback_candidate_async( task or "call", fb_label, fb_timeout, effective_timeout, ) effective_timeout = fb_timeout - fb_base = str(getattr(fb_client, "base_url", "") or "") + destination = _fallback_destination(task, fb_client, fb_model, fb_label) + fallback_messages, fallback_tools = _replan_synchronous_cache_sections( + messages, + tools, + destination=destination, + ) fb_kwargs = _build_call_kwargs( - fb_label, fb_model, messages, + destination.provider, destination.model, fallback_messages, temperature=temperature, max_tokens=max_tokens, - tools=tools, timeout=effective_timeout, + tools=fallback_tools, timeout=effective_timeout, extra_body=effective_extra_body, reasoning_config=reasoning_config, - base_url=fb_base, task=task) + base_url=destination.base_url, task=task) try: return _validate_llm_response( await _relay_async_completion( fb_client, fb_kwargs, - provider=fb_label, + provider=destination.provider, + api_mode=destination.api_mode, ), task, ) except Exception as fb_err: if not _is_auth_error(fb_err): raise - fb_provider = _auth_refresh_provider_for_route(fb_label, fb_base) + fb_provider = _auth_refresh_provider_for_route( + destination.provider, destination.base_url + ) if fb_provider not in {"auto", "", None} and _refresh_provider_credentials(fb_provider): retry_client, retry_model = _get_cached_client( - fb_provider, fb_model, async_mode=True) + fb_provider, + destination.model, + async_mode=True, + base_url=destination.base_url or None, + api_mode=destination.api_mode, + ) if retry_client is not None: + retry_destination = _FallbackDestination( + fb_provider, + destination.base_url + or str(getattr(retry_client, "base_url", "") or ""), + destination.api_mode, + retry_model or destination.model, + ) + retry_messages, retry_tools = _replan_synchronous_cache_sections( + messages, + tools, + destination=retry_destination, + ) retry_kwargs = _build_call_kwargs( - fb_provider, retry_model or fb_model, messages, + retry_destination.provider, + retry_destination.model, + retry_messages, temperature=temperature, max_tokens=max_tokens, - tools=tools, timeout=effective_timeout, + tools=retry_tools, timeout=effective_timeout, extra_body=effective_extra_body, reasoning_config=reasoning_config, - base_url=str(getattr(retry_client, "base_url", "") or fb_base), task=task) + base_url=retry_destination.base_url, task=task) try: return _validate_llm_response( await _relay_async_completion( retry_client, retry_kwargs, - provider=fb_provider, + provider=retry_destination.provider, + api_mode=retry_destination.api_mode, ), task, ) @@ -4735,14 +5235,15 @@ def _try_configured_fallback_for_unavailable_client( def _fallback_entry_api_key(entry: Dict[str, Any]) -> Optional[str]: - """Resolve inline or env-backed API key from a fallback-chain entry.""" - explicit = str(entry.get("api_key") or "").strip() - if explicit: - return explicit - key_env = str(entry.get("key_env") or entry.get("api_key_env") or "").strip() - if key_env: - return os.getenv(key_env, "").strip() or None - return None + """Resolve inline or env-backed API key from a fallback-chain entry. + + Delegates to the centralized, secret-scope-aware resolver so this path + doesn't leak another profile's credential via a raw ``os.getenv`` under + gateway multiplexing (see ``hermes_cli.fallback_config.resolve_entry_api_key``). + """ + from hermes_cli.fallback_config import resolve_entry_api_key + + return resolve_entry_api_key(entry) def _resolve_fallback_entry(entry: Dict[str, Any]) -> Tuple[Optional[Any], Optional[str]]: @@ -4754,13 +5255,21 @@ def _resolve_fallback_entry(entry: Dict[str, Any]) -> Tuple[Optional[Any], Optio base_url = str(entry.get("base_url") or "").strip() or None api_key = _fallback_entry_api_key(entry) api_mode = str(entry.get("api_mode") or entry.get("transport") or "").strip() or None - return resolve_provider_client( + client, resolved_model = resolve_provider_client( provider, model=model, explicit_base_url=base_url, explicit_api_key=api_key, api_mode=api_mode, ) + if client is not None: + try: + client._hermes_fallback_destination = _fallback_destination_from_entry( + entry, client, resolved_model + ) + except Exception: + pass + return client, resolved_model def _try_main_fallback_chain( @@ -5442,7 +5951,7 @@ def resolve_provider_client( custom_base = _to_openai_base_url(explicit_base_url).strip() custom_key = ( (explicit_api_key or "").strip() - or os.getenv("OPENAI_API_KEY", "").strip() + or _scoped_key_env("OPENAI_API_KEY") or _read_main_api_key_if_same_host(custom_base) or "no-key-required" # local servers don't need auth ) @@ -5538,7 +6047,7 @@ def resolve_provider_client( custom_key = (custom_entry.get("api_key") or "").strip() custom_key_env = (custom_entry.get("key_env") or custom_entry.get("api_key_env") or "").strip() if not custom_key and custom_key_env: - custom_key = os.getenv(custom_key_env, "").strip() + custom_key = _scoped_key_env(custom_key_env) custom_key = custom_key or "no-key-required" if custom_key == "no-key-required": logger.warning( @@ -6360,7 +6869,7 @@ def auxiliary_max_tokens_param(value: int, *, model: Optional[str] = None) -> di misses the case where a custom base URL serves e.g. ``gpt-5.4``. """ custom_base = _current_custom_base_url() - or_key = os.getenv("OPENROUTER_API_KEY") + or_key = _scoped_key_env("OPENROUTER_API_KEY") # Use max_completion_tokens for direct OpenAI-compatible providers that reject # max_tokens on newer GPT-4o/o-series/GPT-5-style models. _custom_host = base_url_hostname(custom_base) or "" @@ -6821,7 +7330,7 @@ def _resolve_task_provider_model( task_config.get("key_env") or task_config.get("api_key_env") or "" ).strip() if cfg_key_env: - cfg_api_key = os.getenv(cfg_key_env, "").strip() or None + cfg_api_key = _scoped_key_env(cfg_key_env) or None cfg_api_mode = str(task_config.get("api_mode", "")).strip() or None # 'auto' is a sentinel meaning "inherit from main runtime / auto-detect", not @@ -8130,6 +8639,16 @@ def call_llm( kwargs["stream"] = True if stream_options: kwargs["stream_options"] = stream_options + if task == "moa_aggregator" and isinstance(client, CodexAuxiliaryClient): + # CodexAuxiliaryClient (openai-codex, xai-oauth, and any other + # Responses-shim provider) consumes the provider stream internally + # and returns a completed response object. Routing that nested + # MoA stream through Relay's generic managed stream makes the + # manager iterate the completed SimpleNamespace itself (#55933). + # Return the provider call directly; the MoA facade converts a + # completed response into a one-chunk delta iterator at its + # boundary. + return client.chat.completions.create(**kwargs) return _relay_sync_stream( client, kwargs, diff --git a/agent/azure_identity_adapter.py b/agent/azure_identity_adapter.py index 9506715019..dd0f62ab97 100644 --- a/agent/azure_identity_adapter.py +++ b/agent/azure_identity_adapter.py @@ -367,11 +367,27 @@ def describe_active_credential(config: Optional[EntraIdentityConfig] = None, info["tenant_id_env"] = os.environ["AZURE_TENANT_ID"].strip() # Surface which env-var sources are present without minting yet. + # Credential-bearing vars (AZURE_CLIENT_SECRET, AZURE_FEDERATED_TOKEN_FILE) + # are read through the profile secret scope so a multiplexed profile's + # diagnostics don't report another profile's env-bridged credentials; + # unscoped CLI probes keep the legacy env read (Slack pattern). + def _scoped_env(name: str) -> str: + try: + from agent.secret_scope import UnscopedSecretError, get_secret + + try: + return (get_secret(name) or "").strip() + except UnscopedSecretError: + pass + except Exception: + pass + return os.environ.get(name, "").strip() + env_sources = [] - if os.environ.get("AZURE_FEDERATED_TOKEN_FILE", "").strip(): + if _scoped_env("AZURE_FEDERATED_TOKEN_FILE"): env_sources.append("WorkloadIdentityCredential (AZURE_FEDERATED_TOKEN_FILE)") if (os.environ.get("AZURE_CLIENT_ID", "").strip() - and os.environ.get("AZURE_CLIENT_SECRET", "").strip() + and _scoped_env("AZURE_CLIENT_SECRET") and os.environ.get("AZURE_TENANT_ID", "").strip()): env_sources.append("EnvironmentCredential (client secret)") if os.environ.get("IDENTITY_ENDPOINT", "").strip() or os.environ.get("MSI_ENDPOINT", "").strip(): diff --git a/agent/background_review.py b/agent/background_review.py index bc58ac59d9..01820cfffd 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -18,6 +18,7 @@ for invariants and PR review criteria. from __future__ import annotations +import copy import json import logging import os @@ -284,6 +285,15 @@ _SKILL_REVIEW_PROMPT = ( " • One-off task narratives. A user asking 'summarize today's " "market' or 'analyze this PR' is not a class of work that warrants " "a skill.\n\n" + " • Unresolved failures: if the session ended WITHOUT actually " + "finding a working method — you tried several things, none worked, " + "and told the user to check manually — do NOT write those attempts " + "up as a 'reliable workflow' or 'recommended approach'. That presents " + "an untested sequence of failures as validated guidance a future " + "session will trust and repeat. Either say 'Nothing to save', or, " + "only if you are independently confident of a real working alternative " + "(not something you are merely guessing might work), capture ONLY that " + "alternative — never the dead ends, and never dressed up as best practice.\n\n" "If a tool failed because of setup state, capture the FIX (install " "command, config step, env var to set) under an existing setup or " "troubleshooting skill — never 'this tool does not work' as a " @@ -377,6 +387,15 @@ _COMBINED_REVIEW_PROMPT = ( " • One-off task narratives. A user asking 'summarize today's " "market' or 'analyze this PR' is not a class of work that warrants " "a skill.\n\n" + " • Unresolved failures: if the session ended WITHOUT actually " + "finding a working method — you tried several things, none worked, " + "and told the user to check manually — do NOT write those attempts " + "up as a 'reliable workflow' or 'recommended approach'. That presents " + "an untested sequence of failures as validated guidance a future " + "session will trust and repeat. Either say 'Nothing to save', or, " + "only if you are independently confident of a real working alternative " + "(not something you are merely guessing might work), capture ONLY that " + "alternative — never the dead ends, and never dressed up as best practice.\n\n" "If a tool failed because of setup state, capture the FIX (install " "command, config step, env var to set) under an existing setup or " "troubleshooting skill — never 'this tool does not work' as a " @@ -727,6 +746,43 @@ def _run_review_in_thread( # _cached_system_prompt below. if not _routed: _fork_kwargs["reasoning_config"] = getattr(agent, "reasoning_config", None) + # Gateway session context is appended to the parent's cached + # system prompt at API-call time through this field. Preserve + # it on same-model forks so the complete effective system + # prompt remains byte-identical and can reuse the warm prefix. + _fork_kwargs["ephemeral_system_prompt"] = getattr( + agent, "ephemeral_system_prompt", None + ) + # Prefill messages are inserted immediately after the system + # message at API-call time (chat_completion_helpers.py / + # conversation_loop.py), so a parent with prefill configured + # (gateway prefill_messages_file) would otherwise diverge + # from the warm prefix at message index 1 — same bug class + # as the ephemeral prompt above, one position later. + # Deep copy: the unicode-error recovery path mutates + # prefill entries IN PLACE (_sanitize_messages_surrogates + # via conversation_loop), so sharing dicts would let a + # fork-side sanitize rewrite the parent's prefill bytes. + _parent_prefill = copy.deepcopy( + getattr(agent, "prefill_messages", None) or [] + ) + if _parent_prefill: + _fork_kwargs["prefill_messages"] = _parent_prefill + # OpenRouter provider-routing pins: prompt caches live per + # UPSTREAM provider, so a fork without the parent's pins can + # be routed to a different upstream and miss the warm cache + # even with byte-identical prompt/tools bytes. + for _pref_attr in ( + "providers_allowed", + "providers_ignored", + "providers_order", + "provider_sort", + "provider_require_parameters", + "provider_data_collection", + ): + _pref_val = getattr(agent, _pref_attr, None) + if _pref_val: + _fork_kwargs[_pref_attr] = _pref_val review_agent = AIAgent( model=_rt.get("model") or agent.model, max_iterations=16, diff --git a/agent/browser_provider.py b/agent/browser_provider.py index 75e88e584f..3e96fa9c85 100644 --- a/agent/browser_provider.py +++ b/agent/browser_provider.py @@ -26,6 +26,7 @@ Session metadata contract (preserved from the legacy ``CloudBrowserProvider``):: "session_name": str, # unique name for agent-browser --session "bb_session_id": str, # provider session ID (for close/cleanup) "cdp_url": str, # CDP websocket URL + "expires_at": str, # optional provider-authoritative ISO timestamp "features": dict, # feature flags that were enabled "external_call_id": str, # optional, managed-gateway billing key } @@ -96,6 +97,7 @@ class BrowserProvider(abc.ABC): "session_name": str, # unique name for agent-browser --session "bb_session_id": str, # provider session ID (for close/cleanup) "cdp_url": str, # CDP websocket URL + "expires_at": str, # optional provider-authoritative ISO timestamp "features": dict, # feature flags that were enabled } diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 1c49a59e2a..e6e8ab7fdc 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1120,9 +1120,10 @@ def interruptible_api_call(agent, api_kwargs: dict): -def build_api_kwargs(agent, api_messages: list) -> dict: +def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict: """Build the keyword arguments dict for the active API mode.""" - tools_for_api = agent.tools + if tools_for_api is None: + tools_for_api = agent.tools if agent.api_mode == "anthropic_messages": _transport = agent._get_transport() @@ -1788,19 +1789,17 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool # Pass base_url and api_key from fallback config so custom # endpoints (e.g. Ollama Cloud) resolve correctly instead of # falling through to OpenRouter defaults. + from hermes_cli.fallback_config import resolve_entry_api_key + fb_base_url_hint = (fb.get("base_url") or "").strip() or None - fb_api_key_hint = (fb.get("api_key") or "").strip() or None - if not fb_api_key_hint: - # key_env and api_key_env are both documented aliases (see - # _normalize_custom_provider_entry in hermes_cli/config.py). - fb_key_env = (fb.get("key_env") or fb.get("api_key_env") or "").strip() - if fb_key_env: - fb_api_key_hint = os.getenv(fb_key_env, "").strip() or None + fb_api_key_hint = resolve_entry_api_key(fb) # For Ollama Cloud endpoints, pull OLLAMA_API_KEY from env # when no explicit key is in the fallback config. Host match # (not substring) — see GHSA-76xc-57q6-vm5m. if fb_base_url_hint and base_url_host_matches(fb_base_url_hint, "ollama.com") and not fb_api_key_hint: - fb_api_key_hint = os.getenv("OLLAMA_API_KEY") or None + from agent.secret_scope import get_secret + + fb_api_key_hint = get_secret("OLLAMA_API_KEY") or None fb_client, _resolved_fb_model = resolve_provider_client( fb_provider, model=fb_model, raw_codex=True, explicit_base_url=fb_base_url_hint, @@ -2384,7 +2383,7 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str: final_response = "I reached the iteration limit and couldn't generate a summary." except Exception as e: - logger.warning(f"Failed to get summary response: {e}") + logger.warning("Failed to get summary response: %s", e) final_response = f"I reached the maximum iterations ({agent.max_iterations}) but couldn't summarize. Error: {str(e)}" finally: from agent import relay_llm @@ -2425,7 +2424,7 @@ def cleanup_task_resources(agent, task_id: str) -> None: _ra().cleanup_vm(task_id) except Exception as e: if agent.verbose_logging: - logger.warning(f"Failed to cleanup VM for task {task_id}: {e}") + logger.warning("Failed to cleanup VM for task %s: %s", task_id, e) try: headed = False try: @@ -2443,7 +2442,7 @@ def cleanup_task_resources(agent, task_id: str) -> None: _ra().cleanup_browser(task_id) except Exception as e: if agent.verbose_logging: - logger.warning(f"Failed to cleanup browser for task {task_id}: {e}") + logger.warning("Failed to cleanup browser for task %s: %s", task_id, e) def _build_partial_stream_stub( @@ -3979,7 +3978,7 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta= " To avoid this delay, set display.streaming: false " "in config.yaml\n" ) - logger.info( + logger.exception( "Streaming failed before delivery: %s", e, ) diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 23708eca97..8f64f64b76 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -14,6 +14,7 @@ import hashlib import json import logging import re +import unicodedata import uuid from types import SimpleNamespace from typing import Any, Dict, List, Optional @@ -73,6 +74,79 @@ _TOOL_CALL_LEAK_PATTERN = re.compile( ) +# The ChatGPT Codex backend reserves these Harmony wire tokens. If their +# literal spellings are replayed anywhere in request text, the backend rejects +# the request before inference with ``invalid_prompt: Request blocked.``. +# Category-Cf handling covers persisted sessions from an earlier U+200B weak +# defang; fullwidth bars survive format-character stripping while keeping the +# inspected source legible. +_HARMONY_CONTROL_TOKEN_RE = re.compile( + r"<\|(start|end|channel|message|constrain|return|call)\|>" +) +_FULLWIDTH_PIPE = "\uff5c" + + +def _neutralize_harmony_tokens(text: str) -> str: + """Keep Harmony source readable without emitting reserved wire tokens.""" + if not text or "<" not in text or "|" not in text: + return text + + replacement = rf"<{_FULLWIDTH_PIPE}\1{_FULLWIDTH_PIPE}>" + if not any(unicodedata.category(char) == "Cf" for char in text): + return _HARMONY_CONTROL_TOKEN_RE.sub(replacement, text) + + # U+200B is confirmed to be stripped by the Codex backend before its + # reserved-token check. Treat every Unicode format control equivalently so + # moving the character elsewhere in the token (or swapping in another Cf) + # cannot recreate the same visually hidden form. + visible_chars: List[str] = [] + original_positions: List[int] = [] + for index, char in enumerate(text): + if unicodedata.category(char) == "Cf": + continue + visible_chars.append(char) + original_positions.append(index) + + visible_text = "".join(visible_chars) + matches = list(_HARMONY_CONTROL_TOKEN_RE.finditer(visible_text)) + if not matches: + return text + + result: List[str] = [] + original_cursor = 0 + for match in matches: + original_start = original_positions[match.start()] + original_end = original_positions[match.end() - 1] + 1 + result.append(text[original_cursor:original_start]) + result.append(f"<{_FULLWIDTH_PIPE}{match.group(1)}{_FULLWIDTH_PIPE}>") + original_cursor = original_end + result.append(text[original_cursor:]) + return "".join(result) + + +def _neutralize_harmony_structure(value: Any) -> Any: + """Neutralize JSON-like values; normalize tuples and reject unsafe keys. + + Rewriting an object key could desynchronize a tool schema from the executor + contract, so a reserved token there is rejected explicitly instead. + """ + if isinstance(value, str): + return _neutralize_harmony_tokens(value) + if isinstance(value, (list, tuple)): + return [_neutralize_harmony_structure(item) for item in value] + if isinstance(value, dict): + normalized = {} + for key, item in value.items(): + if isinstance(key, str) and _neutralize_harmony_tokens(key) != key: + raise ValueError( + "Reserved Harmony tokens in a JSON object key cannot be " + "neutralized without changing its contract." + ) + normalized[key] = _neutralize_harmony_structure(item) + return normalized + return value + + # --------------------------------------------------------------------------- # Multimodal content helpers # --------------------------------------------------------------------------- @@ -627,10 +701,16 @@ def _preflight_codex_input_items( raw_items: Any, *, is_github_responses: bool = False, + sanitize_harmony_tokens: bool = False, ) -> List[Dict[str, Any]]: if not isinstance(raw_items, list): raise ValueError("Codex Responses input must be a list of input items.") + sanitize_text = ( + _neutralize_harmony_tokens + if sanitize_harmony_tokens + else lambda text: text + ) normalized: List[Dict[str, Any]] = [] seen_ids: set = set() for idx, item in enumerate(raw_items): @@ -651,7 +731,7 @@ def _preflight_codex_input_items( arguments = json.dumps(arguments, ensure_ascii=False) elif not isinstance(arguments, str): arguments = str(arguments) - arguments = arguments.strip() or "{}" + arguments = sanitize_text(arguments.strip() or "{}") normalized.append( { @@ -685,7 +765,7 @@ def _preflight_codex_input_items( if ptype == "input_text": text = part.get("text") if isinstance(text, str) and text: - cleaned.append({"type": "input_text", "text": text}) + cleaned.append({"type": "input_text", "text": sanitize_text(text)}) elif ptype == "input_image": url = part.get("image_url") if isinstance(url, str) and url: @@ -709,7 +789,7 @@ def _preflight_codex_input_items( { "type": "function_call_output", "call_id": call_id.strip(), - "output": output, + "output": sanitize_text(output), } ) continue @@ -722,14 +802,21 @@ def _preflight_codex_input_items( if item_id in seen_ids: continue seen_ids.add(item_id) - reasoning_item = {"type": "reasoning", "encrypted_content": encrypted} + reasoning_item: Dict[str, Any] = { + "type": "reasoning", + "encrypted_content": encrypted, + } # Do NOT include the "id" in the outgoing item — with # store=False (our default) the API tries to resolve the # id server-side and returns 404. The id is still used # above for local deduplication via seen_ids. summary = item.get("summary") if isinstance(summary, list): - reasoning_item["summary"] = summary + reasoning_item["summary"] = ( + _neutralize_harmony_structure(summary) + if sanitize_harmony_tokens + else summary + ) else: reasoning_item["summary"] = [] normalized.append(reasoning_item) @@ -758,7 +845,7 @@ def _preflight_codex_input_items( text = "" if not isinstance(text, str): text = str(text) - normalized_content.append({"type": "output_text", "text": text}) + normalized_content.append({"type": "output_text", "text": sanitize_text(text)}) if not normalized_content: raise ValueError(f"Codex Responses input[{idx}] message item must contain at least one text part.") normalized_item: Dict[str, Any] = { @@ -798,7 +885,7 @@ def _preflight_codex_input_items( for part_idx, part in enumerate(content): if isinstance(part, str): if part: - validated.append({"type": text_type, "text": part}) + validated.append({"type": text_type, "text": sanitize_text(part)}) continue if not isinstance(part, dict): raise ValueError( @@ -809,7 +896,7 @@ def _preflight_codex_input_items( text = part.get("text", "") if not isinstance(text, str): text = str(text or "") - validated.append({"type": text_type, "text": text}) + validated.append({"type": text_type, "text": sanitize_text(text)}) elif ptype in {"input_image", "image_url"}: image_ref = part.get("image_url", "") detail = part.get("detail") @@ -833,7 +920,7 @@ def _preflight_codex_input_items( if not isinstance(content, str): content = str(content) - normalized.append({"role": role, "content": content}) + normalized.append({"role": role, "content": sanitize_text(content)}) continue raise ValueError( @@ -848,6 +935,7 @@ def _preflight_codex_api_kwargs( *, allow_stream: bool = False, is_github_responses: bool = False, + sanitize_harmony_tokens: bool = False, ) -> Dict[str, Any]: if not isinstance(api_kwargs, dict): raise ValueError("Codex Responses request must be a dict.") @@ -868,10 +956,13 @@ def _preflight_codex_api_kwargs( if not isinstance(instructions, str): instructions = str(instructions) instructions = instructions.strip() or DEFAULT_AGENT_IDENTITY + if sanitize_harmony_tokens: + instructions = _neutralize_harmony_tokens(instructions) normalized_input = _preflight_codex_input_items( api_kwargs.get("input"), is_github_responses=is_github_responses, + sanitize_harmony_tokens=sanitize_harmony_tokens, ) tools = api_kwargs.get("tools") @@ -928,6 +1019,9 @@ def _preflight_codex_api_kwargs( } ) + if sanitize_harmony_tokens and normalized_tools is not None: + normalized_tools = _neutralize_harmony_structure(normalized_tools) + store = api_kwargs.get("store", False) if store is not False: raise ValueError("Codex Responses contract requires 'store' to be false.") diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index 7b8bed6137..c01084c4a4 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -1365,12 +1365,6 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta on_event=_on_event, interrupt_check=_interrupt_or_superseded, ) - # The terminal SSE frame is contractually last. Request the - # end-of-stream marker so Relay can run its response finalizer - # and close the physical attempt scope before Hermes returns. - if not agent._interrupt_requested: - for _ignored in event_stream: - pass except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc: if attempt < max_stream_retries: logger.debug( @@ -1386,6 +1380,25 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta return event_stream.final_response raise + # A terminal response has already been assembled at this point + # (``final`` is built), so a transport error while draining the + # rest of the iterator — done only to let Relay run its response + # finalizer — must NOT discard it or trigger a new physical + # request. Record it as a non-fatal finalization warning and + # still return the already-completed, already-billed response. + if not agent._interrupt_requested: + try: + for _ignored in event_stream: + pass + except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc: + logger.warning( + "Codex Responses stream transport finalization failed " + "after a terminal response was already received; " + "returning the completed response instead of " + "retrying. %s error=%s", + agent._client_log_context(), exc, + ) + if final.status in {"incomplete", "failed"}: logger.warning( "Codex Responses stream terminal status=%s " diff --git a/agent/context_compressor.py b/agent/context_compressor.py index c4a58c10ac..fbb7e6c5e8 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -25,7 +25,12 @@ import time import uuid from typing import Any, Dict, List, Optional -from agent.auxiliary_client import call_llm, _is_connection_error, aux_interrupt_protection +from agent.auxiliary_client import ( + AuxiliaryExplicitCancellation, + _is_connection_error, + aux_interrupt_protection, + call_llm, +) from agent.context_engine import ContextEngine, sanitize_memory_context from agent.error_classifier import FailoverReason, classify_api_error from agent.model_metadata import ( @@ -636,6 +641,12 @@ _ACTIVE_TASK_MAX_CHARS = 1400 # high for small/light tails, but using all 20 as a hard floor here would bring # back the old large-tool-output case where nothing can be compacted. _MAX_TAIL_MESSAGE_FLOOR = 8 + +# Pre-LLM feasibility skip (#60451): when the compressible middle is below +# this fraction of threshold_tokens (and a prior real-usage ineffectiveness +# strike exists), skip the LLM summary call — deterministic dropping alone +# recovers the negligible savings such a summary could deliver. +_FEASIBILITY_SKIP_MIDDLE_FRACTION = 0.10 # Under context pressure (protected-tail tool bodies alone exceed the soft # tail budget), demote large completed tool/file outputs even inside the # protected region — but always keep this many trailing messages verbatim so @@ -764,15 +775,52 @@ def _serialized_length_for_budget(value: Any) -> int: # Responses sessions in particular carry ``codex_reasoning_items`` blobs of # ``encrypted_content`` that can dominate the serialized session (a measured # 214-turn session held ~115K tokens / 27% of its payload there — #55572). +# +# ``reasoning_details`` is handled separately (see +# ``_reasoning_details_text_chars``): its signed/base64 envelope is excluded +# from the budget, mirroring the preflight estimator's exclusion in +# ``model_metadata._estimate_message_tokens_without_images`` (#73298). _REPLAY_BUDGET_KEYS = ( "reasoning", "reasoning_content", - "reasoning_details", "codex_reasoning_items", "codex_message_items", ) +def _reasoning_details_text_chars(value: Any) -> int: + """Textual thinking chars inside a ``reasoning_details`` envelope. + + ``reasoning_details`` carries provider thinking blocks: the actual + thinking TEXT plus opaque signed/base64 envelope blobs (Anthropic + ``signature``, redacted ``data``, encrypted payloads). The envelope is + never billed at anything near chars/4 by the provider and — on every + transport except Codex Responses — is replayed for at most the newest + assistant turn, so charging it on every message inflated the tail-budget + walk and silently shrank the surviving tail (#73298, second site). + + Count only the thinking text (the #51800 lesson: real reasoning text + MUST stay visible to the budget), skip everything else. + """ + if not value: + return 0 + if isinstance(value, str): + return len(value) + total = 0 + if isinstance(value, dict): + value = [value] + if isinstance(value, list): + for part in value: + if isinstance(part, str): + total += len(part) + elif isinstance(part, dict): + for text_key in ("thinking", "text", "summary"): + text = part.get(text_key) + if isinstance(text, str): + total += len(text) + return total + + def _estimate_msg_budget_tokens(msg: dict) -> int: """Token estimate for one message in the tail-protection budget walks. @@ -804,6 +852,17 @@ def _estimate_msg_budget_tokens(msg: dict) -> int: tokens += estimate_tokens_rough(str(tc)) for key in _REPLAY_BUDGET_KEYS: tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN + # reasoning_details: charge only the thinking TEXT, never the signed / + # base64 envelope (#73298 second site; mirrors the preflight estimator's + # exclusion in model_metadata). When the same thinking text already rides + # in ``reasoning``/``reasoning_content`` (measured byte-identical on + # Anthropic-wire sessions), skip it here entirely so the prose is not + # charged twice on top of the envelope exclusion. + if not (msg.get("reasoning") or msg.get("reasoning_content")): + tokens += ( + _reasoning_details_text_chars(msg.get("reasoning_details")) + // _CHARS_PER_TOKEN + ) return tokens @@ -1281,11 +1340,13 @@ class ContextCompressor(ContextEngine): self._consecutive_timeout_failures = 0 self._last_summary_dropped_count = 0 self._last_summary_fallback_used = False + self._last_feasibility_skip = False self._last_aux_model_failure_error = None self._last_aux_model_failure_model = None self._last_compression_savings_pct = 100.0 self._ineffective_compression_count = 0 self._anti_thrash_recovery_deadline = 0.0 + self._prellm_skip_count = 0 self._fallback_compression_streak = 0 self._verify_compaction_cleared_threshold = False self._last_compression_made_progress = False @@ -1337,6 +1398,7 @@ class ContextCompressor(ContextEngine): "protected_head_tokens": None, "protected_tail_tokens": None, "middle_window_tokens": None, + "prellm_skip_count": 0, "aux_prompt_tokens": None, "aux_output_reservation": None, "aux_provider": "", @@ -1547,11 +1609,13 @@ class ContextCompressor(ContextEngine): self._consecutive_timeout_failures = 0 self._last_summary_dropped_count = 0 self._last_summary_fallback_used = False + self._last_feasibility_skip = False self._last_aux_model_failure_error = None self._last_aux_model_failure_model = None self._last_compression_savings_pct = 100.0 self._ineffective_compression_count = 0 self._anti_thrash_recovery_deadline = 0.0 + self._prellm_skip_count = 0 self._fallback_compression_streak = 0 self._verify_compaction_cleared_threshold = False self._last_compression_made_progress = False @@ -1578,6 +1642,7 @@ class ContextCompressor(ContextEngine): self._consecutive_timeout_failures = 0 self._fallback_compression_streak = 0 self._ineffective_compression_count = 0 + self._prellm_skip_count = 0 self._anti_thrash_recovery_deadline = 0.0 self.get_active_compression_failure_cooldown() self._load_fallback_compression_streak() @@ -1722,9 +1787,34 @@ class ContextCompressor(ContextEngine): self._ineffective_compression_count = count self._persist_ineffective_compression_count() - def record_completed_compaction(self, *, used_fallback: bool = False) -> None: - """Record one completed boundary and its summary quality.""" + def record_completed_compaction( + self, *, used_fallback: bool = False, feasibility_skip: bool = False, + ) -> None: + """Record one completed boundary and its summary quality. + + ``feasibility_skip=True`` marks a deliberate pre-LLM skip (#60451): + the boundary is streak-NEUTRAL for ``_fallback_compression_streak`` + (neither incremented nor reset). It still arms the real-usage + effectiveness verdict (``_verify_compaction_cleared_threshold``) on + purpose — a skipped-summary drop that fails to clear the threshold is + exactly the incompressible-transcript case the ineffective-strike + breaker exists for, and its recovery probe bounds the block. + """ self._verify_compaction_cleared_threshold = True + if feasibility_skip: + # A deliberate pre-LLM feasibility skip (#60451) is not a + # summary-quality verdict: it must neither extend a fallback + # streak (two skips would otherwise latch the >= 2 breaker and + # disable compression entirely — including the cheap deterministic + # dropping the skip exists to reach) nor reset one (a skip proves + # nothing about the summary model's health). + if not self.quiet_mode: + logger.info( + "Compaction completed via pre-LLM feasibility skip; " + "fallback_compression_streak unchanged (%d)", + self._fallback_compression_streak, + ) + return if used_fallback: self._fallback_compression_streak += 1 if not self.quiet_mode: @@ -1743,6 +1833,11 @@ class ContextCompressor(ContextEngine): refresh: bool = False, ) -> Optional[Dict[str, Any]]: """Return the live compression-failure cooldown for the bound session.""" + if refresh: + # Transaction rollback must distinguish an authoritative empty row + # from a failed/unavailable durable read. The public return value + # cannot do so because it deliberately falls back to local state. + self._last_cooldown_refresh_was_authoritative = None now_mono = time.monotonic() local_state = None if self._summary_failure_cooldown_until > now_mono: @@ -1767,10 +1862,16 @@ class ContextCompressor(ContextEngine): try: state = getter(session_id) except sqlite3.Error as exc: + if refresh: + self._last_cooldown_refresh_was_authoritative = False logger.debug("compression failure cooldown lookup failed: %s", exc) return local_state except Exception: + if refresh: + self._last_cooldown_refresh_was_authoritative = False return local_state + if refresh: + self._last_cooldown_refresh_was_authoritative = True if not state: if refresh: if local_state is not None and self._cooldown_persist_failed: @@ -1831,7 +1932,43 @@ class ContextCompressor(ContextEngine): self._cooldown_persist_failed = True logger.debug("compression failure cooldown persist failed (non-sqlite): %s", exc) + def record_timeout_failure(self, error: str) -> None: + """Record a consecutive timeout failure using the shared cooldown ladder. + + Used by both the summary-LLM exception handler (inline at line ~3714) + and the host-level ``compress_context`` timeout wrapper in + ``run_compress_context_with_progress_timeout``. Avoids re-implementing + the ladder at each call site (#62452). + """ + _TIMEOUT_COOLDOWN_LADDER = (60, 300, 900) + self._consecutive_timeout_failures = ( + getattr(self, "_consecutive_timeout_failures", 0) + 1 + ) + cooldown = _TIMEOUT_COOLDOWN_LADDER[ + min(self._consecutive_timeout_failures, + len(_TIMEOUT_COOLDOWN_LADDER)) - 1 + ] + self._record_compression_failure_cooldown(float(cooldown), error) + def _clear_compression_failure_cooldown(self) -> None: + # #76354 review F4: fence check BEFORE cooldown-clear. A late worker + # whose host already timed out (and recorded a timeout cooldown) must + # not undo that cooldown when its summary eventually succeeds. The + # hook is installed by compress_context for the duration of the + # fenced call; when it reports cancellation, keep the host's cooldown. + cancelled_check = getattr(self, "_compression_cancelled_check", None) + if callable(cancelled_check): + try: + if cancelled_check(): + logger.info( + "Skipping compression cooldown clear: host already " + "cancelled this compression attempt" + ) + return + except Exception: + logger.debug( + "compression cancellation check failed", exc_info=True + ) self._summary_failure_cooldown_until = 0.0 self._last_summary_error = None self._consecutive_timeout_failures = 0 @@ -1934,6 +2071,7 @@ class ContextCompressor(ContextEngine): # trigger invalidates them. Keep the durable copy in sync so a # restart doesn't resurrect strikes this recalibration just voided. self._record_ineffective_compression_verdict(0) + self._prellm_skip_count = 0 if runtime_changed: self._fallback_compression_streak = 0 self._persist_fallback_compression_streak() @@ -2166,6 +2304,10 @@ class ContextCompressor(ContextEngine): self._micro_compact_consecutive_failures: int = 0 self._micro_compact_last_failure_cursor: int = -1 self._micro_compact_defrag_threshold_tokens: int = 2000 + # Set by _defrag_rolling_summary when it pops _DB_PERSISTED_MARKER + # from a live dict in place; consumed by finalize_turn to invalidate + # the agent's bounded flush-scan cursor (sibling of the #75170 site). + self._flush_scan_cursor_invalidated: bool = False self._micro_compact_passes: int = 0 self._micro_compact_tokens_saved_total: int = 0 # Cadence: run a pass every Nth completed turn. Each pass rewrites @@ -2227,6 +2369,9 @@ class ContextCompressor(ContextEngine): # restart with a persisted tripped counter (#69872) waits a full fresh # window before probing (#54923: restart must never disarm a guard). self._anti_thrash_recovery_deadline: float = 0.0 + # Pre-LLM feasibility skips (#60451). Observability only; NEVER feeds + # the ineffectiveness strike latch or the fallback streak breaker. + self._prellm_skip_count: int = 0 # Consecutive completed deterministic-fallback boundaries. Unlike the # real-usage effectiveness counter, ordinary fitting responses must not # reset this breaker; only a healthy completed summary does. @@ -2248,6 +2393,7 @@ class ContextCompressor(ContextEngine): # (gateway hygiene, /compress) can surface a visible warning. self._last_summary_dropped_count: int = 0 self._last_summary_fallback_used: bool = False + self._last_feasibility_skip: bool = False # When summary generation fails we now ABORT compression entirely # and return the original messages unchanged instead of dropping # the middle window with a static placeholder. Callers inspect @@ -4236,6 +4382,12 @@ This compaction should PRIORITISE preserving all information related to the focu end: int, ) -> list[tuple[int, str]]: """Find handoff summaries inside a compression window.""" + n = len(messages) + # Defensive: clamp bounds so a caller passing an out-of-range end + # (e.g. tail-cut returning len(messages)+1 when head_end >= n) + # cannot trigger IndexError. (#75588) + start = max(0, min(start, n)) + end = max(start, min(end, n)) summaries: list[tuple[int, str]] = [] for idx in range(start, end): content = messages[idx].get("content") @@ -5005,7 +5157,7 @@ This compaction should PRIORITISE preserving all information related to the focu # exists to prevent. Re-align FORWARD (never backward, which would give # the floor's message back) so a raised cut skips to the end of the # group and the whole call/result pair is summarised together. - return self._align_boundary_forward(messages, max(cut_idx, head_end + 1)) + return min(n, self._align_boundary_forward(messages, max(cut_idx, head_end + 1))) # ------------------------------------------------------------------ # ContextEngine: manual /compress preflight @@ -5322,6 +5474,15 @@ This compaction should PRIORITISE preserving all information related to the focu # Content changed after a possible flush — clear the persisted # stamp so the DB sync/flush rewrites the row. entry.pop(_DB_PERSISTED_MARKER, None) + # Sibling of the finalize_turn pop site (#75170): this pop + # also strips the marker from a LIVE dict in place, so the + # bounded flush-scan cursor would identity-skip the rewritten + # marker and the defragged summary would never reach state.db. + # The compressor holds no agent reference, so raise a flag the + # finalizer consumes to invalidate agent._db_flush_scan_prefix. + # (The pop sites at module scope — fresh copies in + # strip-marker helpers — break identity and need no flag.) + self._flush_scan_cursor_invalidated = True break logger.info( "Micro-compaction defrag: rolling summary re-summarized " @@ -5784,7 +5945,11 @@ This compaction should PRIORITISE preserving all information related to the focu 1. Prune old tool results (cheap pre-pass, no LLM call) 2. Protect head messages (system prompt + first exchange) 3. Find tail boundary by token budget (~20K tokens of recent context) - 4. Summarize middle turns with structured LLM prompt + 4. Summarize middle turns with structured LLM prompt (skipped + pre-LLM when the middle is below + ``_FEASIBILITY_SKIP_MIDDLE_FRACTION`` of the threshold after a + prior real-usage ineffectiveness strike — the deterministic + fallback drop recovers the negligible savings instead) 5. On re-compression, iteratively update the previous summary Blank platform-echo user rows trailing the latest actionable user @@ -5804,7 +5969,9 @@ This compaction should PRIORITISE preserving all information related to the focu everything else. Inspired by Claude Code's ``/compact``. force: If True, clear any active summary-failure cooldown before running so a manual ``/compress`` can retry immediately after - an auto-compression abort. Auto-compress callers pass False. + an auto-compression abort, and bypass the pre-LLM feasibility + skip so an explicit user request always exercises the full + summary path. Auto-compress callers pass False. memory_context: Optional provider-supplied context to preserve in the summary prompt. Whitespace-only values are ignored. """ @@ -5812,6 +5979,7 @@ This compaction should PRIORITISE preserving all information related to the focu # after compress() returns to decide whether to surface a warning. self._last_summary_dropped_count = 0 self._last_summary_fallback_used = False + self._last_feasibility_skip = False self._last_summary_error = None self._last_aux_model_failure_error = None self._last_aux_model_failure_model = None @@ -5940,6 +6108,7 @@ This compaction should PRIORITISE preserving all information related to the focu # — take the narrow rescan, miss a beyond-window fossil, and discard the # rehydrated state as cross-session leakage (#57835). _previous_summary_before_scan = self._previous_summary + _summary_has_user_turn_before_scan = getattr(self, "_summary_has_user_turn", None) # A persisted handoff summary can sit in the protected head after a # resume (commonly immediately after the system prompt). Search from # the first non-system message through the compression window. On the @@ -6094,12 +6263,73 @@ This compaction should PRIORITISE preserving all information related to the focu ) # Phase 3: Generate structured summary - summary_focus_topic = focus_topic or self._derive_auto_focus_topic(messages) - summary = self._generate_summary( - turns_to_summarize, - focus_topic=summary_focus_topic, - memory_context=memory_context, - ) + + # Pre-LLM feasibility check: if the middle section is too small to + # yield meaningful token savings, skip the expensive LLM summarization + # call and fall through to the deterministic message-dropping path + # (which is cheap and always applicable). Without this guard a + # tool-heavy session where the protected tail already holds most of + # the tokens can burn 500+ seconds on a summary call that replaces a + # few lightweight messages, leaving the total token count essentially + # unchanged. + # + # Only fires after at least one prior real-usage ineffectiveness + # strike. The check READS ``_ineffective_compression_count`` but + # never writes it: that strike counter is fed exclusively by real + # provider token counts (see the anti-thrashing verdict in + # _update_token_usage), and consumers latch at >= 2 to disable + # compression entirely. Feasibility skips are tracked separately + # in ``_prellm_skip_count`` for observability. + # + # Skipped when ``force=True`` (manual /compress) so auth/error + # handling paths are always exercised on explicit user request. + feasibility_skip = False + if not force and self._ineffective_compression_count >= 1: + # _record_compression_regions already estimated this exact window + # into the telemetry dict above; reuse it so the log line and + # telemetry can never disagree. The regions helper no-ops when the + # telemetry attr isn't a dict, so fall back to a fresh estimate + # when the key is absent/None (0 is a legitimate value). + middle_tokens = telemetry.get("middle_window_tokens") + if middle_tokens is None: + middle_tokens = estimate_messages_tokens_rough(turns_to_summarize) + if middle_tokens < int( + self.threshold_tokens * _FEASIBILITY_SKIP_MIDDLE_FRACTION + ): + feasibility_skip = True + self._last_feasibility_skip = True + self._prellm_skip_count += 1 + telemetry["prellm_skip_count"] = self._prellm_skip_count + if not self.quiet_mode: + logger.warning( + "Compression: middle section (%d tokens at indices " + "%d-%d) is below %.0f%% of threshold (%d tokens) — " + "skipping LLM summarization, proceeding with " + "deterministic message dropping. prellm_skip_count=%d", + middle_tokens, compress_start, compress_end, + _FEASIBILITY_SKIP_MIDDLE_FRACTION * 100, + self.threshold_tokens, self._prellm_skip_count, + ) + + if feasibility_skip: + summary = None # No LLM call; Phase 4 inserts the deterministic fallback + else: + # Deriving the auto focus topic scans recent user turns — only pay + # for it when a summary will actually be generated. + summary_focus_topic = focus_topic or self._derive_auto_focus_topic(messages) + try: + summary = self._generate_summary( + turns_to_summarize, + focus_topic=summary_focus_topic, + memory_context=memory_context, + ) + except AuxiliaryExplicitCancellation: + # Explicit cancellation is a true no-op. Restore state mutated by + # the resume/handoff self-heal scan before the exception escapes to + # the outer transaction, which restores the transcript and lease. + self._previous_summary = _previous_summary_before_scan + self._summary_has_user_turn = _summary_has_user_turn_before_scan + raise # If summary generation failed, behavior splits on # ``abort_on_summary_failure`` (config: compression.abort_on_summary_failure): @@ -6121,7 +6351,7 @@ This compaction should PRIORITISE preserving all information related to the focu # of these cases, rotating into a child session with a placeholder # summary degrades the conversation for zero benefit. Preserve it # unchanged until access is restored or connectivity recovers. - if not summary and ( + if not summary and not feasibility_skip and ( self.abort_on_summary_failure or self._last_summary_auth_failure or self._last_summary_network_failure @@ -6201,15 +6431,26 @@ This compaction should PRIORITISE preserving all information related to the focu # content-free "N messages were removed" marker. if not summary: if not self.quiet_mode: - logger.warning("Summary generation failed — inserting deterministic fallback context summary") + if feasibility_skip: + logger.info("Feasibility skip — inserting deterministic fallback context summary") + else: + logger.warning("Summary generation failed — inserting deterministic fallback context summary") n_dropped = compress_end - compress_start self._last_summary_dropped_count = n_dropped self._last_summary_fallback_used = True telemetry["fallback_used"] = True - telemetry["failure_class"] = telemetry.get("failure_class") or "summary_generation_failed" + if feasibility_skip: + # Deliberate optimization, not a summary failure — keep the + # telemetry class distinct so dashboards don't count skips + # as aux-model breakage. + telemetry["failure_class"] = telemetry.get("failure_class") or "feasibility_skip" + else: + telemetry["failure_class"] = telemetry.get("failure_class") or "summary_generation_failed" summary = self._build_static_fallback_summary( turns_to_summarize, - reason=self._last_summary_error, + # A stale error from an earlier real failure must not be + # embedded into a deliberate feasibility skip's fallback. + reason=None if feasibility_skip else self._last_summary_error, ) tail_messages: List[Dict[str, Any]] = [] @@ -6258,16 +6499,18 @@ This compaction should PRIORITISE preserving all information related to the focu None, ) first_tail_role = None + first_tail_visible_idx: Optional[int] = None if tail_messages: - first_tail_role = next( + first_tail_visible_idx, first_tail_role = next( ( - role - for role in ( - _template_visible_role(m) for m in tail_messages + (idx, role) + for idx, role in ( + (idx, _template_visible_role(m)) + for idx, m in enumerate(tail_messages) ) if role is not None ), - None, + (None, None), ) # When the only protected head message is the system prompt, the # summary becomes the first *visible* message in the API request @@ -6294,11 +6537,27 @@ This compaction should PRIORITISE preserving all information related to the focu # If no user-role message survives in either the protected head or the # preserved tail, the summary MUST carry role="user" so the request # always has at least one user turn. + # + # A bare role check is not enough: the tail's sole surviving user + # turn can be image-only (a screenshot with no caption). The newest + # image-bearing user message is the ``_strip_historical_media`` + # anchor and is kept byte-for-byte, so it never gains a text + # placeholder — its role is "user" but its text content is empty, + # which backends checking for actual query text still reject. Count + # only user messages with non-empty text as "surviving"; when the + # guard fires, the real (never fabricated) summary text lands in a + # role="user" slot, which is always non-empty (falls back to + # ``_build_static_fallback_summary`` above when generation fails). if not _force_user_leading: + def _is_nonempty_user_turn(message: Dict[str, Any]) -> bool: + return message.get("role") == "user" and bool( + _content_text_for_contains(message.get("content")).strip() + ) + _user_survives = any( - message.get("role") == "user" for message in compressed + _is_nonempty_user_turn(message) for message in compressed ) or any( - message.get("role") == "user" for message in tail_messages + _is_nonempty_user_turn(message) for message in tail_messages ) if not _user_survives: _force_user_leading = True @@ -6354,9 +6613,27 @@ This compaction should PRIORITISE preserving all information related to the focu ), }) + # Default merge target: literal tail index 0. For an ordinary + # alternation collision the summary only has to stay *invisible* to + # the template, and a leading template-exempt row (bare tool-call + # assistant message, tool result) is the ideal carrier — it absorbs + # the summary without adding a visible turn, and it leaves the live + # tail user message intact as the model's actual prompt. Retargeting + # to the first template-visible row here would convert that live + # request into the summary carrier for no benefit. + # + # The forced repair path is the exception. There the merge is not + # about alternation but about guaranteeing at least one genuinely + # non-empty role="user" message (an image-only or otherwise + # text-empty surviving user row). An exempt carrier cannot satisfy + # that invariant, so the summary text must land on the + # template-visible row itself. + _merge_target_idx = 0 + if _force_user_leading and first_tail_visible_idx is not None: + _merge_target_idx = first_tail_visible_idx for tail_idx, msg in enumerate(tail_messages): - if _merge_summary_into_tail and tail_idx == 0: - # Merge the summary into the first (post-strip) tail message. + if _merge_summary_into_tail and tail_idx == _merge_target_idx: + # Merge the summary into the tail message that collided. old_content = msg.get("content", "") if _force_user_leading and summary_role == "user": # The summary must be part of the first user-visible diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 8308b618da..3256bca9ec 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -24,10 +24,34 @@ Three concerns live here: ``run_agent`` keeps thin wrappers for each so existing call sites (``self._compress_context(...)``) keep working. Tests that exercise these paths see no behavioural change. + +Thread-safety contract for extension points (#76354 review) +------------------------------------------------------------ + +When the host-level progress-aware timeout is enabled (the default: +``compression.context_timeout_seconds > 0``), the WHOLE compression pass — +including plugin/legacy **context engines** (``compress()`` / +``on_session_start`` / boundary callbacks) and **memory providers** +(``on_pre_compress`` / ``on_session_switch``) — runs on a pooled daemon +thread, not the conversation thread. Extension authors must assume: + +* Calls may arrive on an arbitrary pooled thread; do not rely on + thread-affinity or ``threading.local`` state shared with the caller. +* The input message list is a private deep snapshot owned by the worker; + engines MAY mutate it in place (legacy contract preserved), and that + mutation is invisible to the live conversation unless the pass commits. +* Publication to caller-visible / durable state happens ONLY on an admitted + commit (:class:`CompressionCommitFence`); after a host timeout the still- + running engine's work is discarded. +* Two compression passes never run concurrently for one session (durable + per-session lock), but passes for DIFFERENT sessions may run concurrently + on pool siblings — engine/provider instances shared across sessions must + be thread-safe or internally locked. """ from __future__ import annotations +import concurrent.futures import copy import inspect import json @@ -40,16 +64,30 @@ import uuid import threading from datetime import datetime from pathlib import Path -from typing import Any, Dict, List, Optional, Tuple +from typing import Any, Callable, Dict, List, Optional, Tuple +from agent.auxiliary_client import AuxiliaryExplicitCancellation from agent.context_engine import ( automatic_compaction_status_message, sanitize_memory_context, ) from agent.model_metadata import estimate_request_tokens_rough +from agent.session_activity import ActivityProvenance, normalize_activity_provenance logger = logging.getLogger(__name__) +# Terminal compression outcomes published by host/hygiene timeout or cooldown +# writers. Detached heartbeat workers must not clobber these back to +# agent.compression after cancel (otherwise timeout is unobservable). Observing +# a terminal stamp (or a cancelled commit fence) also latches the heartbeat +# silent so a later UNKNOWN rewrite cannot re-arm a zombie worker. +_TERMINAL_COMPRESSION_PROVENANCES = frozenset( + { + ActivityProvenance.AGENT_COMPRESSION_TIMEOUT, + ActivityProvenance.AGENT_COMPRESSION_COOLDOWN, + } +) + # Stable marker the gateway matches on to re-tag the auto-compaction lifecycle # status as ``kind="compacting"`` (tui_gateway/server.py::_status_update), so # drivers like the desktop app can show an explicit "Summarizing…" indicator @@ -207,6 +245,202 @@ def _cached_prompt_reflects_builtin_memory(agent: Any, cached_prompt: str) -> bo return True +_COMPRESSOR_ATTEMPT_STATE_FIELDS = ( + "_previous_summary", + "_summary_has_user_turn", + "compression_count", + "_last_compression_savings_pct", + "_ineffective_compression_count", + "_anti_thrash_recovery_deadline", + "_fallback_compression_streak", + "_verify_compaction_cleared_threshold", + "_last_compression_made_progress", + "_summary_failure_cooldown_until", + "_cooldown_persist_failed", + "_last_summary_error", + "_consecutive_timeout_failures", + "_last_summary_dropped_count", + "_last_summary_fallback_used", + "_last_compress_aborted", + "_last_summary_auth_failure", + "_last_summary_network_failure", + "_last_aux_model_failure_error", + "_last_aux_model_failure_model", + "_summary_model_fallen_back", + "summary_model", + "_last_compression_telemetry", + "_active_compression_telemetry", + "_compression_telemetry_seed", +) + +_COMPRESSOR_COOLDOWN_STATE_FIELDS = ( + "_summary_failure_cooldown_until", + "_last_summary_error", + "_cooldown_persist_failed", +) + + +def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]: + """Copy only mutable bookkeeping owned by one compression attempt. + + The explicit allow-list avoids copying provider clients, SessionDB handles, + locks, and plugin resources. Missing fields are intentionally ignored so + legacy and third-party compressors keep their existing contract. + """ + try: + values = vars(compressor) + except TypeError: + return {} + selected = { + name: values[name] + for name in _COMPRESSOR_ATTEMPT_STATE_FIELDS + if name in values + } + # Copy the collection as one object so aliases between fields (notably + # _active_compression_telemetry and _last_compression_telemetry) survive. + return copy.deepcopy(selected) + + +def _restore_compressor_attempt_state( + compressor: Any, + snapshot: dict[str, Any], + *, + durable_cooldown_authoritative: Optional[bool] = None, + durable_cooldown_state: Optional[dict[str, Any]] = None, +) -> None: + """Restore the safe per-attempt snapshot after a pre-commit hard cancel.""" + # A successful summary clears the durable cooldown before the outer commit + # boundary. Recreate (or clear) that row before restoring exact in-memory + # values, otherwise the next refresh would overwrite this rollback. Unknown + # durable state and intentionally unpersisted local cooldowns are never + # converted into destructive DB writes during cancellation. + if ( + "_summary_failure_cooldown_until" in snapshot + and durable_cooldown_authoritative is not False + and ( + durable_cooldown_authoritative is True + or not bool(snapshot.get("_cooldown_persist_failed", False)) + ) + ): + session_db = vars(compressor).get("_session_db") + session_id = vars(compressor).get("_session_id") + if session_db is not None and session_id: + if durable_cooldown_authoritative is True: + restorer = getattr( + type(session_db), + "restore_compression_failure_cooldown_row", + None, + ) + if not callable(restorer) or durable_cooldown_state is None: + raise RuntimeError( + "exact compression cooldown rollback API is unavailable" + ) + # This API restores raw columns (including expired and null + # combinations), verifies the read-back, and propagates failure. + restorer( + session_db, + session_id, + copy.deepcopy(durable_cooldown_state), + ) + else: + try: + deadline = float( + snapshot["_summary_failure_cooldown_until"] or 0.0 + ) + remaining = max(0.0, deadline - time.monotonic()) + durable_deadline = time.time() + remaining + durable_error = snapshot.get("_last_summary_error") + if remaining > 0: + recorder = getattr( + type(session_db), + "record_compression_failure_cooldown", + None, + ) + if callable(recorder): + recorder( + session_db, + session_id, + durable_deadline, + durable_error, + ) + else: + clearer = getattr( + type(session_db), + "clear_compression_failure_cooldown", + None, + ) + if callable(clearer): + clearer(session_db, session_id) + except Exception: + # Legacy/third-party compatibility path: its existing APIs + # do not provide a verifiable transaction contract. + logger.debug( + "compression cooldown persistence rollback failed", + exc_info=True, + ) + restored = copy.deepcopy(snapshot) + for name, value in restored.items(): + setattr(compressor, name, value) + + +def _capture_authoritative_cooldown_under_lease( + compressor: Any, + attempt_snapshot: dict[str, Any], +) -> tuple[Optional[bool], Optional[dict[str, Any]]]: + """Refresh and snapshot built-in durable cooldown state under the lease. + + Third-party compressors are deliberately not invoked here: arbitrary plugin + callbacks must not run while the session lease is held. A durable read + failure returns ``False`` so rollback cannot mistake unknown durable state + for an authoritative empty row and clear it; an unavailable legacy API + returns ``None`` and preserves the compatibility path. + """ + try: + from agent.context_compressor import ContextCompressor + + if not isinstance(compressor, ContextCompressor): + return None, None + values = vars(compressor) + session_db = values.get("_session_db") + session_id = values.get("_session_id") + raw_reader = ( + getattr( + type(session_db), "get_compression_failure_cooldown_row", None + ) + if session_db is not None + else None + ) + if session_db is None or not session_id: + # Unbound compressors have no durable row to mutate or restore. + return None, None + if not callable(raw_reader): + return False, None + # Capture the exact persisted representation first. The active getter + # intentionally filters expired rows and therefore cannot serve as a + # lossless rollback snapshot. + durable_state = raw_reader(session_db, session_id) + if not isinstance(durable_state, dict): + raise TypeError("raw compression cooldown snapshot must be a mapping") + ContextCompressor.get_active_compression_failure_cooldown( + compressor, + refresh=True, + ) + except Exception as exc: + logger.debug("authoritative compression cooldown capture failed: %s", exc) + return False, None + authoritative = getattr( + compressor, "_last_cooldown_refresh_was_authoritative", None + ) + if authoritative is not True: + return authoritative, None + + values = vars(compressor) + for name in _COMPRESSOR_COOLDOWN_STATE_FIELDS: + if name in values: + attempt_snapshot[name] = copy.deepcopy(values[name]) + return True, copy.deepcopy(durable_state) + + class CompressionCommitFence: """Fence timeout cancellation against post-summary session mutation. @@ -221,6 +455,31 @@ class CompressionCommitFence: self._lock = threading.Lock() self._cancelled = False self._commit_started = False + # Lock-free commit-phase marker (#76354 review F1). ``begin_commit`` + # RETAINS ``self._lock`` until ``finish_commit``, so any host-side + # observation that needs the lock (``try_cancel_before_commit``) + # blocks/space-outs for the whole commit. This Event is set inside + # ``begin_commit`` while the lock is held but is READABLE WITHOUT the + # lock, so a host can observe "a commit was admitted and may be in + # flight" even while the commit itself is hung — which is exactly when + # the overrun warning must be able to fire. + self._commit_phase = threading.Event() + # Lock-free admission revocation (#76354 review F2). Set by + # :meth:`revoke_commit_admission` on ANY host unwind (KeyboardInterrupt, + # cancellation, unexpected exception) without touching the fence lock, + # so a host that cannot afford to block behind an in-flight commit can + # still guarantee no FUTURE commit is admitted. Plain bool store — + # atomic in CPython. + self._admission_revoked = False + # Holder-qualified durable-lock release hook (#76354 review F4; + # transplanted from PR #71569 by @ciabata-git). The worker publishes an + # idempotent, holder-scoped release callable once it owns the durable + # compression lock; a timed-out host invokes it to free the lease + # without racing a NEW holder (DB release is holder-qualified, so a + # stale release can never delete a replacement's row — no ABA). + self._lock_release_guard = threading.Lock() + self._cancelled_lock_release: Optional[Callable[[], None]] = None + self._cancelled_lock_release_requested = False # Forward-progress telemetry: the compression worker touches this # whenever the streamed summary call produces a token (see # ContextCompressor._call_summary_llm). Waiters use it to distinguish @@ -241,7 +500,7 @@ class CompressionCommitFence: """Seconds since the worker last reported forward progress.""" return max(0.0, time.monotonic() - self._last_progress) - def cancel_before_commit(self) -> bool: + def cancel_before_commit(self, cancel_event: Any = None) -> bool: """Cancel a pending commit, or wait for an active commit to finish. Returns ``True`` when cancellation won before the commit boundary. @@ -250,8 +509,12 @@ class CompressionCommitFence: """ with self._lock: if self._commit_started: + if cancel_event is not None: + cancel_event.set() return False self._cancelled = True + if cancel_event is not None: + cancel_event.set() return True def try_cancel_before_commit(self) -> Optional[bool]: @@ -270,18 +533,558 @@ class CompressionCommitFence: finally: self._lock.release() - def begin_commit(self) -> bool: - """Enter the commit boundary unless cancellation already won.""" + def begin_commit(self, cancel_event: Any = None) -> bool: + """Atomically admit commit unless a hard cancellation already won.""" self._lock.acquire() - if self._cancelled: + if ( + self._cancelled + or self._admission_revoked + or (cancel_event is not None and bool(cancel_event.is_set())) + ): + self._cancelled = True self._lock.release() + if self._admission_revoked: + # Round-2 #1: a revoke that lost the fence-lock race to this + # very begin_commit deferred its lease release; the commit was + # refused, so the release is safe (and idempotent with the + # worker's own holder-qualified cleanup) right now. + self.release_cancelled_compression_lock() return False self._commit_started = True + # Set while the fence lock is held so observers can never see + # commit_in_flight=True for a commit that lost to cancellation. + self._commit_phase.set() return True def finish_commit(self) -> None: """Leave a commit boundary entered by :meth:`begin_commit`.""" + self._commit_phase.clear() self._lock.release() + if self._admission_revoked: + # Round-2 #1: a revoke that arrived while THIS commit was in + # flight deferred its durable-lease release rather than freeing + # the lock out from under an active SessionDB mutation. The + # commit is now fully complete, so perform the deferred release + # here — promptly, without relying on the (possibly parked) + # worker thread's outer cleanup. Idempotent with that cleanup: + # the DB release is holder-qualified. + self.release_cancelled_compression_lock() + + @property + def commit_in_flight(self) -> bool: + """Lock-free read: an admitted commit has begun and not yet finished. + + Safe to call from the host while the worker holds the fence lock for + the whole commit (a hung SessionDB write). Hosts use this to reach + their overrun-warning loop WHILE the commit is blocked instead of + spinning on ``try_cancel_before_commit`` (which needs the lock the + worker retains until ``finish_commit``). + """ + return self._commit_phase.is_set() + + @property + def is_cancelled(self) -> bool: + """True after cancellation won before the commit boundary.""" + return self._cancelled or self._admission_revoked + + def revoke_commit_admission(self) -> None: + """Revoke FUTURE commit admission without blocking on the fence lock. + + #76354 review F2: every host unwind path (KeyboardInterrupt, task + cancellation, unexpected exception while waiting) must guarantee a + detached worker cannot later enter the commit boundary and mutate + durable/session state. The flag store is lock-free: a commit that is + ALREADY in flight cannot be safely abandoned (the invariant "commit + never abandoned mid-mutation" holds), but no NEW commit will be + admitted after this call — ``begin_commit`` re-checks the flag under + the fence lock. + + Round-2 #1 (durable-lease timing): the worker's holder-qualified + lease release (F4) must NOT run while an admitted commit is still + mutating SessionDB — a second compressor could otherwise acquire the + durable lock mid-commit and interleave with the first commit's + writes. The release decision is therefore made under the fence lock: + + - non-blocking acquire succeeds → no commit is in flight (an + admitted commit RETAINS the lock until ``finish_commit``), so the + lease is released immediately, while still holding the lock so a + concurrent ``begin_commit`` cannot slip in between the check and + the release (it would be refused anyway — the flag is already set). + - acquire fails → the lock holder is either an in-flight commit or a + transient boundary (lock-setup / cancel admission). Defer: the + release then runs in ``finish_commit`` (after the mutation fully + completes) or on the ``begin_commit``-refusal path, whichever the + worker reaches first. Both are idempotent with the worker's own + outer cleanup because the DB release is holder-qualified. + """ + self._admission_revoked = True + if self._lock.acquire(blocking=False): + try: + self.release_cancelled_compression_lock() + finally: + self._lock.release() + # else: deferred — finish_commit()/begin_commit() re-check + # _admission_revoked and perform the release once no commit can be + # mid-mutation. + + # ── Holder-qualified durable-lease cancellation (#76354 F4) ────────── + # Transplanted from PR #71569 (@ciabata-git): the worker publishes an + # idempotent, holder-scoped release hook once it owns the durable + # compression lock, and the host invokes it after winning cancellation. + # ABA safety comes from SessionDB.release_compression_lock being + # holder-qualified (DELETE ... WHERE holder = ?), so a stale release can + # never free a NEW holder's lease. + + def begin_lock_setup(self) -> bool: + """Fence durable-lock acquisition and release-hook publication. + + The caller keeps the fence until it has either published the exact + holder-qualified release hook or established that no lock was + acquired. A timeout cannot therefore win in the gap between acquiring + the durable lock and making its cancellation cleanup callable. + """ + self._lock.acquire() + if self._cancelled or self._admission_revoked: + self._lock.release() + return False + return True + + def finish_lock_setup(self) -> None: + """Leave a lock setup boundary entered by :meth:`begin_lock_setup`.""" + self._lock.release() + + def register_cancelled_lock_release( + self, release: Callable[[], None] + ) -> bool: + """Publish the timed-out worker's holder-qualified lock release. + + Returns whether cancellation cleanup was requested before publication. + In that race, the release runs synchronously before this method returns. + """ + with self._lock_release_guard: + self._cancelled_lock_release = release + requested = self._cancelled_lock_release_requested + if requested: + release() + return requested + + def clear_cancelled_lock_release(self, release: Callable[[], None]) -> None: + """Forget ``release`` after the worker's normal cleanup finishes.""" + with self._lock_release_guard: + if self._cancelled_lock_release is release: + self._cancelled_lock_release = None + + def release_cancelled_compression_lock(self) -> None: + """Release the cancelled worker's lock without finalizing its clients. + + Callers invoke this only after cancellation won (fence cancelled or + admission revoked). A request that races ahead of lock-hook + publication is retained and fulfilled synchronously when the worker + publishes the hook. + """ + with self._lock_release_guard: + self._cancelled_lock_release_requested = True + release = self._cancelled_lock_release + if release is not None: + release() + + +# Defaults for the in-agent (non-hygiene) progress-aware compress_context wrap. +# Mirror hermes_cli.config.DEFAULT_CONFIG["compression"] keys of the same name. +DEFAULT_CONTEXT_TIMEOUT_SECONDS = 120.0 +DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0 + +# Shared daemon pool for sync compress_context timeout wraps — analogous to +# asyncio's default executor used by gateway session hygiene's +# ``loop.run_in_executor(None, ...)``, but daemon so a fence-cancelled hung +# worker cannot block interpreter exit via concurrent.futures' atexit join. +# Created lazily; never shut down per call (a timed-out worker may still be +# winding down after fence cancel). +_compress_timeout_executor = None +_compress_timeout_executor_lock = threading.Lock() + +# Commit-phase overrun wait slice: once an in-flight SessionDB commit runs +# past the total ceiling, keep waiting in bounded increments of this size so +# every overrun window produces a fresh (escalating) log line instead of one +# silent unbounded future.result(). Clamped down to the ceiling for tiny test +# ceilings so overrun reporting stays observable at test timescales. +_COMMIT_OVERRUN_WAIT_SLICE_SECONDS = 30.0 + +# Bounded admission for the shared compress-timeout pool (#76354 review F6). +# The stdlib executor queue is unbounded: with all four workers wedged in hung +# summaries, a fifth compression would queue silently, wait out its whole +# timeout without ever starting, and remain eligible to run as a stale job +# whenever a worker recovered. Admission is therefore capped at the worker +# count — when every worker slot is occupied (running OR admitted-not-started) +# submission FAILS FAST and the caller continues without compression. +# +# Recovery contract when all workers are wedged: new compressions fail fast +# (no queue growth, conversation continues uncompressed, a warning is logged +# each attempt); wedged workers are fence-cancelled so they cannot publish +# anything when they eventually return, and each recovery frees its admission +# slot via the future done-callback, restoring normal service. If a worker +# NEVER returns, its slot is lost for the process lifetime — bounded, +# observable degradation instead of an unbounded stale-job queue. +_COMPRESS_EXECUTOR_MAX_WORKERS = 4 +_compress_admission_lock = threading.Lock() +_compress_admitted_count = 0 + + +class CompressionExecutorSaturatedError(RuntimeError): + """All compression pool slots are occupied; submission was refused.""" + + +def _try_admit_compression_job() -> bool: + """Reserve one bounded compression-pool admission slot (F6).""" + global _compress_admitted_count + with _compress_admission_lock: + if _compress_admitted_count >= _COMPRESS_EXECUTOR_MAX_WORKERS: + return False + _compress_admitted_count += 1 + return True + + +def _release_compression_admission(_future=None) -> None: + """Free an admission slot (future done-callback or failed submit).""" + global _compress_admitted_count + with _compress_admission_lock: + if _compress_admitted_count > 0: + _compress_admitted_count -= 1 + + +def _get_compress_timeout_executor(): + """Return the process-wide compress-timeout DaemonThreadPoolExecutor.""" + global _compress_timeout_executor + executor = _compress_timeout_executor + if executor is not None: + return executor + from tools.daemon_pool import DaemonThreadPoolExecutor + + with _compress_timeout_executor_lock: + if _compress_timeout_executor is None: + # Small pool: compress is rare and heavy. Sized for a few + # overlapping calls (live compress + fence-cancelled workers + # still winding down), not asyncio's min(32, cpu+4) fan-out. + _compress_timeout_executor = DaemonThreadPoolExecutor( + max_workers=_COMPRESS_EXECUTOR_MAX_WORKERS, + thread_name_prefix="compress-ctx-timeout", + ) + return _compress_timeout_executor + + +def resolve_context_compression_timeouts( + compression_cfg: Optional[dict] = None, +) -> Tuple[float, float]: + """Return ``(idle_timeout_seconds, total_ceiling_seconds)``. + + ``idle_timeout_seconds <= 0`` disables the owned progress-aware wrapper. + The ceiling is clamped to at least one idle window when the idle budget + is positive, matching gateway hygiene semantics. + """ + idle = DEFAULT_CONTEXT_TIMEOUT_SECONDS + ceiling = DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS + cfg = compression_cfg + if cfg is None: + try: + from hermes_cli.config import load_config + + raw = load_config() + maybe = raw.get("compression", {}) if isinstance(raw, dict) else {} + cfg = maybe if isinstance(maybe, dict) else {} + except Exception: + cfg = {} + if isinstance(cfg, dict): + raw_idle = cfg.get("context_timeout_seconds") + if raw_idle is not None: + try: + parsed = float(raw_idle) + # Explicit 0/negative disables; positive values win. + idle = parsed + except (TypeError, ValueError): + pass + raw_ceiling = cfg.get("context_total_ceiling_seconds") + if raw_ceiling is not None: + try: + parsed = float(raw_ceiling) + if parsed > 0: + ceiling = parsed + except (TypeError, ValueError): + pass + if idle > 0: + ceiling = max(ceiling, idle) + return idle, ceiling + + +def run_compress_context_with_progress_timeout( + *, + worker: Callable[[CompressionCommitFence], Tuple[list, str]], + messages: list, + system_prompt_fallback: Any, + idle_timeout_seconds: float, + total_ceiling_seconds: float, + on_timeout: Optional[Callable[[float, float, float], None]] = None, + on_commit_overrun: Optional[Callable[[float, float], None]] = None, + fence: Optional[CompressionCommitFence] = None, + telemetry_agent: Any = None, +) -> Tuple[list, str]: + """Run ``worker(fence)`` under a sync progress-aware timeout. + + The idle budget is inactivity-based (same idea as gateway session hygiene): + streamed summary progress via :meth:`CompressionCommitFence.touch_progress` + extends the wait. A hard ceiling still bounds a degenerate trickle stream. + + When cancellation wins before the commit boundary, returns + ``(messages, system_prompt_fallback)`` immediately and leaves the worker + thread detached — the fence prevents a late commit from mutating session + state. When the worker already entered the commit boundary, waits for that + commit to finish and returns its result. + + Timeout budgets (``idle_timeout_seconds`` / ``total_ceiling_seconds``) cover + the **pre-commit** wait only — the summary / stream phase before + :meth:`CompressionCommitFence.begin_commit`. Once the worker holds the + commit fence, SessionDB mutation is already in flight and cannot be safely + abandoned without risking transcript divergence; the commit is therefore + always allowed to complete. The commit-phase wait is still *bounded in + increments* against the remaining total ceiling: if the commit runs past + ``total_ceiling_seconds``, the overrun is logged loudly (escalating from + WARNING to ERROR on repeat) and surfaced once via ``on_commit_overrun``, + while the host keeps waiting in bounded slices until the commit finishes. + The documented guarantee is: **summary phase bounded by the ceiling; + commit phase logged + surfaced if it exceeds it** (never silently hung, + never abandoned mid-commit). + + ``system_prompt_fallback`` may be a string or a zero-arg callable resolved + only on the timeout path, so successful compression never pays for (or + fails on) an eager prompt rebuild. + """ + if idle_timeout_seconds <= 0: + raise ValueError( + "run_compress_context_with_progress_timeout requires " + "idle_timeout_seconds > 0; call compress_context directly to disable" + ) + + def _resolve_fallback_prompt() -> str: + if callable(system_prompt_fallback): + return system_prompt_fallback() + return system_prompt_fallback + + fence = fence if fence is not None else CompressionCommitFence() + ceiling = max(float(total_ceiling_seconds), float(idle_timeout_seconds)) + idle = float(idle_timeout_seconds) + # Sync mirror of gateway session-hygiene's run_in_executor(None, ...) + + # wait_for loop (gateway/run.py): offload compress_context onto the shared + # daemon pool, poll with an inactivity budget + total ceiling, then + # fence-cancel on timeout so a late commit cannot land. Daemon workers + # match tool_executor: a cancelled hung summary must not block process exit. + from tools.thread_context import propagate_context_to_thread + + executor = _get_compress_timeout_executor() + # Bounded admission (#76354 F6): refuse rather than queue when every pool + # slot is occupied. A queued job would silently wait out its whole budget + # without starting and stay eligible to run as a stale cancelled job when + # a worker recovers. Fail fast: continue without compression this cycle. + if not _try_admit_compression_job(): + logger.warning( + "Context compression pool saturated (%d workers busy) — " + "refusing new compression this cycle and continuing without " + "compression. Wedged workers are fence-cancelled and free their " + "slot when they return; if this persists, check the summary " + "provider health.", + _COMPRESS_EXECUTOR_MAX_WORKERS, + ) + # Round-2 #6: saturation refusals must be visible in the same + # telemetry stream as every other failed attempt, or a wedged pool + # looks like compression simply stopped being attempted. + if telemetry_agent is not None: + _emit_compression_attempt_telemetry( + telemetry_agent, + started_at=time.monotonic(), + commit_status="aborted", + split_status="aborted", + failure_class="pool_saturated", + ) + return messages, _resolve_fallback_prompt() + + def _fence_gated_worker(worker_fence: CompressionCommitFence): + # F6: an admitted job can still start after the host stopped waiting + # (worker slot freed late). Check the fence BEFORE any expensive + # summary work so a stale job never burns an LLM call; its return + # value is discarded by the already-departed host. + if worker_fence.is_cancelled: + logger.info( + "Skipping stale compression job: fence cancelled before start" + ) + return messages, "" + return worker(worker_fence) + + # Bare pool workers start with an empty ContextVar map; propagate the + # parent conversation/approval context into the worker. + try: + future = executor.submit( + propagate_context_to_thread(_fence_gated_worker), fence + ) + except BaseException: + _release_compression_admission() + raise + future.add_done_callback(_release_compression_admission) + wait_started = time.monotonic() + # F2: EVERY host unwind (KeyboardInterrupt, task cancellation, unexpected + # exception while waiting) must revoke future commit admission before the + # host resumes, or a detached worker could later commit and mutate durable + # state behind the caller's back. ``handled_exit`` marks the paths that + # settle admission themselves (worker result returned, or fence cancel + # won); everything else revokes in the ``finally``. + handled_exit = False + try: + while True: + waited = time.monotonic() - wait_started + remaining_ceiling = ceiling - waited + if remaining_ceiling <= 0: + break + # #76354 S3 analogue for this wait: charge the idle budget from + # the LAST PROGRESS event, not from the start of this wait slice. + # Waiting a full ``idle`` after progress that landed early in the + # previous slice would allow silence to approach 2x the budget. + since_progress = fence.seconds_since_progress() + wait_slice = min( + max(idle - since_progress, 0.005), remaining_ceiling + ) + try: + result = future.result(timeout=wait_slice) + handled_exit = True + return result + except concurrent.futures.TimeoutError: + waited = time.monotonic() - wait_started + since_progress = fence.seconds_since_progress() + if since_progress < idle and waited < ceiling: + logger.info( + "Context compression still streaming after %.0fs " + "(last progress %.1fs ago) — extending wait " + "(ceiling %.0fs)", + waited, + since_progress, + ceiling, + ) + continue + break + + # F6: a not-yet-started future must not linger as a stale queued job. + # cancel() is a no-op for a running worker (fence handles that path). + future.cancel() + + cancelled: Optional[bool] = None + while cancelled is None: + # F1: ``begin_commit`` retains the fence lock until + # ``finish_commit``, so a hung commit makes + # ``try_cancel_before_commit`` return None forever. The lock-free + # phase marker breaks the spin so the overrun-warning loop below + # is reachable WHILE the commit is still blocked. + if fence.commit_in_flight: + cancelled = False + break + cancelled = fence.try_cancel_before_commit() + if cancelled is None: + # Round-2 #5: the fence is only held transiently here (lock + # setup / cancel admission — an in-flight commit is caught by + # the commit_in_flight check above), but that window rides + # SessionDB write patience and can last seconds. 25ms keeps + # sub-tick latency without a 1kHz spin. + time.sleep(0.025) + if not cancelled: + # Pre-commit ceiling already elapsed, but begin_commit() won the + # race. Waiting is intentional: SessionDB mutation cannot be + # fence-cancelled. The wait is bounded in increments against the + # remaining ceiling: a commit that overruns total_ceiling_seconds + # is logged loudly and surfaced once (on_commit_overrun), then + # waited on in bounded slices with escalating log level until it + # completes. Guarantee: summary phase bounded by ceiling; commit + # phase logged + surfaced if it exceeds it — never silently hung, + # never abandoned mid-commit. F1: this loop is reachable WHILE + # the commit is blocked (commit_in_flight is lock-free), so the + # warning + on_commit_overrun fire during the hang, not after it. + overrun_surfaced = False + overrun_reports = 0 + while True: + waited = time.monotonic() - wait_started + remaining = ceiling - waited + if remaining <= 0: + # Ceiling breached while the commit is in flight. Wait in + # bounded increments so each overrun window is visible in + # logs rather than one silent unbounded block. + remaining = min( + _COMMIT_OVERRUN_WAIT_SLICE_SECONDS, + max(ceiling, 0.05), + ) + overrun_reports += 1 + log = ( + logger.warning if overrun_reports <= 2 else logger.error + ) + log( + "Context compression SessionDB commit still running " + "%.1fs past the total ceiling (waited %.1fs, ceiling " + "%.1fs); commit cannot be abandoned mid-flight — " + "continuing to wait (check SessionDB health if this " + "persists)", + waited - ceiling, + waited, + ceiling, + ) + if not overrun_surfaced and on_commit_overrun is not None: + overrun_surfaced = True + try: + on_commit_overrun(waited, ceiling) + except Exception: + logger.debug( + "compress_context commit-overrun callback " + "failed", + exc_info=True, + ) + try: + result = future.result(timeout=remaining) + handled_exit = True + return result + except concurrent.futures.TimeoutError: + # Fence progress (commit-phase touch_progress) is + # informative only — the commit must complete regardless; + # loop and re-report with the updated overrun window. + continue + + # Idle-timeout path: cancellation won before the commit boundary. + # The fence already blocks any future commit; F4 additionally frees + # the timed-out worker's durable lease via the holder-qualified hook + # so a NEW compressor can acquire the lock immediately (no ABA: the + # DB release is holder-scoped). + handled_exit = True + fence.release_cancelled_compression_lock() + waited = time.monotonic() - wait_started + since_progress = fence.seconds_since_progress() + if on_timeout is not None: + try: + on_timeout(idle, waited, since_progress) + except Exception: + logger.debug( + "compress_context timeout callback failed", + exc_info=True, + ) + else: + logger.warning( + "Context compression made no progress for %.1fs " + "(total wait %.1fs, ceiling %.1fs); continuing without " + "compression", + since_progress, + waited, + ceiling, + ) + # Leave the future on the shared pool: fence cancel won, so a late + # commit cannot land (same detachment model as gateway hygiene). + return messages, _resolve_fallback_prompt() + finally: + if not handled_exit: + # F2: KeyboardInterrupt / cancellation / any unexpected exception + # while waiting — revoke commit admission (and release the + # worker's durable lease via the holder-qualified hook) before + # the host unwinds, so the detached worker can never publish. + fence.revoke_commit_admission() def _lock_api_is_absent_on_session_db(lock_db: Any) -> bool: @@ -307,13 +1110,21 @@ def _lock_api_is_absent_on_session_db(lock_db: Any) -> bool: return False -def _refresh_persisted_compression_guards(compressor: Any) -> None: +def _refresh_persisted_compression_guards( + compressor: Any, + *, + include_cooldown: bool = True, +) -> None: """Refresh durable automatic-compression guards on a built-in compressor.""" - method_calls = ( - ("get_active_compression_failure_cooldown", {"refresh": True}), + method_calls = [ ("_load_fallback_compression_streak", {}), ("_load_ineffective_compression_count", {}), - ) + ] + if include_cooldown: + method_calls.insert( + 0, + ("get_active_compression_failure_cooldown", {"refresh": True}), + ) for method_name, kwargs in method_calls: method = getattr(type(compressor), method_name, None) if not callable(method): @@ -573,8 +1384,17 @@ def _supported_compression_kwargs( class _CompressionActivityHeartbeat: """Refresh the agent inactivity tracker while compression blocks in an aux call.""" - def __init__(self, agent: Any, interval_seconds: float | None = None) -> None: + def __init__( + self, + agent: Any, + interval_seconds: float | None = None, + commit_fence: Optional[CompressionCommitFence] = None, + ) -> None: self._agent = agent + self._commit_fence = commit_fence + # Latched once host cancel/timeout wins or a terminal stamp is observed, + # so a later UNKNOWN rewrite cannot re-arm a detached zombie heartbeat. + self._suppressed = False if interval_seconds is None: interval_seconds = getattr(agent, "_compression_activity_heartbeat_interval", 60.0) try: @@ -592,7 +1412,10 @@ class _CompressionActivityHeartbeat: ) def start(self) -> "_CompressionActivityHeartbeat": - self._touch("context compression started") + # A new compression episode always republishes agent.compression even + # if a prior timeout/cooldown stamp is still on the agent. + self._suppressed = False + self._touch("context compression started", allow_terminal_overwrite=True) self._thread.start() return self @@ -600,18 +1423,63 @@ class _CompressionActivityHeartbeat: self._stop.set() if self._thread.is_alive() and threading.current_thread() is not self._thread: self._thread.join(timeout=1.0) - self._touch(desc) + # Host timeout already owns the terminal stamp; a detached worker's + # late stop must not republish agent.compression / "completed". + if self._should_suppress(): + return + # Terminal completed/failed must reach SessionDB even inside the + # ordinary 60s activity persist window — otherwise durable labels + # stay on "context compression in progress" after /compress (which + # never hits run_conversation's turn-end clear). + self._touch(desc, force_persist=True) - def _touch(self, desc: str) -> None: + def _fence_cancelled(self) -> bool: + fence = self._commit_fence + return fence is not None and fence.is_cancelled + + def _should_suppress(self) -> bool: + if self._suppressed: + return True + if self._fence_cancelled(): + self._suppressed = True + return True + return False + + def _touch( + self, + desc: str, + *, + allow_terminal_overwrite: bool = False, + force_persist: bool = False, + ) -> None: try: + if not allow_terminal_overwrite: + if self._should_suppress(): + return + current = normalize_activity_provenance( + getattr(self._agent, "_last_activity_provenance", None) + ) + if current in _TERMINAL_COMPRESSION_PROVENANCES: + self._suppressed = True + return touch = getattr(self._agent, "_touch_activity", None) if callable(touch): - touch(desc) + # Re-check after reading provenance: host may cancel/stamp + # TIMEOUT between the earlier guard and the write. + if not allow_terminal_overwrite and self._should_suppress(): + return + touch( + desc, + provenance=ActivityProvenance.AGENT_COMPRESSION, + force_persist=force_persist, + ) except Exception: logger.debug("compression activity heartbeat touch failed", exc_info=True) def _run(self) -> None: while not self._stop.wait(self._interval_seconds): + if self._should_suppress(): + return self._touch("context compression in progress") @@ -1298,6 +2166,11 @@ def compress_context( prompt — the session is NOT rotated. Callers should detect the no-op via ``len(returned) == len(input)`` and stop the retry loop. """ + _compressor_attempt_snapshot = _snapshot_compressor_attempt_state( + agent.context_compressor + ) + _durable_cooldown_authoritative: Optional[bool] = None + _durable_cooldown_state: Optional[dict[str, Any]] = None if ( defer_context_engine_notification and callable(getattr(agent, _PENDING_CONTEXT_ENGINE_NOTIFICATION, None)) @@ -1343,8 +2216,13 @@ def compress_context( if getattr(agent, "api_mode", None) == "codex_app_server": _codex_fence_entered = False if commit_fence is not None: - _codex_fence_entered = commit_fence.begin_commit() + _codex_fence_entered = commit_fence.begin_commit( + getattr(agent, "_hard_interrupt_requested", None) + ) if not _codex_fence_entered: + _restore_compressor_attempt_state( + agent.context_compressor, _compressor_attempt_snapshot + ) existing_prompt = getattr(agent, "_cached_system_prompt", None) if not existing_prompt: existing_prompt = agent._build_system_prompt(system_message) @@ -1505,6 +2383,19 @@ def compress_context( _lock_ttl = 300.0 _lock_refresh_interval = getattr(agent, "_compression_lock_refresh_interval", None) _lock_refresher: Optional[_CompressionLockLeaseRefresher] = None + # F4 (#76354, transplanted from PR #71569 by @ciabata-git): fence the + # durable-lock acquisition + release-hook publication so a host timeout + # can never win in the gap between acquiring the durable lock and having + # a holder-qualified way to release it. + _lock_setup_entered = False + + def _finish_lock_setup() -> None: + nonlocal _lock_setup_entered + if not _lock_setup_entered or commit_fence is None: + return + _lock_setup_entered = False + commit_fence.finish_lock_setup() + if _lock_db is not None and _lock_sid: _lock_holder = _compression_lock_holder(agent) if _lock_lookup_error is not None: @@ -1533,6 +2424,27 @@ def compress_context( ) _lock_acquired = True # acquired-but-unlocked compatibility path else: + if commit_fence is not None: + _lock_setup_entered = commit_fence.begin_lock_setup() + if not _lock_setup_entered: + logger.info( + "Compression commit cancelled before lock acquisition " + "(session=%s).", + agent.session_id or "none", + ) + agent._last_compaction_in_place = False + _existing_sp = getattr(agent, "_cached_system_prompt", None) + if not _existing_sp: + _existing_sp = agent._build_system_prompt(system_message) + _emit_compression_attempt_telemetry( + agent, + started_at=_attempt_started_at, + commit_status="aborted", + split_status="aborted", + failure_class="commit_fence_cancelled", + ) + _complete_compaction_lifecycle() + return messages, _existing_sp try: _lock_acquired = _try_acquire_lock( _lock_sid, _lock_holder, ttl_seconds=_lock_ttl @@ -1560,6 +2472,7 @@ def compress_context( ) _lock_acquired = False if not _lock_acquired: + _finish_lock_setup() try: existing = _lock_db.get_compression_lock_holder(_lock_sid) except Exception: @@ -1604,29 +2517,83 @@ def compress_context( _complete_compaction_lifecycle() return messages, _existing_sp _lock_released = False + _lock_release_guard = threading.Lock() + + def _release_lock_holder_only() -> None: + """Stop this holder's refresher and release only its durable lock. + + Holder-qualified and idempotent (#76354 F4, from PR #71569): safe for + the HOST to invoke after a timeout without an ABA race — the DB + release is scoped to this worker's holder token, so a NEW holder's + lease can never be deleted by this stale release. + """ + nonlocal _lock_released + with _lock_release_guard: + if _lock_released: + return + _lock_released = True + if getattr(agent, "_active_compression_lock_holder", None) == _lock_holder: + agent._active_compression_lock_holder = None + if _lock_refresher is not None: + try: + _lock_refresher.stop() + except Exception as _stop_err: + logger.debug("compression lock refresher stop failed: %s", _stop_err) + if _lock_db is not None and _lock_sid and _lock_holder: + try: + _lock_db.release_compression_lock(_lock_sid, _lock_holder) + except Exception as _rel_err: + logger.debug("compression lock release failed: %s", _rel_err) def _release_lock() -> None: - """Release the lock keyed on the OLD session_id (before rotation).""" - nonlocal _lock_released - _complete_compaction_lifecycle() - if _lock_released: - return - _lock_released = True - if getattr(agent, "_active_compression_lock_holder", None) == _lock_holder: - agent._active_compression_lock_holder = None - if _lock_refresher is not None: + """Finish lifecycle cleanup and release the OLD session lock once.""" + try: + _complete_compaction_lifecycle() + finally: try: - _lock_refresher.stop() - except Exception as _stop_err: - logger.debug("compression lock refresher stop failed: %s", _stop_err) - if _lock_db is not None and _lock_sid and _lock_holder: - try: - _lock_db.release_compression_lock(_lock_sid, _lock_holder) - except Exception as _rel_err: - logger.debug("compression lock release failed: %s", _rel_err) + _release_lock_holder_only() + finally: + try: + if commit_fence is not None: + commit_fence.clear_cancelled_lock_release( + _release_lock_holder_only + ) + finally: + _finish_lock_setup() if _lock_holder is not None: agent._active_compression_lock_holder = _lock_holder + if ( + commit_fence is not None + and commit_fence.register_cancelled_lock_release( + _release_lock_holder_only + ) + ): + # Cancellation already won while we were inside lock setup: the + # hook just ran synchronously, our lease is gone — abort before + # any summary work. + logger.info( + "Compression commit cancelled before summary dispatch " + "(session=%s).", + agent.session_id or "none", + ) + agent._last_compaction_in_place = False + _existing_sp = getattr(agent, "_cached_system_prompt", None) + if not _existing_sp: + _existing_sp = agent._build_system_prompt(system_message) + _emit_compression_attempt_telemetry( + agent, + started_at=_attempt_started_at, + commit_status="aborted", + split_status="aborted", + failure_class="commit_fence_cancelled", + ) + _release_lock() + return messages, _existing_sp + + # Publish the holder-qualified release hook before a timeout can win the + # fence. If no durable lock was acquired there is no hook to publish. + _finish_lock_setup() # A delayed contender can acquire the parent lock after the winning path # has released it and completed rotation. The lock serializes work but does @@ -1671,13 +2638,36 @@ def compress_context( ) return messages, _existing_sp + # Snapshot the authoritative durable cooldown only after this attempt owns + # the session lease. This runs for force=True too, but does not apply the + # automatic breaker gate: manual compression still retries immediately. + _durable_cooldown_authoritative, _durable_cooldown_state = ( + _capture_authoritative_cooldown_under_lease( + agent.context_compressor, + _compressor_attempt_snapshot, + ) + ) + if _durable_cooldown_authoritative is False: + # A bound built-in compressor reached its durable getter and the read + # failed. Proceeding with force=True could clear an unknown newer row + # before cancellation has enough information to restore it. This is a + # persistence-safety abort, not automatic breaker gating. + _release_lock() + existing_prompt = getattr(agent, "_cached_system_prompt", None) + if not existing_prompt: + existing_prompt = agent._build_system_prompt(system_message) + return messages, existing_prompt + # The agent may have been constructed before another path completed an # in-place compaction on the same session. Re-read durable breaker state # after acquiring the session lock so this final gate cannot act on the # stale snapshot loaded by bind_session_state(). if not force: compressor = agent.context_compressor - _refresh_persisted_compression_guards(compressor) + _refresh_persisted_compression_guards( + compressor, + include_cooldown=False, + ) blocked = getattr( type(compressor), "_automatic_compression_blocked", @@ -1691,16 +2681,24 @@ def compress_context( return messages, existing_prompt _activity_heartbeat: Optional[_CompressionActivityHeartbeat] = None + messages_before_compression = None try: if _lock_holder is not None: - _lock_refresher = _CompressionLockLeaseRefresher( + _candidate_refresher = _CompressionLockLeaseRefresher( _lock_db, _lock_sid, _lock_holder, _lock_ttl, _lock_refresh_interval, ) - _lock_refresher.start() + # Cancellation may release the holder after hook publication but + # before this refresher starts. Serialize that check/start with + # the idempotent release path so a refresher is never started for + # an already-released lock (#76354 F4 / PR #71569). + with _lock_release_guard: + if not _lock_released: + _lock_refresher = _candidate_refresher + _lock_refresher.start() # The caller's history snapshot predates lease acquisition. Reload the # durable parent after the lease is live; MORE durable rows than the @@ -1782,7 +2780,9 @@ def compress_context( ) messages_before_compression = copy.deepcopy(messages) - _activity_heartbeat = _CompressionActivityHeartbeat(agent).start() + _activity_heartbeat = _CompressionActivityHeartbeat( + agent, commit_fence=commit_fence + ).start() # Publish forward progress to the commit fence while the summary LLM # call streams. Async hosts (gateway session hygiene) poll # ``commit_fence.seconds_since_progress()`` to extend their deadline @@ -1791,21 +2791,118 @@ def compress_context( # thread-local and the compress call is synchronous on this thread, # so it cannot leak into unrelated auxiliary calls. # - # Fenceless callers (CLI /compress, in-loop auto-compress) install a - # no-op hook: nobody polls their progress, but an ACTIVE hook is what - # switches the summary call onto the streamed path — giving every - # compression path the same two guarantees: the configured timeout - # acts on inactivity (slow models finish), and a byte-trickling - # provider that keeps the connection alive forever is cut off at the + # Callers that pass no commit_fence install a no-op progress hook + # here. AIAgent._compress_context injects an owned fence for + # fenceless callers so the host-level progress-aware wait can + # extend on streamed tokens; gateway hygiene already passes its + # own fence. An ACTIVE hook (even a no-op) is what switches the + # summary call onto the streamed path — giving every compression + # path the same two guarantees: the configured timeout acts on + # inactivity (slow models finish), and a byte-trickling provider + # that keeps the connection alive forever is cut off at the # streamed total ceiling (see _aux_stream_total_ceiling) instead of # outliving the SDK's inactivity timeout indefinitely. - from agent.auxiliary_client import aux_progress_hook + from agent.auxiliary_client import ( + aux_interrupt_protection, + aux_progress_hook, + ) _progress_hook = ( commit_fence.touch_progress if commit_fence is not None else (lambda: None) ) - with aux_progress_hook(_progress_hook): - compressed = compress_fn(messages, **compress_kwargs) + # F4 state-ordering (#76354): a LATE successful summary must not undo + # the timeout cooldown the host recorded. Install a cancellation + # check the compressor consults BEFORE clearing the failure cooldown; + # removed in the finally below so it cannot leak into later attempts + # (e.g. a manual /compress force-clear). + if commit_fence is not None: + try: + agent.context_compressor._compression_cancelled_check = ( + lambda: commit_fence.is_cancelled + ) + except Exception: + pass + # Incoming-message interrupts and active-turn redirects must not tear an + # atomic summary in half (#23975). Explicit stop surfaces set a separate + # Event atomically; never infer cause from the racy message fields. + _hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None) + try: + # F6: never start expensive summary work for an already-cancelled + # fence (a stale queued job admitted after host departure). + if commit_fence is not None and commit_fence.is_cancelled: + logger.info( + "Compression cancelled before summary dispatch " + "(session=%s) — skipping summary work.", + agent.session_id or "none", + ) + compressed = messages + else: + with aux_progress_hook(_progress_hook), aux_interrupt_protection( + cancel_event=_hard_cancel_event + ): + compressed = compress_fn(messages, **compress_kwargs) + # Freeze a hard stop that arrived after the final provider + # attempt unwound but before this transaction can rotate + # session state. + if ( + _hard_cancel_event is not None + and _hard_cancel_event.is_set() + ): + raise AuxiliaryExplicitCancellation() + finally: + if commit_fence is not None: + try: + agent.context_compressor._compression_cancelled_check = None + except Exception: + pass + except AuxiliaryExplicitCancellation: + try: + _restore_compressor_attempt_state( + agent.context_compressor, + _compressor_attempt_snapshot, + durable_cooldown_authoritative=_durable_cooldown_authoritative, + durable_cooldown_state=_durable_cooldown_state, + ) + except BaseException as _rollback_exc: + # Compensation failure must surface, but it must not strand the + # session lease or retain an in-memory transcript mutation. + if ( + messages_before_compression is not None + and messages != messages_before_compression + ): + messages[:] = copy.deepcopy(messages_before_compression) + if _activity_heartbeat is not None: + _activity_heartbeat.stop("context compression rollback failed") + _activity_heartbeat = None + _release_lock() + _emit_compression_attempt_telemetry( + agent, + started_at=_attempt_started_at, + commit_status="aborted", + split_status="aborted", + failure_class=f"rollback:{type(_rollback_exc).__name__}", + ) + raise + if ( + messages_before_compression is not None + and messages != messages_before_compression + ): + messages[:] = copy.deepcopy(messages_before_compression) + if _activity_heartbeat is not None: + _activity_heartbeat.stop("context compression cancelled") + _activity_heartbeat = None + _release_lock() + _emit_compression_attempt_telemetry( + agent, + started_at=_attempt_started_at, + commit_status="aborted", + split_status="aborted", + failure_class="explicit_interrupt", + ) + _existing_sp = getattr(agent, "_cached_system_prompt", None) + if not _existing_sp: + _existing_sp = agent._build_system_prompt(system_message) + return messages, _existing_sp except BaseException as _compress_exc: # ANY exception after lock acquisition — memory hook, capability # inspection, engine lookup, or compress() — must release the lock so @@ -1838,6 +2935,9 @@ def compress_context( _compression_used_fallback = bool( getattr(agent.context_compressor, "_last_summary_fallback_used", False) ) + _compression_feasibility_skip = bool( + getattr(agent.context_compressor, "_last_feasibility_skip", False) + ) # If compression aborted (aux LLM failed to produce a usable summary) # the compressor returns the input messages unchanged. Surface the @@ -1915,8 +3015,19 @@ def compress_context( return messages, _existing_sp if commit_fence is not None: - _commit_fence_entered = commit_fence.begin_commit() + _commit_fence_entered = commit_fence.begin_commit(_hard_cancel_event) if not _commit_fence_entered: + _restore_compressor_attempt_state( + agent.context_compressor, + _compressor_attempt_snapshot, + durable_cooldown_authoritative=_durable_cooldown_authoritative, + durable_cooldown_state=_durable_cooldown_state, + ) + if ( + messages_before_compression is not None + and messages != messages_before_compression + ): + messages[:] = copy.deepcopy(messages_before_compression) logger.info( "Compression commit cancelled before session mutation " "(session=%s).", @@ -2247,6 +3358,31 @@ def compress_context( ) _boundary_parent = _old_sid or agent.session_id or "" + # Round-2 #4: the activity heartbeat's terminal "context compression + # completed" stamp landed on the PARENT row (force-persisted before + # the rotation re-pointed agent.session_id at the child). Without a + # cleanup, the archived parent advertises a fresh last_activity_at + + # "context compression completed" forever — a permanent false-fresh + # row for any activity consumer that scans ended sessions. Clear the + # labels on the parent best-effort (keeps last_activity_at so idle + # clocks stay continuous; the CHILD carries the live labels). + if _old_sid and _session_commit_succeeded: + try: + _labels_db = getattr(agent, "_session_db", None) + _clear_labels = getattr( + type(_labels_db) if _labels_db is not None else None, + "clear_session_activity_labels", + None, + ) + if callable(_clear_labels): + _clear_labels(_labels_db, _old_sid) + except Exception: + logger.debug( + "failed to clear archived compression parent's activity " + "labels (ignored)", + exc_info=True, + ) + # Notify the context engine that a compaction boundary occurred. Plugin # engines (e.g. hermes-lcm) use boundary_reason="compression" to preserve # DAG lineage / checkpoint per-session state across the boundary instead of @@ -2348,6 +3484,7 @@ def compress_context( record_boundary( agent.context_compressor, used_fallback=_compression_used_fallback, + feasibility_skip=_compression_feasibility_skip, ) else: agent.context_compressor._verify_compaction_cleared_threshold = True @@ -2360,6 +3497,13 @@ def compress_context( reset_file_dedup(task_id) except Exception: pass + # Same for the skill_view repeat-view dedup: a post-compression + # re-view must return the full skill content again. + try: + from tools.skills_tool import reset_skill_view_dedup + reset_skill_view_dedup(task_id) + except Exception: + pass logger.info( "context compression done: session=%s messages=%d->%d rough_tokens=~%s awaiting_real_usage=true", diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 3ca96898cf..1884c06489 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -70,8 +70,9 @@ from agent.model_metadata import ( ) from agent.process_bootstrap import _install_safe_stdio from agent.prompt_caching import ( - apply_anthropic_cache_control, + build_prompt_cache_plan, strip_anthropic_cache_control, + strip_anthropic_tool_cache_control, ) from agent.retry_utils import ( adaptive_rate_limit_backoff, @@ -194,6 +195,54 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text agent._stream_needs_break = True +def _is_copilot_provider(agent: Any) -> bool: + """Delegate to ``AIAgent._is_copilot_provider`` (single owner of the check). + + ``agent.provider`` is not always the normalized ``copilot`` slug — + ``/model`` and profile configs can leave the alias ``github-copilot`` (or + ``github``) in place, and a bare ``provider == "copilot"`` gate silently + skips credential recovery for those spellings. + """ + try: + return bool(agent._is_copilot_provider()) + except Exception: + return (getattr(agent, "provider", "") or "").strip().lower() in { + "copilot", + "github-copilot", + "github", + } + + +def _is_stale_copilot_credential_error(status_code: Optional[int], error_message: str) -> bool: + """Detect a Copilot 400 that is really a STALE / DEGRADED credential. + + Copilot surfaces a stale or degraded credential as an HTTP 400 rather than a + clean 401. Two body markers indicate this class: + + - ``model_not_available_for_integrator`` — the request reached the + restricted ``copilot-language-server`` integrator (the server's fallback + when it receives a raw OAuth token instead of an exchanged API token), + whose model allowlist omits enterprise-only models. + - ``model_not_supported`` / "the requested model is not supported" — the + cached bearer's Copilot entitlement rotated out from under a long-lived + process. + + Matched narrowly (status 400 AND a specific marker) so a genuinely wrong + model name — a real 400 — never triggers the single-shot re-exchange. The + caller enforces copilot-provider scoping and the single-shot guard. + """ + lowered = (error_message or "").lower() + is_400 = status_code == 400 or "error code: 400" in lowered + if not is_400: + return False + return ( + "model_not_available_for_integrator" in lowered + or "not available for integrator" in lowered + or "model_not_supported" in lowered + or "the requested model is not supported" in lowered + ) + + def _image_error_max_dimension(error: Exception) -> Optional[int]: """Extract a provider-reported image dimension ceiling, if present.""" parts = [] @@ -689,6 +738,92 @@ _CONTENT_POLICY_RECOVERY_HINT = ( ) +# Memo for the send-path tool-call argument canonicalization inside +# run_conversation(). That pass re-canonicalizes the arguments string of +# EVERY historical tool call on EVERY API-call iteration (quadratic in +# session tool-call count), and the api_messages copies share the exact +# argument string objects with the persisted history, so the same strings +# come through unchanged iteration after iteration. +# +# Soundness: canonicalization is a pure, deterministic function of the +# input string (fixed separators, sort_keys=True), so a value-keyed memo +# is exact — equal inputs always produce the canonical form computed the +# first time. Malformed strings raise out of json.loads BEFORE anything +# is stored, so the repair fallback below is never memoized and reruns on +# every occurrence, exactly as before. Bounded FIFO eviction mirrors the +# _MSG_TOKENS_CACHE idiom in agent/model_metadata.py. +_CANON_ARGS_CACHE: Dict[str, str] = {} +_CANON_ARGS_CACHE_MAX = 4096 +# Count bound alone doesn't bound MEMORY: write_file/patch argument strings +# run 100KB+, so 4096 entries could pin ~800MB in a long-lived gateway +# process. The byte budget keeps the memo effective for the common case +# (args ~0.5-2KB) while bounding the worst case. +_CANON_ARGS_CACHE_MAX_BYTES = 32 * 1024 * 1024 +_canon_args_cache_bytes = 0 + + +def _canonicalize_tool_call_arguments(arg_str: str) -> str: + """Return the canonical wire form of a tool-call arguments JSON string. + + Raises whatever ``json.loads`` raises on malformed input; the caller + falls back to ``_repair_tool_call_arguments``, exactly as before. + """ + global _canon_args_cache_bytes + cached = _CANON_ARGS_CACHE.get(arg_str) + if cached is not None: + return cached + canonical = json.dumps( + json.loads(arg_str), separators=(",", ":"), sort_keys=True, + ) + _CANON_ARGS_CACHE[arg_str] = canonical + _canon_args_cache_bytes += len(arg_str) + len(canonical) + while len(_CANON_ARGS_CACHE) > _CANON_ARGS_CACHE_MAX or ( + _canon_args_cache_bytes > _CANON_ARGS_CACHE_MAX_BYTES + and len(_CANON_ARGS_CACHE) > 1 + ): + try: + evicted_key = next(iter(_CANON_ARGS_CACHE)) + evicted_val = _CANON_ARGS_CACHE.pop(evicted_key) + _canon_args_cache_bytes -= len(evicted_key) + len(evicted_val) + except (StopIteration, KeyError, RuntimeError): + break + return canonical + + +def _canonicalize_api_tool_calls(api_messages) -> None: + """Canonicalize tool-call argument JSON on the send-path message copy. + + Rewrites each message's ``tool_calls`` in place (copy-on-write for the + tool-call dicts it canonicalizes; the persisted history is untouched). + The pass still traverses every message and tool call each iteration; + the memo above bounds the JSON parse/serialize work to one round-trip + per UNIQUE argument string instead of one per string per iteration — + the quadratic part of the cost. The remaining traversal is pointer + chasing and dict copies, cheap next to a json.loads + json.dumps. + """ + for am in api_messages: + tcs = am.get("tool_calls") + if not tcs: + continue + new_tcs = [] + for tc in tcs: + if isinstance(tc, dict) and "function" in tc: + try: + tc = {**tc, "function": { + **tc["function"], + "arguments": _canonicalize_tool_call_arguments( + tc["function"]["arguments"] + ), + }} + except Exception: + tc["function"]["arguments"] = _repair_tool_call_arguments( + tc["function"]["arguments"], + tc["function"].get("name", "?"), + ) + new_tcs.append(tc) + am["tool_calls"] = new_tcs + + def _invalid_tool_name_error_content(name: str, valid_tool_names) -> str: """Error-result content for a tool call whose name isn't a real tool. @@ -893,7 +1028,8 @@ def _redecorate_prompt_cache_for_provider( *, system_message=None, moa_prepared: Optional[Dict[str, Any]] = None, -) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]]: + tools_for_api: Optional[List[Dict[str, Any]]] = None, +) -> tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]] | tuple[List[Dict[str, Any]], Optional[Dict[str, Any]], List[Dict[str, Any]]]: """Strip and re-apply cache_control for the *current* provider policy. Decoration runs once per call block before the retry loop for the primary @@ -903,10 +1039,9 @@ def _redecorate_prompt_cache_for_provider( by reshaping at the top of each retry attempt. The source list is the mutated in-flight request (image shrink / ASCII / - reasoning_details recoveries already applied) — never a pristine - pre-decoration snapshot. MoA guidance is peeled, the base is redecorated, - then ``rebase_prepared_request`` re-attaches guidance outside the cached - span. + reasoning_details recoveries already applied), never a pristine + pre-decoration snapshot. MoA guidance is peeled and rebased without + decoration; the acting aggregator plans its resolved destination later. """ messages: List[Dict[str, Any]] = [ dict(m) if isinstance(m, dict) else m for m in (api_messages or []) @@ -917,6 +1052,21 @@ def _redecorate_prompt_cache_for_provider( messages = _peel_moa_guidance(messages, guidance) strip_anthropic_cache_control(messages) + planned_tools = strip_anthropic_tool_cache_control( + tools_for_api if tools_for_api is not None else getattr(agent, "tools", []) + ) + + if prepared is not None and getattr(agent, "provider", None) == "moa": + # Prepared MoA state is canonical: the synchronous acting-aggregator + # sender owns its destination-local cache plan after it resolves the slot. + completions = getattr(getattr(agent.client, "chat", None), "completions", None) + rebase = getattr(completions, "rebase_prepared_request", None) + if callable(rebase): + prepared = rebase(prepared, messages) + messages = prepared["messages"] + if tools_for_api is None: + return messages, prepared + return messages, prepared, planned_tools # Direct attribute access matches the call-block decoration site — the # flags are unconditionally initialized on AIAgent, and a getattr @@ -924,31 +1074,25 @@ def _redecorate_prompt_cache_for_provider( if agent._use_prompt_caching: _ensure_cached_system_prompt_static(agent, system_message=system_message) static = getattr(agent, "_cached_system_prompt_static", None) - messages = apply_anthropic_cache_control( + direct_tool_cache = getattr( + agent, + "_direct_native_anthropic_tool_cache_capability", + lambda: False, + )() + plan = build_prompt_cache_plan( messages, + planned_tools, cache_ttl=agent._cache_ttl, native_anthropic=agent._use_native_cache_layout, static_system_prefix=static if isinstance(static, str) else None, + direct_native_tool_cache=direct_tool_cache, ) + messages = plan.messages + planned_tools = plan.tools - if ( - prepared is not None - and getattr(agent, "provider", None) == "moa" - ): - # No `and guidance` here: guidance=None is a real prepared shape - # (all-references-failed / silent degraded policy builds the - # prepared request without attaching guidance), and the MoA facade - # sends prepared["messages"] — not api_kwargs["messages"] — so the - # rebase must refresh the prepared object even when there is no - # guidance to re-attach. rebase_prepared_request handles falsy - # guidance by copying the messages and skipping the attach. - completions = getattr(getattr(agent.client, "chat", None), "completions", None) - rebase = getattr(completions, "rebase_prepared_request", None) - if callable(rebase): - prepared = rebase(prepared, messages) - messages = prepared["messages"] - - return messages, prepared + if tools_for_api is None: + return messages, prepared + return messages, prepared, planned_tools def _apply_context_engine_selection( @@ -1661,29 +1805,7 @@ def run_conversation( for am in api_messages: if isinstance(am.get("content"), str): am["content"] = am["content"].strip() - for am in api_messages: - tcs = am.get("tool_calls") - if not tcs: - continue - new_tcs = [] - for tc in tcs: - if isinstance(tc, dict) and "function" in tc: - try: - args_obj = json.loads(tc["function"]["arguments"]) - tc = {**tc, "function": { - **tc["function"], - "arguments": json.dumps( - args_obj, separators=(",", ":"), - sort_keys=True, - ), - }} - except Exception: - tc["function"]["arguments"] = _repair_tool_call_arguments( - tc["function"]["arguments"], - tc["function"].get("name", "?"), - ) - new_tcs.append(tc) - am["tool_calls"] = new_tcs + _canonicalize_api_tool_calls(api_messages) # Proactively strip any surrogate characters before the API call. # Models served via Ollama (Kimi K2.5, GLM-5, Qwen) can return @@ -1700,12 +1822,8 @@ def run_conversation( # regardless of ordering (a single-space pad here previously had to # be sequenced after normalization to survive, forking the concept). - # Apply Anthropic prompt caching for Claude models on native - # Anthropic, OpenRouter, and third-party Anthropic-compatible - # gateways. Auto-detected: if ``_use_prompt_caching`` is set, inject - # cache_control breakpoints for the static system prefix, full system - # prompt, and last two messages (or the legacy system-and-3 layout - # when no static prefix is available). + # Build the request-local cache sections only after every transcript + # mutation. The canonical tool registry stays undecorated. # # Runs LAST, after every message mutation above. Marking earlier # defeats the prefix stability the mutations exist to create: @@ -1720,10 +1838,12 @@ def run_conversation( # exactly the point the breakpoints were meant to protect. Marking # last also keeps breakpoints off messages that the orphan sweep or # the thinking-only drop is about to remove or merge away. - if agent._use_prompt_caching: + tools_for_api = agent.tools + if agent._use_prompt_caching and agent.provider != "moa": _static_system_prefix = getattr(agent, "_cached_system_prompt_static", None) - api_messages = apply_anthropic_cache_control( + _initial_cache_plan = build_prompt_cache_plan( api_messages, + tools_for_api, cache_ttl=agent._cache_ttl, native_anthropic=agent._use_native_cache_layout, static_system_prefix=( @@ -1731,7 +1851,10 @@ def run_conversation( if isinstance(_static_system_prefix, str) else None ), + direct_native_tool_cache=agent._direct_native_anthropic_tool_cache_capability(), ) + api_messages = _initial_cache_plan.messages + tools_for_api = _initial_cache_plan.tools # Build a persistent-MoA request before measuring compression pressure. # MoA reference output is injected into the aggregator prompt, but it @@ -2067,15 +2190,22 @@ def run_conversation( # fallback refreshes the policy flags, but the decorated list # still carries the primary's breakpoints (or none). Strip and # re-render for the current provider before building kwargs. - api_messages, _moa_prepared_request = ( + api_messages, _moa_prepared_request, tools_for_api = ( _redecorate_prompt_cache_for_provider( agent, api_messages, system_message=system_message, moa_prepared=_moa_prepared_request, + tools_for_api=tools_for_api, ) ) - api_kwargs = agent._build_api_kwargs(api_messages) + if tools_for_api == agent.tools: + api_kwargs = agent._build_api_kwargs(api_messages) + else: + api_kwargs = agent._build_api_kwargs( + api_messages, + tools_for_api=tools_for_api, + ) if agent._force_ascii_payload: _sanitize_structure_non_ascii(api_kwargs) if agent.api_mode == "codex_responses": @@ -2083,6 +2213,7 @@ def run_conversation( api_kwargs, allow_stream=False, is_github_responses=agent._is_copilot_url(), + sanitize_harmony_tokens=agent._is_codex_backend(), ) # Copilot x-initiator: the first API call of a user turn is # marked "user" so Copilot bills a premium request; tool-loop @@ -2242,6 +2373,7 @@ def run_conversation( next_api_kwargs, allow_stream=False, is_github_responses=agent._is_copilot_url(), + sanitize_harmony_tokens=agent._is_codex_backend(), ) if _use_streaming: return agent._interruptible_streaming_api_call( @@ -2540,7 +2672,7 @@ def run_conversation( # Terminal — flush buffered retry trace so user sees what happened. agent._flush_status_buffer() agent._emit_status(f"❌ Max retries ({max_retries}) exceeded for invalid responses. Giving up.") - logger.error(f"{agent.log_prefix}Invalid API response after {max_retries} retries.") + logger.error("%sInvalid API response after %d retries.", agent.log_prefix, max_retries) agent._persist_session(messages, conversation_history) _final_response = f"Invalid API response after {max_retries} retries: {_failure_hint}" return { @@ -2555,7 +2687,7 @@ def run_conversation( # Backoff before retry — jittered exponential: 5s base, 120s cap wait_time = jittered_backoff(retry_count, base_delay=5.0, max_delay=120.0) agent._buffer_vprint(f"⏳ Retrying in {wait_time:.1f}s ({_failure_hint})...") - logger.warning(f"Invalid API response (retry {retry_count}/{max_retries}): {', '.join(error_details)} | Provider: {provider_name}") + logger.warning("Invalid API response (retry %d/%d): %s | Provider: %s", retry_count, max_retries, ', '.join(error_details), provider_name) # Sleep in small increments to stay responsive to interrupts sleep_end = time.time() + wait_time @@ -3859,7 +3991,7 @@ def run_conversation( print(f"{agent.log_prefix} • Verify stored credentials: {_dhh}/auth.json") print(f"{agent.log_prefix} • Switch providers temporarily: /model --provider openrouter") if ( - agent.provider == "copilot" + _is_copilot_provider(agent) and status_code == 401 and not _retry.copilot_auth_retry_attempted ): @@ -4467,7 +4599,7 @@ def run_conversation( agent._flush_status_buffer() agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached for payload-too-large error.", force=True) agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error(f"{agent.log_prefix}413 compression failed after {max_compression_attempts} attempts.") + logger.error("%s413 compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) agent._persist_session(messages, conversation_history) _final_response = f"Request payload too large: max compression attempts ({max_compression_attempts}) reached." return { @@ -4536,7 +4668,7 @@ def run_conversation( agent._flush_status_buffer() agent._vprint(f"{agent.log_prefix}❌ Payload too large and cannot compress further.", force=True) agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error(f"{agent.log_prefix}413 payload too large. Cannot compress further.") + logger.error("%s413 payload too large. Cannot compress further.", agent.log_prefix) agent._persist_session(messages, conversation_history) _final_response = "Request payload too large (413). Cannot compress further." return { @@ -4609,7 +4741,7 @@ def run_conversation( agent._flush_status_buffer() agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True) agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error(f"{agent.log_prefix}Context compression failed after {max_compression_attempts} attempts.") + logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) agent._persist_session(messages, conversation_history) _final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached." return { @@ -4698,6 +4830,13 @@ def run_conversation( provider=agent.provider, api_mode=agent.api_mode, ) + # Persist an explicit provider-reported limit before + # compression/retry. The next request can be rate + # limited, omit usage, or the process can restart; none + # of those should discard metadata the provider already + # confirmed. Keep the probe flags as a best-effort + # post-success retry if this write cannot complete. + save_context_length(agent.model, agent.base_url, new_ctx) # Context probing flags — only set on built-in # compressor (plugin engines manage their own). This # value came from the provider, so it is safe to cache. @@ -4721,7 +4860,7 @@ def run_conversation( agent._flush_status_buffer() agent._vprint(f"{agent.log_prefix}❌ Max compression attempts ({max_compression_attempts}) reached.", force=True) agent._vprint(f"{agent.log_prefix} 💡 Try /new to start a fresh conversation, or /compress to retry compression.", force=True) - logger.error(f"{agent.log_prefix}Context compression failed after {max_compression_attempts} attempts.") + logger.error("%sContext compression failed after %d attempts.", agent.log_prefix, max_compression_attempts) agent._persist_session(messages, conversation_history) _final_response = f"Context length exceeded: max compression attempts ({max_compression_attempts}) reached." return { @@ -4779,7 +4918,7 @@ def run_conversation( agent._flush_status_buffer() agent._vprint(f"{agent.log_prefix}❌ Context length exceeded and cannot compress further.", force=True) agent._vprint(f"{agent.log_prefix} 💡 The conversation has accumulated too much content. Try /new to start fresh, or /compress to manually trigger compression.", force=True) - logger.error(f"{agent.log_prefix}Context length exceeded: {new_tokens:,} tokens. Cannot compress further.") + logger.error("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}") agent._persist_session(messages, conversation_history) _final_response = f"Context length exceeded ({new_tokens:,} tokens). Cannot compress further." return { @@ -4865,6 +5004,32 @@ def run_conversation( ) and not is_context_length_error if is_client_error: + # Copilot self-heal BEFORE fallback: a stale/degraded + # credential surfaces as a 400 + # ``model_not_available_for_integrator`` / + # ``model_not_supported`` (not a clean 401), so the 401 + # refresh path above never fired. Force a fresh token + # exchange + client rebuild and retry once on the SAME + # provider — a fresh 437-char API token routes to the + # correct integrator and the model becomes available again. + # Single-shot guard prevents looping on a genuinely + # unavailable model. Copilot-scoped so other providers' + # real 400s are untouched. + if ( + _is_copilot_provider(agent) + and not _retry.copilot_stale_cred_retry_attempted + and _is_stale_copilot_credential_error( + status_code, str(getattr(api_error, "message", "") or api_error) + ) + ): + _retry.copilot_stale_cred_retry_attempted = True + if agent._try_recover_stale_copilot_credential(): + agent._buffer_vprint( + "🔐 Copilot credential re-exchanged after " + "model_not_available 400. Retrying request..." + ) + retry_count = 0 + continue # Try fallback before aborting — a different provider may # not have the same issue (rate limit, auth, etc.). Only # announce the attempt when a fallback chain actually @@ -5019,7 +5184,7 @@ def run_conversation( f"{agent.log_prefix} for localhost, or add the server's cert to your trust store.", force=True, ) - logger.error(f"{agent.log_prefix}Non-retryable client error: {api_error}") + logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error) # Skip session persistence when the error is likely # context-overflow related (status 400 + large session). # Persisting the failed user message would make the diff --git a/agent/credential_pool.py b/agent/credential_pool.py index 08b0c0ea6b..adb24f75a6 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -28,10 +28,13 @@ from hermes_cli.auth import ( _auth_store_lock, _codex_access_token_is_expiring, _decode_jwt_claims, + _global_auth_file_path, _load_auth_store, _load_provider_state, + _load_provider_state_with_source, _resolve_kimi_base_url, _resolve_zai_base_url, + _same_path, _save_auth_store, _save_provider_state, _store_provider_state, @@ -792,7 +795,7 @@ class CredentialPool: device_code-sourced entries; env/API-key-sourced entries have no auth.json shadow to sync from. """ - if self.provider != "openai-codex" or entry.source != "device_code": + if self.provider != "openai-codex" or entry.source not in ("device_code", "manual:device_code"): return entry try: with _auth_store_lock(): @@ -808,19 +811,43 @@ class CredentialPool: # Adopt auth.json tokens when either side differs. Codex refresh # tokens are single-use too, so a fresh refresh_token from # another process means our entry's pair is consumed/stale. + # + # Also adopt when the store has a refresh_token but no + # access_token — another process may have rotated the pair + # and the store entry's access_token was already consumed; + # the important signal is the refresh_token difference. entry_access = entry.access_token or "" entry_refresh = entry.refresh_token or "" + should_adopt = False if store_access and ( store_access != entry_access or (store_refresh and store_refresh != entry_refresh) ): + should_adopt = True + elif ( + store_refresh + and store_refresh != entry_refresh + and not store_access + ): + # Store has only a refresh_token (no access_token) — + # another process rotated the pair. Adopt the + # refresh_token so we don't replay the consumed one. + logger.info( + "Pool entry %s: auth.json has newer refresh_token " + "but no access_token; adopting refresh_token to " + "avoid replaying consumed token", + entry.id, + ) + should_adopt = True + + if should_adopt: logger.debug( "Pool entry %s: syncing Codex tokens from auth.json " "(refreshed by another process)", entry.id, ) field_updates: Dict[str, Any] = { - "access_token": store_access, + "access_token": store_access or entry.access_token, "refresh_token": store_refresh or entry.refresh_token, "last_status": None, "last_status_at": None, @@ -1039,32 +1066,58 @@ class CredentialPool: try: with _auth_store_lock(): auth_store = _load_auth_store() - # Decide BEFORE writing whether this profile is reading the - # grant from the global root (no own providers. block) vs. - # genuinely shadowing it. A pool refresh rotates single-use - # OAuth refresh tokens, so a profile that resolved the grant - # from root MUST write the rotated chain back to root too — - # otherwise root keeps a revoked refresh token and every other - # profile reading the stale root grant dies with - # refresh_token_reused / invalid_grant once its access token - # expires. This mirrors the xAI write-through in - # hermes_cli.auth._save_xai_oauth_tokens (#43589); the pool - # refresh path is the Codex/xAI analog reported in #48415. _wt_provider_id = { "nous": "nous", "openai-codex": "openai-codex", "xai-oauth": "xai-oauth", }.get(self.provider) - write_through_to_root = bool(_wt_provider_id) and not ( - isinstance(auth_store.get("providers"), dict) - and isinstance( - auth_store["providers"].get(_wt_provider_id), dict - ) - ) + # Resolve state and track which store it came from — the + # source path tells us whether this profile genuinely owns + # its provider block or is reading from the global root. + # #74339: the old key-presence check decided write-through + # on whether the profile had ``providers.`` BEFORE the + # save — correct for the first refresh but self-sealing + # because ``_store_provider_state`` unconditionally creates + # that key inside the same function. Once the profile has + # the key, every subsequent refresh silently disables the + # root write-through and root keeps a revoked refresh token. + # + # Fix: use ``_load_provider_state_with_source`` to learn + # where the state was resolved from. When the grant was + # resolved from the global root, write back *only* to root + # and skip ``_store_provider_state`` for the profile so the + # profile does not accrue a shadowing ``providers.`` + # key that blocks both the root fallback and the write-through + # on subsequent calls. if self.provider == "nous": - state = _load_provider_state(auth_store, "nous") + state, source_path = _load_provider_state_with_source( + auth_store, "nous" + ) if state is None: return + elif self.provider == "openai-codex": + state, source_path = _load_provider_state_with_source( + auth_store, "openai-codex" + ) + if not isinstance(state, dict): + return + elif self.provider == "xai-oauth": + state, source_path = _load_provider_state_with_source( + auth_store, "xai-oauth" + ) + if not isinstance(state, dict): + return + else: + return + + global_root = _global_auth_file_path() + is_from_root = bool( + source_path is not None + and global_root is not None + and _same_path(source_path, global_root) + ) + + if self.provider == "nous": state["access_token"] = entry.access_token if entry.refresh_token: state["refresh_token"] = entry.refresh_token @@ -1082,12 +1135,8 @@ class CredentialPool: state[extra_key] = val if entry.inference_base_url: state["inference_base_url"] = entry.inference_base_url - _store_provider_state(auth_store, "nous", state, set_active=False) elif self.provider == "openai-codex": - state = _load_provider_state(auth_store, "openai-codex") - if not isinstance(state, dict): - return tokens = state.get("tokens") if not isinstance(tokens, dict): return @@ -1096,12 +1145,8 @@ class CredentialPool: tokens["refresh_token"] = entry.refresh_token if entry.last_refresh: state["last_refresh"] = entry.last_refresh - _store_provider_state(auth_store, "openai-codex", state, set_active=False) elif self.provider == "xai-oauth": - state = _load_provider_state(auth_store, "xai-oauth") - if not isinstance(state, dict): - return tokens = state.get("tokens") if not isinstance(tokens, dict): return @@ -1110,16 +1155,26 @@ class CredentialPool: tokens["refresh_token"] = entry.refresh_token if entry.last_refresh: state["last_refresh"] = entry.last_refresh - _store_provider_state(auth_store, "xai-oauth", state, set_active=False) - else: - return - - _save_auth_store(auth_store) - if write_through_to_root and _wt_provider_id: + if is_from_root and _wt_provider_id: + # Grant was resolved from root — write back to root + # only. Do NOT call _store_provider_state on the + # profile auth_store (it would create a shadowing + # providers. key that disables write-through on + # the next refresh — #74339). + # _load_provider_state has root fallback, so the + # profile can always read fresh tokens from root + # without needing its own providers block. _write_through_provider_state_to_global_root( _wt_provider_id, state ) + else: + # Profile genuinely owns this provider — write to + # the profile store as normal. + _store_provider_state( + auth_store, self.provider, state, set_active=False + ) + _save_auth_store(auth_store) except Exception as exc: logger.debug("Failed to sync %s pool entry back to auth store: %s", self.provider, exc) @@ -2336,6 +2391,20 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup token, source = resolve_copilot_token() if token: api_token, enterprise_base_url = get_copilot_api_token(token) + # Observability: get_copilot_api_token falls back to returning + # the RAW token when the exchange fails. A raw ~40-char token + # sent to the Copilot API is routed to the fallback + # "copilot-language-server" integrator, whose allowlist omits + # enterprise-only models (claude-opus-4.8) → HTTP 400 on every + # turn. exchange_copilot_token now retries + reuses a persisted + # JWT, so this should be rare; surface it at WARNING so a + # recurrence is visible in logs instead of failing silently. + if api_token == token and not enterprise_base_url: + logger.warning( + "Copilot token exchange degraded to RAW token (exchange " + "unavailable); enterprise-only models may 400 with " + "model_not_available_for_integrator until exchange recovers." + ) source_name = "gh_cli" if "gh" in source.lower() else f"env:{source}" if not _is_suppressed(provider, source_name): active_sources.add(source_name) @@ -2506,6 +2575,20 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool changed = False active_sources: Set[str] = set() + # Copilot has its own dedicated seeding branch (see `_seed_credentials` + # for provider == "copilot") which exchanges the raw ghu_ OAuth token + # for the ~437-char api token via `get_copilot_api_token`. If we let + # the generic env-var loop below run for copilot, it re-reads + # COPILOT_GITHUB_TOKEN from .env and shoves the RAW 40-char token in + # as `access_token`, overwriting the correctly-exchanged token. That + # bypasses the Copilot token exchange entirely and causes 400s with + # "not available for integrator copilot-language-server" (the server's + # fallback integrator when it receives a raw OAuth token instead of + # an api token). Skip the generic loop here — the copilot-specific + # branch is authoritative. + if provider == "copilot": + return False, active_sources + # Prefer ~/.hermes/.env over os.environ — the user's config file is the # authoritative source for Hermes credentials. Stale env vars from parent # processes (Codex CLI, test scripts, etc.) should not override deliberate diff --git a/agent/curator_backup.py b/agent/curator_backup.py index ca0cc76362..8a65825464 100644 --- a/agent/curator_backup.py +++ b/agent/curator_backup.py @@ -541,6 +541,33 @@ def _restore_cron_skill_links(snapshot_dir: Path) -> Dict[str, Any]: +def _unstage(moved: List[Tuple[Path, Path]]) -> List[str]: + """Move staged entries back to their original paths. + + ``shutil.move`` moves *into* an existing destination directory rather than + replacing it, so a partially-completed extract leaves debris that would + otherwise bury the user's real skill one level deeper + (``skills/foo/foo/``) while the tree still looks populated. Clear whatever + the failed extract created at each original path first. The staged copy is + authoritative, and the pre-rollback safety snapshot is the undo handle for + the extract's own output. + + Returns the names that could not be restored, so the caller can report an + incomplete recovery instead of claiming the state was restored. + """ + failed: List[str] = [] + for orig, dest in moved: + try: + if orig.is_dir() and not orig.is_symlink(): + shutil.rmtree(orig) + elif orig.exists() or orig.is_symlink(): + orig.unlink() + shutil.move(str(dest), str(orig)) + except OSError: + failed.append(orig.name) + return failed + + def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]]: """Restore ``~/.hermes/skills/`` from a snapshot. @@ -609,11 +636,7 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path] moved.append((entry, dest)) except OSError as e: # Best-effort rollback of the move - for orig, dest in moved: - try: - shutil.move(str(dest), str(orig)) - except OSError: - pass + _unstage(moved) try: shutil.rmtree(staged, ignore_errors=True) except OSError: @@ -638,12 +661,30 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path] # Python < 3.12 — no filter kwarg tf.extractall(str(skills)) except (OSError, tarfile.TarError) as e: - # Best-effort recover: move staged contents back - for orig, dest in moved: + # Best-effort recover. A partial extract can leave entries the + # original tree never had, so drop those first, otherwise the + # "restored" tree is the user's skills plus a slice of the snapshot. + staged_names = {orig.name for orig, _ in moved} + for entry in list(skills.iterdir()): + if entry.name in _EXCLUDE_TOP_LEVEL or entry.name in staged_names: + continue try: - shutil.move(str(dest), str(orig)) + if entry.is_dir() and not entry.is_symlink(): + shutil.rmtree(entry) + else: + entry.unlink() except OSError: pass + unrestored = _unstage(moved) + if unrestored: + # Do not claim a clean restore we did not achieve, and keep the + # staging dir so the entries can be recovered by hand. + return ( + False, + f"snapshot extract failed: {e} - could not restore " + f"{', '.join(sorted(unrestored))}; staged copies kept at {staged}", + None, + ) try: shutil.rmtree(staged, ignore_errors=True) except OSError: diff --git a/agent/delegation_context.py b/agent/delegation_context.py index b80bbbe00f..41fe7f56ac 100644 --- a/agent/delegation_context.py +++ b/agent/delegation_context.py @@ -31,11 +31,21 @@ KANBAN_ENV_KEYS: tuple[str, ...] = ( @contextmanager -def delegated_child_context() -> Iterator[None]: - """Mark the current execution context as a delegate_task child.""" +def delegated_child_context(session_id: str | None = None) -> Iterator[None]: + """Mark child execution and isolate its task-local session identity. + + Child construction calls ``set_current_session_id`` internally, so even a + context entered without an id must restore the parent's ContextVar. Child + execution passes its explicit id and receives it only for this scope. + """ token = _DELEGATED_CHILD_CONTEXT.set(True) try: - yield + # Import lazily: session_context calls is_delegated_child_context() when + # deciding whether the compatibility os.environ mirror is safe. + from gateway.session_context import scoped_current_session_id + + with scoped_current_session_id(session_id): + yield finally: _DELEGATED_CHILD_CONTEXT.reset(token) diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 27df08f58e..8ac0b6c872 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -836,6 +836,19 @@ def classify_api_error( if classified is not None: return classified + # Local MoA streaming compatibility errors are adapter-shape bugs, not a + # provider outage. Falling back to another model would silently switch the + # user's selected MoA route to a single-model answer (#55933 follow-up). + if provider_lower == "moa" and ( + "'types.SimpleNamespace' object is not iterable" in str(error) + or "'types.SimpleNamespace' object has no attribute 'index'" in str(error) + ): + return _result( + FailoverReason.format_error, + retryable=False, + should_fallback=False, + ) + # Local MoA config drift is deterministic: a persisted session can retain # a preset name that was later renamed/deleted. Retrying the same lookup # cannot recover and makes a clear config error look like an API outage. diff --git a/agent/interrupt_compat.py b/agent/interrupt_compat.py new file mode 100644 index 0000000000..bf56849495 --- /dev/null +++ b/agent/interrupt_compat.py @@ -0,0 +1,35 @@ +"""Compatibility helper for explicit agent stop producers.""" + +from __future__ import annotations + +import inspect +from typing import Any + + +def request_hard_interrupt(agent: Any, message: str | None = None) -> bool: + """Request an explicit stop, falling back to the legacy interrupt ABI. + + New agents expose ``hard_interrupt(message=None)``. Third-party agents and + old test doubles may only expose ``interrupt(message=None)``; keep those + usable without sending the newer ``hard_cancel=`` keyword they do not know. + Returns ``False`` only when neither callable is available. + """ + # Avoid treating a dynamic ``__getattr__`` proxy (notably an unspecced + # ``MagicMock`` or a third-party RPC facade) as if it genuinely implements + # the new ABI. Static lookup proves the attribute exists on the instance or + # its type before normal descriptor binding retrieves the callable. + try: + inspect.getattr_static(agent, "hard_interrupt") + except AttributeError: + interrupt = None + else: + interrupt = getattr(agent, "hard_interrupt", None) + if not callable(interrupt): + interrupt = getattr(agent, "interrupt", None) + if not callable(interrupt): + return False + if message is None: + interrupt() + else: + interrupt(message) + return True diff --git a/agent/lsp/install.py b/agent/lsp/install.py index 2671e7ccd3..fc9bea5930 100644 --- a/agent/lsp/install.py +++ b/agent/lsp/install.py @@ -35,6 +35,7 @@ from pathlib import Path from typing import Any, Dict, Optional from hermes_cli._subprocess_compat import windows_hide_flags +from hermes_constants import find_node_executable logger = logging.getLogger("agent.lsp.install") @@ -249,9 +250,12 @@ def _install_npm( peer deps that npm doesn't auto-pull (typescript-language-server needs ``typescript`` next to it; intelephense ships standalone). """ - npm = shutil.which("npm") + # Managed npm first: $HERMES_HOME/node is not on an arbitrary process's + # PATH, so a bare which() misses the Node that Hermes installed and + # reports "npm not on PATH" on a machine that has a perfectly good one. + npm = find_node_executable("npm") if npm is None: - logger.info("[install] cannot install %s: npm not on PATH", pkg) + logger.info("[install] cannot install %s: no usable npm found", pkg) return None staging = hermes_lsp_bin_dir().parent # /lsp/ install_targets = [pkg] + list(extra_pkgs or []) diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 173816c8cb..26e9523ec3 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -13,6 +13,7 @@ import logging import re import threading from concurrent.futures import ThreadPoolExecutor, wait as _futures_wait +from types import SimpleNamespace from typing import Any from agent.auxiliary_client import call_llm @@ -382,6 +383,8 @@ def _merge_slot_extra_body( def _maybe_apply_moa_cache_control( messages: list[dict[str, Any]], runtime: dict[str, Any], + *, + cache_disabled: bool | None = None, ) -> list[dict[str, Any]]: """Decorate an advisor or aggregator request with cache_control when its route honors it. @@ -395,17 +398,27 @@ def _maybe_apply_moa_cache_control( Returns the messages unchanged on any resolution error or when the policy says the route doesn't honor markers. + + ``cache_disabled`` (or the live config when omitted) is stamped onto the + policy stub so ``prompt_caching.cache_ttl: off`` is not bypassed by the + blank-agent pattern (#76085). """ try: - from types import SimpleNamespace - - from agent.agent_runtime_helpers import anthropic_prompt_cache_policy + from agent.agent_runtime_helpers import ( + anthropic_prompt_cache_policy, + blank_cache_policy_stub, + ) from agent.prompt_caching import apply_anthropic_cache_control + # Prefer an explicit kwarg, then a snapshot on the runtime dict + # (threaded from the live agent), else config via the stub factory. + if cache_disabled is None and "_cache_disabled" in runtime: + cache_disabled = runtime.get("_cache_disabled") + # The policy function reads agent.* only as fallbacks for kwargs we - # don't pass; provide a stub so the slot is judged purely on its own - # resolved runtime. - stub = SimpleNamespace(provider="", base_url="", api_mode="", model="") + # don't pass; blank_cache_policy_stub is the only sanctioned stub + # so _cache_disabled cannot be left off again (#76085). + stub = blank_cache_policy_stub(cache_disabled) should_cache, native_layout = anthropic_prompt_cache_policy( stub, provider=runtime.get("provider") or "", @@ -431,6 +444,7 @@ def _run_reference( max_tokens: int | None = None, reference_timeout: float | None = None, context_length_cache: Any = None, + cache_disabled: bool | None = None, ) -> tuple[str, str, Any]: """Call one reference model and return ``(label, text, accounting)``. @@ -492,7 +506,12 @@ def _run_reference( # caching is opt-in per request. OpenAI-family advisors are untouched # (their caching is automatic; markers are ignored harmlessly, but we # only decorate when the policy says the route honors them). - messages = _maybe_apply_moa_cache_control(messages, runtime) + # Pin the live agent disable onto the runtime so advisor decoration + # tracks conversation state, not a fresh config re-read (#76085). + cache_runtime = runtime + if cache_disabled is not None: + cache_runtime = {**runtime, "_cache_disabled": cache_disabled} + messages = _maybe_apply_moa_cache_control(messages, cache_runtime) # Per-slot max_tokens takes precedence over the preset-level # reference_max_tokens passed in by the caller. This lets each # reference model have its own output cap independently. @@ -796,6 +815,9 @@ def _run_references_parallel( # instead of re-probing metadata sources per reference (dict get/set is # GIL-atomic; a rare duplicate probe on a first-use race is harmless). _ctx_len_cache: dict[tuple[str, str], int | None] = {} + cache_disabled = ( + getattr(agent, "_cache_disabled", None) if agent is not None else None + ) try: for idx, slot in enumerate(reference_models): if slot.get("provider") == "moa": @@ -814,6 +836,7 @@ def _run_references_parallel( max_tokens=max_tokens, reference_timeout=reference_timeout, context_length_cache=_ctx_len_cache, + cache_disabled=cache_disabled, ) ] = idx @@ -1261,6 +1284,19 @@ def aggregate_moa_context( agg_label = _slot_label(aggregator) agg_runtime = _slot_runtime(aggregator) + # Pin the live agent disable onto synthesis decoration so mid-session + # config flips cannot re-enable markers on this path alone (#76085). + # Same not-None guard as _run_reference: stamping None would be a no-op + # (present-None falls through to the config fallback anyway). + agg_cache_runtime = agg_runtime + _agg_cache_disabled = ( + getattr(agent, "_cache_disabled", None) if agent is not None else None + ) + if _agg_cache_disabled is not None: + agg_cache_runtime = { + **agg_runtime, + "_cache_disabled": _agg_cache_disabled, + } try: # Same cache_control decoration as _run_reference's advisor calls # (see _maybe_apply_moa_cache_control) — this synthesis call is a @@ -1273,7 +1309,7 @@ def aggregate_moa_context( # breakpoints, even when the resolved aggregator slot is a # cache-honoring route (e.g. Claude on OpenRouter/native Anthropic). agg_messages = _maybe_apply_moa_cache_control( - [{"role": "user", "content": synth_prompt}], agg_runtime + [{"role": "user", "content": synth_prompt}], agg_cache_runtime ) response = call_llm( task="moa_aggregator", @@ -1300,6 +1336,53 @@ def aggregate_moa_context( ) +def _completed_response_as_stream_chunk(response: Any) -> Any: + """Convert a completed Chat Completions response into one delta stream chunk. + + MoA's outer streaming consumer expects ``choices[0].delta`` chunks. A + completed aggregator response carries ``choices[0].message`` instead; adapt + it here, at the MoA facade boundary, so provider-specific Relay behavior and + other transports remain untouched. + """ + + choices = getattr(response, "choices", None) + first_choice = choices[0] if isinstance(choices, (list, tuple)) and choices else None + message = getattr(first_choice, "message", None) + raw_tool_calls = getattr(message, "tool_calls", None) + tool_call_deltas = None + if isinstance(raw_tool_calls, (list, tuple)) and raw_tool_calls: + tool_call_deltas = [] + for index, tc in enumerate(raw_tool_calls): + function = getattr(tc, "function", None) + tool_call_deltas.append(SimpleNamespace( + index=getattr(tc, "index", index), + id=getattr(tc, "id", None), + type=getattr(tc, "type", None) or "function", + function=SimpleNamespace( + name=getattr(function, "name", None), + arguments=getattr(function, "arguments", None), + ), + )) + delta = SimpleNamespace( + content=getattr(message, "content", None), + tool_calls=tool_call_deltas, + reasoning_content=getattr(message, "reasoning_content", None), + reasoning=getattr(message, "reasoning", None), + reasoning_details=getattr(message, "reasoning_details", None), + ) + choice = SimpleNamespace( + index=getattr(first_choice, "index", 0), + delta=delta, + finish_reason=getattr(first_choice, "finish_reason", None) or "stop", + ) + return SimpleNamespace( + id=getattr(response, "id", None), + model=getattr(response, "model", None), + choices=[choice], + usage=getattr(response, "usage", None), + ) + + def _attach_reference_guidance(agg_messages: list[dict[str, Any]], guidance: str) -> None: """Attach the per-turn reference block at the END of the aggregator prompt. @@ -1609,6 +1692,52 @@ class MoAChatCompletions: max_tokens: Any = agg_kwargs.get("max_tokens") tools: Any = agg_kwargs.get("tools") extra_body: Any = agg_kwargs.get("extra_body") + agg_runtime = _slot_runtime(aggregator) + try: + from agent.agent_runtime_helpers import ( + plan_cache_sections_for_destination, + ) + + guidance = prepared.get("guidance") + planning_messages = agg_messages + if guidance: + planning_messages = peel_reference_guidance( + agg_messages, + str(guidance), + ) + # plan_cache_sections_for_destination never mutates its inputs + # and always returns request-local copies, so the prepared + # state stays canonical. + # Tri-state: only pass a bool when a live agent snapshot exists. + # Prepared-aggregator facades built via __new__ have no _agent; + # getattr(self._agent, ...) raises and bool(None-agent) would + # force False and suppress the planner's config fallback (#76085). + _agent = getattr(self, "_agent", None) + _cache_disabled = ( + getattr(_agent, "_cache_disabled", None) + if _agent is not None + else None + ) + agg_messages, tools = plan_cache_sections_for_destination( + planning_messages, + tools, + provider=agg_runtime.get("provider") or "", + base_url=agg_runtime.get("base_url") or "", + api_mode=agg_runtime.get("api_mode") or "", + model=agg_runtime.get("model") or "", + cache_disabled=_cache_disabled, + ) + if guidance: + _attach_reference_guidance(agg_messages, str(guidance)) + except Exception as exc: # pragma: no cover - cache planning must not block MoA + # Warning, not debug: since the call-block site skips MoA, this + # block is the aggregator's ONLY decoration path — a silent + # failure here ships an undecorated request and regresses the + # exact 0%-cache MoA failure the planning exists to prevent. + logger.warning( + "MoA aggregator cache plan failed — sending undecorated " + "request (cache misses expected): %s", exc, + ) # Record the exact aggregator INPUT (incl. the injected reference # context) into the pending trace so a trace captures what the # aggregator actually saw, not a reconstruction. Traces are a @@ -1649,7 +1778,6 @@ class MoAChatCompletions: # actually governs the aggregator stream, not just call_llm's default. if api_kwargs.get("timeout") is not None: stream_kwargs["timeout"] = api_kwargs["timeout"] - agg_runtime = _slot_runtime(aggregator) # _slot_runtime may carry the provider's request_overrides.extra_body; # pop it and merge with the caller's extra_body (caller wins) so the # explicit kwarg below never collides with **agg_runtime. @@ -1685,6 +1813,14 @@ class MoAChatCompletions: self._pending_trace["aggregator_output"] = _extract_text(_agg_response) except Exception: # pragma: no cover - defensive self._pending_trace["aggregator_output"] = None + if stream and hasattr(_agg_response, "choices"): + # Some aggregator adapters (notably openai-codex Responses) consume + # their provider stream internally and return a completed response + # object even when the acting consumer requested token streaming. + # The outer chat-completions streaming loop expects delta chunks; + # hand it a one-chunk iterator instead of letting it iterate the + # SimpleNamespace response itself (#55933). + return iter((_completed_response_as_stream_chunk(_agg_response),)) return _agg_response def create(self, **api_kwargs: Any) -> Any: @@ -2174,8 +2310,24 @@ def build_moa_facade(agent, preset_name: Any = None) -> MoAClient: except Exception: pass + resolved_preset = preset_name + if resolved_preset is None and getattr(agent, "provider", None) == "moa": + resolved_preset = getattr(agent, "model", None) + + resolved_preset = str(resolved_preset or "default") + try: + from hermes_cli.config import load_config + from hermes_cli.moa_config import normalize_moa_config + + moa_cfg = normalize_moa_config(load_config().get("moa") or {}) + presets = moa_cfg.get("presets") or {} + if resolved_preset not in presets: + resolved_preset = moa_cfg.get("default_preset") or "default" + except Exception: + resolved_preset = "default" + return MoAClient( - str(preset_name or getattr(agent, "model", None) or "default"), + resolved_preset, reference_callback=_moa_reference_relay, # Thread the agent through so the reference fan-out wait can be # aborted on a user interrupt (see _run_references_parallel). diff --git a/agent/model_metadata.py b/agent/model_metadata.py index cec891dbb7..f89d25f4c8 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -273,6 +273,27 @@ CONTEXT_PROBE_TIERS = [ # Default context length when no detection method succeeds. DEFAULT_FALLBACK_CONTEXT = CONTEXT_PROBE_TIERS[0] +# (model, base_url) pairs that already emitted the fallback warning. +# The fallback result itself is deliberately never cached, so without this +# the warning would repeat on every resolution for the same unknown model. +_FALLBACK_WARNED: set = set() + + +def _warn_context_length_fallback(model: str, base_url: str) -> None: + """Warn (once per model+endpoint) that context detection failed and the + hard default is being used, so small-context models (8K, 32K) don't + silently get 256K and cause hard-to-debug API failures.""" + key = (model, base_url or "") + if key in _FALLBACK_WARNED: + return + _FALLBACK_WARNED.add(key) + logger.warning( + "Could not determine context length for model %r (base_url=%s) " + "— falling back to %s tokens. Set model.context_length in " + "config.yaml to override.", + model, base_url or "default", f"{DEFAULT_FALLBACK_CONTEXT:,}", + ) + # Minimum context length required to run Hermes Agent. Models with fewer # tokens cannot maintain enough working memory for tool-calling workflows. # Sessions, model switches, and cron jobs should reject models below this. @@ -634,12 +655,17 @@ def _is_known_provider_base_url(base_url: str) -> bool: def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]: - """Return metadata confirmed only for the Kimi Coding endpoint. + """Return context metadata confirmed for one provider endpoint. Kimi Coding serves K3 under the bare slug ``k3``, but users may also configure or select the public-facing aliases ``kimi-k3`` and ``kimi-k3-cot``. Only canonical ``https://api.kimi.com/coding`` endpoints (legacy Moonshot keys do not serve K3) get the 1 Mi context window. + + NVIDIA NIM serves ``deepseek-ai/deepseek-v4-pro`` with a 262,144-token + window even though DeepSeek's native endpoint serves the V4 family with a + 1M window. Keep the lower limit scoped to NVIDIA instead of weakening the + global model-family metadata. """ normalized = _normalize_base_url(base_url) try: @@ -659,6 +685,18 @@ def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]: and model.strip().lower() in {"k3", "kimi-k3", "kimi-k3-cot"} ): return 1_048_576 + if ( + parsed.scheme.lower() == "https" + and (parsed.hostname or "").lower() == "integrate.api.nvidia.com" + and port in (None, 443) + and parsed.username is None + and parsed.password is None + and parsed.path.rstrip("/") == "/v1" + and not parsed.query + and not parsed.fragment + and model.strip().lower() == "deepseek-ai/deepseek-v4-pro" + ): + return 262_144 return None @@ -1046,7 +1084,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any return cache except Exception as e: - logger.warning(f"Failed to fetch model metadata from OpenRouter: {e}") + logger.warning("Failed to fetch model metadata from OpenRouter: %s", e) if _model_metadata_cache: return _model_metadata_cache disk_cache = _load_model_metadata_disk_cache() @@ -1147,8 +1185,22 @@ def fetch_endpoint_model_metadata( for candidate in candidates: url = candidate.rstrip("/") + "/models" + response = None try: - response = requests.get(url, headers=headers, timeout=(5, 10), verify=_resolve_requests_verify()) + response = requests.get( + url, + headers=headers, + timeout=(5, 10), + verify=_resolve_requests_verify(), + stream=True, + ) + if response.status_code in (401, 403): + logger.debug( + "Model metadata probe received HTTP %s from %s; stopping candidate probing", + response.status_code, + url, + ) + break response.raise_for_status() payload = response.json() cache: Dict[str, Dict[str, Any]] = {} @@ -1198,6 +1250,9 @@ def fetch_endpoint_model_metadata( return cache except Exception as exc: last_error = exc + finally: + if response is not None: + response.close() if last_error: logger.debug("Failed to fetch model metadata from %s/models: %s", normalized, last_error) @@ -2582,6 +2637,9 @@ def get_model_context_length( f"{length:,}", model, default_model, ) return length + # Same silent-256K bug class as the step-9 fallback below — + # warn here too so custom/local endpoints aren't left invisible. + _warn_context_length_fallback(model, base_url) return DEFAULT_FALLBACK_CONTEXT # 4. Anthropic /v1/models API (only for regular API keys, not OAuth) @@ -2754,7 +2812,10 @@ def get_model_context_length( if default_model in model_lower: return length - # 9. Default fallback — 256K + # 9. Default fallback — warn (deduped per model+endpoint) so + # small-context models don't silently get 256K. See + # _warn_context_length_fallback for rationale. + _warn_context_length_fallback(model, base_url) return DEFAULT_FALLBACK_CONTEXT @@ -2965,6 +3026,68 @@ def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int: return count * cost_per_image +def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]: + """Shadow of a message holding only what the provider actually receives. + + Two adjustments to the raw persisted dict: + + * ``api_content`` is a SUBSTITUTE for ``content``, not an addition to it. + ``turn_context.substitute_api_content()`` pops the sidecar and overwrites + ``content`` at every API-bound build site, so exactly one of the two is + ever sent. Counting both double-counts any message whose sidecar differs + from its clean stored content (2.00x on a 40KB sidecar). + + The substitution mirrors that helper's guard exactly: only a non-empty + STRING sidecar on a ``user``/``assistant`` row displaces ``content``. + Any other sidecar shape is popped and discarded on the wire without + touching ``content``, so a shadow that substituted unconditionally + would UNDERcount those rows — the dangerous direction, since it makes + compaction fire too late and the turn dies on a hard context error. + * Base64 image payloads are replaced with a placeholder; they are charged + separately at a flat rate by ``_count_image_tokens``, and counting their + raw chars here would massively overestimate usage. + """ + sidecar = msg.get("api_content") + sidecar_wins = ( + isinstance(sidecar, str) + and bool(sidecar) + and msg.get("role") in ("user", "assistant") + ) + shadow: Dict[str, Any] = {} + for k, v in msg.items(): + if k in ("_anthropic_content_blocks", "reasoning_details"): + continue + if k == "api_content": + # Always popped before the request is built; only counted when it + # actually replaces ``content``. + if sidecar_wins: + shadow["content"] = v + continue + if k == "content": + if sidecar_wins: + # The sidecar wins on the wire; skip the clean copy so the + # same logical content is not counted twice. + continue + if isinstance(v, list): + cleaned = [] + for part in v: + if isinstance(part, dict): + if part.get("type") in {"image", "image_url", "input_image"}: + cleaned.append({"type": part.get("type"), "image": "[stripped]"}) + else: + cleaned.append(part) + else: + cleaned.append(part) + shadow[k] = cleaned + elif isinstance(v, dict) and v.get("_multimodal"): + shadow[k] = v.get("text_summary", "") + else: + shadow[k] = v + else: + shadow[k] = v + return shadow + + def _estimate_message_chars(msg: Dict[str, Any]) -> int: """Char count for token estimation, excluding base64 image data. @@ -2973,58 +3096,14 @@ def _estimate_message_chars(msg: Dict[str, Any]) -> int: """ if not isinstance(msg, dict): return len(str(msg)) - shadow: Dict[str, Any] = {} - for k, v in msg.items(): - if k == "_anthropic_content_blocks": - continue - if k == "content": - if isinstance(v, list): - cleaned = [] - for part in v: - if isinstance(part, dict): - if part.get("type") in {"image", "image_url", "input_image"}: - cleaned.append({"type": part.get("type"), "image": "[stripped]"}) - else: - cleaned.append(part) - else: - cleaned.append(part) - shadow[k] = cleaned - elif isinstance(v, dict) and v.get("_multimodal"): - shadow[k] = v.get("text_summary", "") - else: - shadow[k] = v - else: - shadow[k] = v - return len(str(shadow)) + return len(str(_wire_message_shadow(msg))) def _estimate_message_tokens_without_images(msg: Dict[str, Any]) -> int: """Token estimate for a message shadow with image payloads stripped.""" if not isinstance(msg, dict): return estimate_tokens_rough(str(msg)) - shadow: Dict[str, Any] = {} - for k, v in msg.items(): - if k == "_anthropic_content_blocks": - continue - if k == "content": - if isinstance(v, list): - cleaned = [] - for part in v: - if isinstance(part, dict): - if part.get("type") in {"image", "image_url", "input_image"}: - cleaned.append({"type": part.get("type"), "image": "[stripped]"}) - else: - cleaned.append(part) - else: - cleaned.append(part) - shadow[k] = cleaned - elif isinstance(v, dict) and v.get("_multimodal"): - shadow[k] = v.get("text_summary", "") - else: - shadow[k] = v - else: - shadow[k] = v - return estimate_tokens_rough(str(shadow)) + return estimate_tokens_rough(str(_wire_message_shadow(msg))) def estimate_request_tokens_rough( diff --git a/agent/outbound_webhooks.py b/agent/outbound_webhooks.py new file mode 100644 index 0000000000..f437b809f3 --- /dev/null +++ b/agent/outbound_webhooks.py @@ -0,0 +1,569 @@ +""" +Outbound webhook notifications. + +Reads the ``hooks.outbound:`` list from ``config.yaml`` and registers +notify-only callbacks on the existing plugin hook manager, so every +``invoke_hook()`` site can push lifecycle events to external HTTP +endpoints — CI systems, dashboards, other agents — with zero changes to +call sites and zero polling on the receiving end. + +This is the outbound mirror of the inbound webhook platform +(``gateway/platforms/webhook.py``): inbound wakes Hermes when the world +changes; outbound tells the world when Hermes does something. + +Design notes +------------ +* Delivery is fire-and-forget through a bounded in-process queue and a + single daemon worker thread. ``invoke_hook()`` runs inside the agent + loop, so callbacks must never block on network I/O — they serialize, + enqueue, and return ``None`` immediately. Outbound targets can never + block a tool call, inject context, or otherwise influence agent flow. +* Payloads are signed with HMAC-SHA256 (GitHub-style + ``X-Hermes-Signature-256: sha256=`` over the raw body) when + a secret is configured. Receivers verify exactly like they verify + GitHub webhooks. +* No consent prompt: unlike shell hooks, an outbound target executes no + code on this machine — it POSTs JSON to a URL the user themselves put + in config. ``HERMES_SAFE_MODE=1`` still skips registration, matching + plugins / MCP / shell hooks. +* Registration is idempotent — safe to invoke from both the CLI entry + point and the gateway entry point. + +Config schema (``~/.hermes/config.yaml``):: + + hooks: + outbound: + - url: https://ci.example.com/hermes-events + events: [on_session_end, subagent_stop] + # secret literal (discouraged) or env var name (preferred): + secret_env: HERMES_OUTBOUND_WEBHOOK_SECRET + # optional regex, honored for pre/post_tool_call only: + matcher: "terminal|delegate_task" + timeout: 10 # per-attempt seconds, clamped to [1, 60] + name: ci-notify # optional label for logs / `hermes hooks list` + +Wire format (POST body):: + + { + "hook_event_name": "on_session_end", + "tool_name": null, + "tool_input": null, + "session_id": "sess_abc123", + "cwd": "/home/user/project", + "extra": {...}, # event-specific kwargs + "delivery_id": "3f2c...", # uuid4, unique per POST + "timestamp": "2026-07-22T14:00:00Z" + } + +Headers:: + + Content-Type: application/json + User-Agent: Hermes-Agent-Outbound-Webhook + X-Hermes-Event: + X-Hermes-Delivery: + X-Hermes-Signature-256: sha256= # only when secret set +""" + +from __future__ import annotations + +import atexit +import hashlib +import hmac +import json +import logging +import os +import queue +import re +import threading +import time +import uuid +from dataclasses import dataclass, field +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Dict, List, Optional, Set, Tuple +from urllib import error as urlerror +from urllib import request as urlrequest + +logger = logging.getLogger(__name__) + +DEFAULT_TIMEOUT_SECONDS = 10 +MAX_TIMEOUT_SECONDS = 60 +MAX_DELIVERY_ATTEMPTS = 2 +RETRY_BACKOFF_SECONDS = 1.0 +QUEUE_MAX_SIZE = 256 + +# Events whose ``matcher`` field is honored (mirrors shell hooks). +_TOOL_SCOPED_EVENTS = {"pre_tool_call", "post_tool_call"} + +# kwargs promoted to top-level payload keys (mirrors shell hooks wire). +_TOP_LEVEL_PAYLOAD_KEYS = {"tool_name", "args", "session_id", "parent_session_id"} + +# (event, url) pairs already wired to the plugin manager in this process. +_registered: Set[Tuple[str, str]] = set() +_registered_lock = threading.Lock() + +_delivery_queue: "queue.Queue[Optional[Dict[str, Any]]]" = queue.Queue( + maxsize=QUEUE_MAX_SIZE +) +_worker_lock = threading.Lock() +_worker: Optional[threading.Thread] = None + + +@dataclass +class WebhookTarget: + """Parsed and validated representation of one ``hooks.outbound`` entry.""" + + url: str + events: List[str] + name: str = "" + secret: Optional[str] = None + matcher: Optional[str] = None + timeout: int = DEFAULT_TIMEOUT_SECONDS + compiled_matcher: Optional[re.Pattern] = field(default=None, repr=False) + + def __post_init__(self) -> None: + if isinstance(self.matcher, str): + stripped = self.matcher.strip() + self.matcher = stripped if stripped else None + if self.matcher: + try: + self.compiled_matcher = re.compile(self.matcher) + except re.error as exc: + logger.warning( + "outbound webhook matcher %r is invalid (%s) — treating " + "as literal equality", self.matcher, exc, + ) + self.compiled_matcher = None + + @property + def label(self) -> str: + return self.name or self.url + + def matches_tool(self, tool_name: Optional[str]) -> bool: + if not self.matcher: + return True + if tool_name is None: + return False + if self.compiled_matcher is not None: + return self.compiled_matcher.fullmatch(tool_name) is not None + return tool_name == self.matcher + + +# --------------------------------------------------------------------------- +# Public API +# --------------------------------------------------------------------------- + +def register_from_config(cfg: Optional[Dict[str, Any]]) -> List[WebhookTarget]: + """Register every configured outbound webhook on the plugin manager. + + ``cfg`` is the full parsed config dict. Missing, empty, or malformed + ``hooks.outbound`` is treated as zero targets — config parsing never + raises, because a broken webhook entry must not crash the agent. + + Returns the targets that ended up wired (deduplicated across repeat + calls, so the CLI and gateway can both invoke this safely). + """ + if not isinstance(cfg, dict): + return [] + + from utils import env_var_enabled + + if env_var_enabled("HERMES_SAFE_MODE"): + logger.info("HERMES_SAFE_MODE=1 — outbound webhook registration skipped") + return [] + + hooks_cfg = cfg.get("hooks") + targets = _parse_outbound_block( + hooks_cfg.get("outbound") if isinstance(hooks_cfg, dict) else None + ) + if not targets: + return [] + + from hermes_cli.plugins import get_plugin_manager + + manager = get_plugin_manager() + + registered: List[WebhookTarget] = [] + with _registered_lock: + for target in targets: + wired_any = False + for event in target.events: + key = (event, target.url) + if key in _registered: + continue + manager._hooks.setdefault(event, []).append( + _make_callback(event, target) + ) + _registered.add(key) + wired_any = True + logger.info( + "outbound webhook registered: %s -> %s (matcher=%s, " + "timeout=%ds)", + event, target.label, target.matcher, target.timeout, + ) + if wired_any: + registered.append(target) + + return registered + + +def iter_configured_targets(cfg: Optional[Dict[str, Any]]) -> List[WebhookTarget]: + """Parse ``hooks.outbound`` without registering anything. + Used by ``hermes hooks list``.""" + if not isinstance(cfg, dict): + return [] + hooks_cfg = cfg.get("hooks") + return _parse_outbound_block( + hooks_cfg.get("outbound") if isinstance(hooks_cfg, dict) else None + ) + + +def flush(timeout: float = 5.0) -> bool: + """Block until all queued deliveries are done (or *timeout* elapses). + Returns ``True`` when the queue fully drained. Test/shutdown helper.""" + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + with _delivery_queue.all_tasks_done: + if _delivery_queue.unfinished_tasks == 0: + return True + time.sleep(0.02) + with _delivery_queue.all_tasks_done: + return _delivery_queue.unfinished_tasks == 0 + + +def reset_for_tests() -> None: + """Clear the idempotence set and drain the queue. Test-only helper.""" + with _registered_lock: + _registered.clear() + try: + while True: + _delivery_queue.get_nowait() + _delivery_queue.task_done() + except queue.Empty: + pass + + +# --------------------------------------------------------------------------- +# Config parsing +# --------------------------------------------------------------------------- + +def _parse_outbound_block(raw: Any) -> List[WebhookTarget]: + if raw is None: + return [] + if not isinstance(raw, list): + logger.warning( + "hooks.outbound must be a list of webhook targets; got %s", + type(raw).__name__, + ) + return [] + + targets: List[WebhookTarget] = [] + for i, entry in enumerate(raw): + target = _parse_single_target(i, entry) + if target is not None: + targets.append(target) + return targets + + +def _parse_single_target(index: int, raw: Any) -> Optional[WebhookTarget]: + from hermes_cli.plugins import VALID_HOOKS + + if not isinstance(raw, dict): + logger.warning( + "hooks.outbound[%d] must be a mapping with 'url' and 'events' " + "keys; got %s", index, type(raw).__name__, + ) + return None + + url = raw.get("url") + if not isinstance(url, str) or not url.strip(): + logger.warning("hooks.outbound[%d] is missing a non-empty 'url'", index) + return None + url = url.strip() + if not url.lower().startswith(("http://", "https://")): + logger.warning( + "hooks.outbound[%d].url must be http(s); got %r — skipped", + index, url, + ) + return None + if url.lower().startswith("http://"): + logger.warning( + "hooks.outbound[%d].url uses plain http:// — payloads (including " + "tool inputs) travel unencrypted. Prefer https.", index, + ) + + events_raw = raw.get("events") + if not isinstance(events_raw, list) or not events_raw: + logger.warning( + "hooks.outbound[%d] needs a non-empty 'events' list (valid: %s)", + index, ", ".join(sorted(VALID_HOOKS)), + ) + return None + events: List[str] = [] + for ev in events_raw: + if ev in VALID_HOOKS: + events.append(ev) + else: + logger.warning( + "hooks.outbound[%d]: unknown event %r ignored (valid: %s)", + index, ev, ", ".join(sorted(VALID_HOOKS)), + ) + if not events: + logger.warning( + "hooks.outbound[%d] has no valid events — skipped", index, + ) + return None + + matcher = raw.get("matcher") + if matcher is not None and not isinstance(matcher, str): + logger.warning( + "hooks.outbound[%d].matcher must be a string regex; ignoring", + index, + ) + matcher = None + if matcher is not None and not any(e in _TOOL_SCOPED_EVENTS for e in events): + logger.warning( + "hooks.outbound[%d].matcher=%r will be ignored — matcher is only " + "honored for pre_tool_call / post_tool_call.", index, matcher, + ) + matcher = None + + timeout_raw = raw.get("timeout", DEFAULT_TIMEOUT_SECONDS) + try: + timeout = int(timeout_raw) + except (TypeError, ValueError): + logger.warning( + "hooks.outbound[%d].timeout must be an int (got %r); using " + "default %ds", index, timeout_raw, DEFAULT_TIMEOUT_SECONDS, + ) + timeout = DEFAULT_TIMEOUT_SECONDS + timeout = max(1, min(timeout, MAX_TIMEOUT_SECONDS)) + + secret = _resolve_secret(index, raw) + + name = raw.get("name") + if not isinstance(name, str): + name = "" + + return WebhookTarget( + url=url, + events=events, + name=name.strip(), + secret=secret, + matcher=matcher, + timeout=timeout, + ) + + +def _resolve_secret(index: int, raw: Dict[str, Any]) -> Optional[str]: + """``secret_env`` (env var name, preferred) wins over inline ``secret``.""" + secret_env = raw.get("secret_env") + if isinstance(secret_env, str) and secret_env.strip(): + value = os.environ.get(secret_env.strip(), "") + if value: + return value + logger.warning( + "hooks.outbound[%d].secret_env=%r is not set in the environment " + "— deliveries will be UNSIGNED", index, secret_env.strip(), + ) + return None + secret = raw.get("secret") + if isinstance(secret, str) and secret: + return secret + return None + + +# --------------------------------------------------------------------------- +# Callback + delivery +# --------------------------------------------------------------------------- + +def _make_callback(event: str, target: WebhookTarget): + """Build the notify-only closure ``invoke_hook()`` calls per firing.""" + + def _callback(**kwargs: Any) -> None: + if event in _TOOL_SCOPED_EVENTS: + if not target.matches_tool(kwargs.get("tool_name")): + return None + delivery_id = uuid.uuid4().hex + try: + body = _serialize_payload(event, kwargs, delivery_id) + except Exception: # defensive — a bad payload must not hurt the loop + logger.warning( + "outbound webhook payload serialization failed (event=%s " + "target=%s)", event, target.label, exc_info=True, + ) + return None + _enqueue(_build_delivery(event, target, body, delivery_id)) + return None + + _callback.__name__ = f"outbound_webhook[{event}:{target.label}]" + _callback.__qualname__ = _callback.__name__ + return _callback + + +def _serialize_payload( + event: str, kwargs: Dict[str, Any], delivery_id: str, +) -> bytes: + """Render the POST body. Same top-level shape as shell hooks' stdin + (documented in :mod:`agent.shell_hooks`), plus delivery metadata. + + ``delivery_id`` is shared with the ``X-Hermes-Delivery`` header so + receivers can dedupe on either — and since it (plus ``timestamp``) + lives inside the HMAC-signed body, it doubles as replay protection. + """ + extras = {k: v for k, v in kwargs.items() if k not in _TOP_LEVEL_PAYLOAD_KEYS} + try: + cwd = str(Path.cwd()) + except OSError: + cwd = "" + payload = { + "hook_event_name": event, + "tool_name": kwargs.get("tool_name"), + "tool_input": kwargs.get("args") if isinstance(kwargs.get("args"), dict) else None, + "session_id": kwargs.get("session_id") or kwargs.get("parent_session_id") or "", + "cwd": cwd, + "extra": extras, + "delivery_id": delivery_id, + "timestamp": datetime.now(tz=timezone.utc) + .isoformat() + .replace("+00:00", "Z"), + } + return json.dumps(payload, ensure_ascii=False, default=str).encode("utf-8") + + +def _build_delivery( + event: str, target: WebhookTarget, body: bytes, delivery_id: str, +) -> Dict[str, Any]: + headers = { + "Content-Type": "application/json", + "User-Agent": "Hermes-Agent-Outbound-Webhook", + "X-Hermes-Event": event, + "X-Hermes-Delivery": delivery_id, + } + if target.secret: + digest = hmac.new( + target.secret.encode("utf-8"), body, hashlib.sha256 + ).hexdigest() + headers["X-Hermes-Signature-256"] = f"sha256={digest}" + return { + "url": target.url, + "label": target.label, + "event": event, + "body": body, + "headers": headers, + "timeout": target.timeout, + } + + +def _enqueue(delivery: Dict[str, Any]) -> None: + _ensure_worker() + try: + _delivery_queue.put_nowait(delivery) + except queue.Full: + logger.warning( + "outbound webhook queue full (%d pending) — dropping %s event " + "for %s", QUEUE_MAX_SIZE, delivery["event"], delivery["label"], + ) + + +def _ensure_worker() -> None: + global _worker + if _worker is not None and _worker.is_alive(): + return + with _worker_lock: + if _worker is not None and _worker.is_alive(): + return + _worker = threading.Thread( + target=_worker_loop, name="outbound-webhooks", daemon=True, + ) + _worker.start() + # The worker is a daemon thread, so a short-lived process (a `-q` + # CLI run, a cron session) can exit right after enqueuing the + # final events — silently dropping on_session_end, the headline + # use case. Drain the queue at interpreter shutdown, bounded so + # a dead endpoint can only delay exit, never hang it. + atexit.register(flush, timeout=5.0) + + +def _worker_loop() -> None: + while True: + delivery = _delivery_queue.get() + try: + if delivery is not None: + _deliver(delivery) + except Exception: # pragma: no cover — defensive + logger.warning( + "outbound webhook delivery crashed (target=%s)", + delivery.get("label") if isinstance(delivery, dict) else "?", + exc_info=True, + ) + finally: + _delivery_queue.task_done() + + +class _NoRedirectHandler(urlrequest.HTTPRedirectHandler): + """Refuse to follow redirects. + + urllib's default handler converts a redirected POST into a body-less + GET — the signed payload would be silently dropped and the headers + re-sent to a location the user never configured. Treat any 3xx as a + delivery failure instead (surfaced as HTTPError by returning None). + """ + + def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D102 + return None + + +_opener = urlrequest.build_opener(_NoRedirectHandler) + + +def _deliver(delivery: Dict[str, Any]) -> None: + """POST with bounded retries. Retries on connection errors and 5xx; + 4xx is the receiver telling us the request itself is wrong — no retry. + 3xx redirects are never followed (misconfiguration — fix the URL).""" + last_error = "" + for attempt in range(1, MAX_DELIVERY_ATTEMPTS + 1): + req = urlrequest.Request( + delivery["url"], + data=delivery["body"], + headers=delivery["headers"], + method="POST", + ) + try: + with _opener.open(req, timeout=delivery["timeout"]) as resp: + status = getattr(resp, "status", 200) + if 200 <= status < 300: + logger.debug( + "outbound webhook delivered: %s -> %s (HTTP %d)", + delivery["event"], delivery["label"], status, + ) + return + last_error = f"HTTP {status}" + except urlerror.HTTPError as exc: + last_error = f"HTTP {exc.code}" + if 300 <= exc.code < 400: + logger.warning( + "outbound webhook target redirected (event=%s target=%s): " + "%s -> %s — redirects are not followed; update the " + "configured url", delivery["event"], delivery["label"], + last_error, exc.headers.get("Location", "?"), + ) + return + if 400 <= exc.code < 500: + logger.warning( + "outbound webhook rejected (event=%s target=%s): %s — " + "not retrying", delivery["event"], delivery["label"], + last_error, + ) + return + except Exception as exc: + last_error = str(exc) or type(exc).__name__ + + if attempt < MAX_DELIVERY_ATTEMPTS: + time.sleep(RETRY_BACKOFF_SECONDS * attempt) + + logger.warning( + "outbound webhook delivery failed after %d attempt(s) (event=%s " + "target=%s): %s", + MAX_DELIVERY_ATTEMPTS, delivery["event"], delivery["label"], last_error, + ) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 6bd88fa194..0f748f8b2b 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -1083,6 +1083,7 @@ def _probe_remote_backend(env_type: str) -> str | None: "docker_env": config.get("docker_env", {}), "docker_run_as_host_user": config.get("docker_run_as_host_user", False), "docker_extra_args": config.get("docker_extra_args", []), + "docker_shm_size": config.get("docker_shm_size", "1g"), "docker_persist_across_processes": config.get("docker_persist_across_processes", True), "docker_orphan_reaper": config.get("docker_orphan_reaper", True), } diff --git a/agent/prompt_caching.py b/agent/prompt_caching.py index 1a0326c79d..a012f143f2 100644 --- a/agent/prompt_caching.py +++ b/agent/prompt_caching.py @@ -11,9 +11,27 @@ Pure functions -- no class state, no AIAgent dependency. """ import copy +from dataclasses import dataclass from typing import Any, Dict, List +@dataclass(frozen=True) +class PromptCachePlan: + """Request-local message and tool sections with their cache markers.""" + + messages: List[Dict[str, Any]] + tools: List[Dict[str, Any]] + + @property + def marker_count(self) -> int: + """Wire-visible cache markers in this plan (computed on demand). + + Only tests consume this; keeping it lazy avoids walking every + message part and tool schema on the per-request hot path. + """ + return _count_cache_markers(self.messages, self.tools) + + def _apply_cache_marker(msg: dict, cache_marker: dict, native_anthropic: bool = False) -> None: """Add cache_control to a single message, handling all format variations.""" role = msg.get("role", "") @@ -89,12 +107,28 @@ def _apply_system_cache_markers( static_system_prefix: str | None, *, native_anthropic: bool, + mark_suffix: bool = True, + fallback_to_whole: bool = True, ) -> int: - """Mark the static system prefix and full prompt when they can be split. + """Mark the static system prefix (and optionally the full prompt). The system prompt remains one stored string. Splitting it only in the outgoing request keeps session persistence and non-Anthropic transports unchanged while making the stable prefix independently cacheable. + + ``mark_suffix=False`` is the tool-cache-plan layout: only the static + prefix carries a marker, the volatile suffix rides unmarked (its + breakpoint budget is spent on the tools array instead). + + ``fallback_to_whole=False`` skips marking entirely when the prefix + split is not possible (no prefix, mismatched prefix, non-string + content) instead of marking the whole message. + + When the prompt IS exactly the static prefix (empty suffix), the whole + message is marked as a single block — never a two-part split with an + empty text block, which Anthropic rejects. + + Returns the number of markers applied (0, 1, or 2). """ content = message.get("content") if ( @@ -105,16 +139,26 @@ def _apply_system_cache_markers( ): suffix = content[len(static_system_prefix):] if suffix: + suffix_part: dict = {"type": "text", "text": suffix} + if mark_suffix: + suffix_part["cache_control"] = cache_marker message["content"] = [ { "type": "text", "text": static_system_prefix, "cache_control": cache_marker, }, - {"type": "text", "text": suffix, "cache_control": cache_marker}, + suffix_part, ] - return 2 + return 2 if mark_suffix else 1 + # Empty suffix: the stored prompt IS the static prefix. Mark it as + # one whole block — a [marked-prefix, ""] split would put an empty + # text block on the wire (HTTP 400 on native Anthropic). + _apply_cache_marker(message, cache_marker, native_anthropic=native_anthropic) + return 1 + if not fallback_to_whole: + return 0 _apply_cache_marker(message, cache_marker, native_anthropic=native_anthropic) return 1 @@ -172,6 +216,135 @@ def strip_anthropic_cache_control( return api_messages +def strip_anthropic_tool_cache_control(tools: List[Dict[str, Any]] | None) -> List[Dict[str, Any]]: + """Return copied tools without request-local Anthropic cache markers.""" + cleaned = copy.deepcopy(tools or []) + for tool in cleaned: + if isinstance(tool, dict): + tool.pop("cache_control", None) + return cleaned + + +def _count_cache_markers(messages: List[Dict[str, Any]], tools: List[Dict[str, Any]]) -> int: + """Count the wire-visible cache markers in a request-local plan.""" + count = sum( + 1 + for message in messages + if isinstance(message, dict) and "cache_control" in message + ) + count += sum( + 1 + for message in messages + if isinstance(message, dict) and isinstance(message.get("content"), list) + for part in message["content"] + if isinstance(part, dict) and "cache_control" in part + ) + return count + sum( + 1 for tool in tools if isinstance(tool, dict) and "cache_control" in tool + ) + + +def _completed_transaction_endpoint_indexes( + messages: List[Dict[str, Any]], *, native_anthropic: bool, +) -> List[int]: + """Select legal ends of completed tool runs and ordinary turns.""" + endpoints: List[int] = [] + index = 0 + while index < len(messages): + message = messages[index] + if not isinstance(message, dict) or message.get("role") == "system": + index += 1 + continue + + if message.get("role") == "assistant" and message.get("tool_calls"): + result_start = index + 1 + result_end = result_start + while result_end < len(messages): + result = messages[result_end] + if not isinstance(result, dict) or result.get("role") != "tool": + break + result_end += 1 + if result_end > result_start: + endpoint = result_end - 1 + if _can_carry_marker(messages[endpoint], native_anthropic): + endpoints.append(endpoint) + index = result_end + continue + + if message.get("role") == "tool": + while index < len(messages): + result = messages[index] + if not isinstance(result, dict) or result.get("role") != "tool": + break + index += 1 + continue + + if message.get("role") == "user" and index + 1 < len(messages): + index += 1 + continue + + if ( + message.get("role") == "assistant" + and message.get("content") in (None, "") + ): + index += 1 + continue + + if _can_carry_marker(message, native_anthropic): + endpoints.append(index) + index += 1 + return endpoints + + +def build_prompt_cache_plan( + api_messages: List[Dict[str, Any]], + tools: List[Dict[str, Any]] | None, + *, + cache_ttl: str = "5m", + native_anthropic: bool = False, + static_system_prefix: str | None = None, + direct_native_tool_cache: bool = False, +) -> PromptCachePlan: + """Build isolated cache sections for one resolved request destination.""" + messages = copy.deepcopy(api_messages or []) + strip_anthropic_cache_control(messages) + planned_tools = strip_anthropic_tool_cache_control(tools) + + if not direct_native_tool_cache or not planned_tools: + planned_messages = apply_anthropic_cache_control( + messages, + cache_ttl=cache_ttl, + native_anthropic=native_anthropic, + static_system_prefix=static_system_prefix, + ) + return PromptCachePlan(messages=planned_messages, tools=planned_tools) + + marker = _build_marker(cache_ttl) + if ( + messages + and isinstance(messages[0], dict) + and messages[0].get("role") == "system" + ): + # Tool-cache layout: only the static prefix carries a system-side + # marker; the volatile suffix's budget is spent on the tools array. + _apply_system_cache_markers( + messages[0], + marker, + static_system_prefix, + native_anthropic=True, + mark_suffix=False, + fallback_to_whole=False, + ) + planned_tools[-1]["cache_control"] = dict(marker) + for endpoint in _completed_transaction_endpoint_indexes( + messages, + native_anthropic=True, + )[-2:]: + _apply_cache_marker(messages[endpoint], marker, native_anthropic=True) + + return PromptCachePlan(messages=messages, tools=planned_tools) + + def apply_anthropic_cache_control( api_messages: List[Dict[str, Any]], cache_ttl: str = "5m", diff --git a/agent/redact.py b/agent/redact.py index b104ad4c22..ea70246a90 100644 --- a/agent/redact.py +++ b/agent/redact.py @@ -111,6 +111,23 @@ _PREFIX_PATTERNS = [ r"fw-[A-Za-z0-9]{30,}", # Fireworks AI API key r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key + # GitLab token families (each pattern keeps a full literal prefix so the + # _PREFIX_SUBSTRINGS pre-screen stays false-negative-free). Ported from + # openclaw/openclaw#112954; follow-up invited in #4541. + r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token + r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret + r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token + r"glrt-[A-Za-z0-9_.\-]{10,}", # GitLab runner authentication token (routable tokens are dotted) + r"glrtr-[A-Za-z0-9_.\-]{10,}", # GitLab runner registration token (routable) + r"glcbt-[A-Za-z0-9_\-]{10,}", # GitLab CI/CD job token + r"glptt-[A-Za-z0-9_\-]{10,}", # GitLab pipeline trigger token + r"glft-[A-Za-z0-9_\-]{10,}", # GitLab feed token + r"glimt-[A-Za-z0-9_\-]{10,}", # GitLab incoming mail token + r"glagent-[A-Za-z0-9_\-]{10,}", # GitLab agent (KAS) token + r"glsoat-[A-Za-z0-9_\-]{10,}", # GitLab service-account access token + r"glffct-[A-Za-z0-9_\-]{10,}", # GitLab feature-flags client token + r"glwt-[A-Za-z0-9_\-]{10,}", # GitLab workspace token + r"GR1348941[A-Za-z0-9_\-]{10,}", # GitLab legacy runner registration token ] # ENV assignment patterns: KEY=value where KEY contains a secret-like name. @@ -154,9 +171,13 @@ _ENV_LOOKUP_VALUE_RE = re.compile( r"^(?:os\.(?:getenv|environ)|process\.env|\$ENV\{)" ) # Namespaced (dotted) key: the secret word may sit anywhere in a dotted path. +# NOTE(perf): possessive quantifiers (py3.11+) replace the nested quantifier +# ``(?:[A-Za-z0-9_\-]+\.)+`` (exponential backtracking on long dotted runs). +# The ``*`` runs bordering {_SECRET_CFG_NAMES} must stay backtrackable +# (secret words are matchable by the class, e.g. ``app.api.key=…``). _CFG_DOTTED_RE = re.compile( - rf"((?:[A-Za-z0-9_\-]+\.)+[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*" - rf"|[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*\.[A-Za-z0-9_.\-]+)" + rf"([A-Za-z0-9_\-]++\.[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*+" + rf"|[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*\.[A-Za-z0-9_.\-]++)" rf"={_CFG_VALUE}", re.IGNORECASE, ) @@ -175,8 +196,10 @@ _CFG_ANCHORED_RE = re.compile( # is masked by _AUTH_HEADER_RE); ``auth_token``/``auth-token`` still match via # the ``token`` keyword. Quoted values defer to _JSON_FIELD_RE via the lookahead. _YAML_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential)" +# NOTE(perf): possessive quantifiers wherever the successor is disjoint; the +# leading ``[A-Za-z0-9_.\-]*`` stays backtrackable (see _CFG_DOTTED_RE note). _YAML_ASSIGN_RE = re.compile( - rf"(^[ \t]*[A-Za-z0-9_.\-]*{_YAML_CFG_NAMES}[A-Za-z0-9_.\-]*)(:[ \t]*)(?!['\"])([^\s&]+)", + rf"(^[ \t]*+[A-Za-z0-9_.\-]*{_YAML_CFG_NAMES}[A-Za-z0-9_.\-]*+)(:[ \t]*+)(?!['\"])([^\s&]++)", re.IGNORECASE | re.MULTILINE, ) diff --git a/agent/relay_llm.py b/agent/relay_llm.py index 535925f9a6..984fcae3eb 100644 --- a/agent/relay_llm.py +++ b/agent/relay_llm.py @@ -247,6 +247,14 @@ async def execute_current_async( ) +def _has_running_event_loop() -> bool: + try: + asyncio.get_running_loop() + except RuntimeError: + return False + return True + + def stream_current( request: dict[str, Any], stream_factory: Callable[[dict[str, Any]], Any], @@ -256,12 +264,34 @@ def stream_current( finalizer: Callable[[], Any], metadata: dict[str, Any] | None = None, defer_logical_completion: bool = False, + completed_response_predicate: Callable[[Any], bool] | None = None, ) -> Any: - """Run a provider stream under the inherited Hermes turn when present.""" + """Run a provider stream under the inherited Hermes turn when present. + + When ``completed_response_predicate`` is set and the stream_factory returns + a complete response instead of an iterator (e.g. AnthropicAuxiliaryClient + and other shims that ignore ``stream=True``), unwrap and return the + completed response directly. This mirrors the pre-Relay behavior where + ``call_llm(stream=True)`` returned the raw response and the consumer's + own ``hasattr(stream, "choices")`` check handled it (#11732, #55933) — + without the unwrap the response stays trapped as ``final_response`` on the + inner ManagedLlmStream and the outer consumer sees an empty stream. + """ turn = relay_runtime.active_turn() if turn is None: return stream_factory(request) - return stream( + if _has_running_event_loop(): + # Managed provider callbacks execute on the Relay session's event + # loop. A nested ManagedLlmStream built here would be synchronously + # iterated on that same loop thread, which asyncio forbids + # ("Cannot run the event loop while another loop is running"). + # Return the raw factory result instead: the outer managed stream + # already provides Relay tracking for the enclosing attempt, and its + # own completed_response_predicate traps a completed response (e.g. + # the MoA facade's auxiliary ``call_llm(stream=True)`` returning a + # full response when an adapter ignores ``stream=True``). + return stream_factory(request) + managed = stream( request, stream_factory, session_id=turn.lease.session_id, @@ -270,7 +300,17 @@ def stream_current( finalizer=finalizer, metadata=metadata, defer_logical_completion=defer_logical_completion, + completed_response_predicate=completed_response_predicate, ) + # In the non-managed path the factory already ran eagerly during __init__, + # so a completed response is visible immediately and must surface raw. + # In the managed path the factory runs lazily on first pull, so + # final_response is still None here and the managed stream is returned. + if completed_response_predicate is not None: + completed = getattr(managed, "final_response", None) + if completed is not None: + return completed + return managed def stream( diff --git a/agent/secret_scope.py b/agent/secret_scope.py index 8b376d5fce..919fe3e27b 100644 --- a/agent/secret_scope.py +++ b/agent/secret_scope.py @@ -23,6 +23,7 @@ Design rationale lives in ``docs/design/multiplexing-gateway.md`` (Workstream A) from __future__ import annotations import os +import re from contextvars import ContextVar, Token from pathlib import Path from typing import Dict, Mapping, Optional @@ -105,6 +106,14 @@ _GLOBAL_ENV_EXACT = frozenset({ "VIRTUAL_ENV", "PYTHONPATH", "SSL_CERT_FILE", # Kanban paths (per-board, not per-profile-secret) "HERMES_KANBAN_DB", "HERMES_KANBAN_WORKSPACES_ROOT", "HERMES_KANBAN_BOARD", + # API-server LISTENER settings — deployment config (Docker compose + # ``environment:`` block, systemd ``Environment=``), not profile secrets. + # The scoped runner reload (#64674) must keep seeing them or container + # deployments silently lose the api_server platform (#69379). NOTE: + # API_SERVER_KEY is deliberately NOT here — it IS a credential and stays + # profile-scoped. + "API_SERVER_ENABLED", "API_SERVER_HOST", "API_SERVER_PORT", + "API_SERVER_CORS_ORIGINS", }) _GLOBAL_ENV_PREFIXES = ( "HERMES_KANBAN_", @@ -177,20 +186,72 @@ def get_secret(name: str, default: Optional[str] = None) -> Optional[str]: return val if val is not None else default +def _strip_inline_comment(value: str) -> str: + """Strip a dotenv-style inline comment from a raw ``.env`` value. + + Mirrors python-dotenv (1.2.2) semantics, verified empirically: + + - Quoted values: scan for the matching close quote + (backslash-escape-aware for double quotes, since ``save_env_value`` + writes ``\\"``/``\\\\`` escapes). Everything through the close quote is + kept; a trailing ``# ...`` remainder after it is discarded, so + ``KEY="has # inside" # trailing`` yields ``has # inside``. Non-comment + trailing junk leaves the value untouched (lenient, unlike dotenv's + hard parse error). + - Unquoted values: truncate only at a ``#`` PRECEDED BY WHITESPACE, so + ``KEY=foo#bar`` keeps ``foo#bar`` while ``KEY=value # comment`` keeps + ``value``. A value that *starts* with ``#`` (``KEY=#leading``) is kept. + """ + value = value.strip() + if not value: + return value + quote = value[0] + if quote in ("'", '"'): + i = 1 + while i < len(value): + ch = value[i] + if quote == '"' and ch == "\\": + i += 2 # skip the escaped character + continue + if ch == quote: + remainder = value[i + 1:].lstrip() + if remainder.startswith("#"): + return value[: i + 1] + return value + i += 1 + return value # unterminated quote: leave as-is + return re.split(r"\s+#", value, maxsplit=1)[0].strip() + + def load_env_file(env_path: Path) -> Dict[str, str]: """Parse a ``.env`` file into a plain dict WITHOUT touching ``os.environ``. Used to load a profile's secrets into an isolated mapping for - ``set_secret_scope``. Mirrors python-dotenv's basic parsing (KEY=VALUE, - ``export`` prefix, ``#`` comments, optional matching quotes) but never - mutates the process environment — that isolation is the whole point. + ``set_secret_scope``. Parses the small KEY=VALUE subset Hermes writes + itself (``export`` prefix, ``#`` comments — full-line and + dotenv-compatible inline, matching quotes with the + writer's ``\\"``/``\\\\`` escapes reversed — the same semantics as + ``hermes_cli.config._parse_env_value``) but never mutates the process + environment — that isolation is the whole point. + + Encoding is ``utf-8-sig`` so a leading UTF-8 BOM (Windows Notepad / + PowerShell ``Set-Content -Encoding UTF8``) does not prefix the first + key as ``\\ufeffNAME`` and make ``get_secret('NAME')`` miss under scope. """ secrets: Dict[str, str] = {} try: - text = env_path.read_text(encoding="utf-8") + text = env_path.read_text(encoding="utf-8-sig") except (FileNotFoundError, OSError, UnicodeDecodeError): return secrets + # Parse values with the canonical Hermes parser: save_env_value + # escapes " and \ inside double quotes, and every other reader + # (load_env, python-dotenv) reverses those escapes. Stripping only + # the outer quotes here would corrupt credentials containing " + # or \ — they work interactively but fail in scoped (cron / + # multiplex) resolution. + from hermes_cli.config import _parse_env_value + for raw in text.splitlines(): line = raw.strip() if not line or line.startswith("#"): @@ -203,10 +264,7 @@ def load_env_file(env_path: Path) -> Dict[str, str]: key = key.strip() if not key: continue - value = value.strip() - if len(value) >= 2 and value[0] == value[-1] and value[0] in ("'", '"'): - value = value[1:-1] - secrets[key] = value + secrets[key] = _parse_env_value(_strip_inline_comment(value)) return secrets diff --git a/agent/secret_sources/base.py b/agent/secret_sources/base.py index 886a02950c..a7bc345baa 100644 --- a/agent/secret_sources/base.py +++ b/agent/secret_sources/base.py @@ -39,17 +39,36 @@ from __future__ import annotations import os import re import subprocess +from contextvars import ContextVar, Token from abc import ABC, abstractmethod from dataclasses import dataclass, field from enum import Enum from pathlib import Path -from typing import Dict, FrozenSet, List, Optional, Sequence +from typing import Dict, FrozenSet, List, MutableMapping, Optional, Sequence # Bump ONLY for breaking changes to the required contract surface # (abstract-method signatures, FetchResult required fields). Additive # optional hooks must ship with defaults and must NOT bump this. SECRET_SOURCE_API_VERSION = 1 +_SOURCE_ENVIRONMENT: ContextVar[Optional[MutableMapping[str, str]]] +_SOURCE_ENVIRONMENT = ContextVar("hermes_secret_source_environment", default=None) + + +def set_source_environment(environ: MutableMapping[str, str]) -> Token: + """Install a per-fetch environment view without changing ``os.environ``.""" + return _SOURCE_ENVIRONMENT.set(environ) + + +def reset_source_environment(token: Token) -> None: + _SOURCE_ENVIRONMENT.reset(token) + + +def get_source_environment() -> MutableMapping[str, str]: + """Return the active per-fetch environment, or the process environment.""" + environ = _SOURCE_ENVIRONMENT.get() + return environ if environ is not None else os.environ + # Timeout the orchestrator enforces around fetch() when the source's # config section doesn't override it. Generous because a first run may # include a one-time CLI binary auto-install (e.g. bws download+verify). diff --git a/agent/secret_sources/bitwarden.py b/agent/secret_sources/bitwarden.py index 7e71f0f7fd..357f69cc6f 100644 --- a/agent/secret_sources/bitwarden.py +++ b/agent/secret_sources/bitwarden.py @@ -58,6 +58,7 @@ from agent.secret_sources._cache import ( is_valid_env_name as _is_valid_env_name, ) from agent.secret_sources.base import ErrorKind, SecretSource +from agent.secret_sources.base import get_source_environment logger = logging.getLogger(__name__) @@ -667,10 +668,15 @@ def _run_bws_list( bws: Path, access_token: str, project_id: str, server_url: str = "" ) -> Tuple[Dict[str, str], List[str]]: cmd = [str(bws), "secret", "list", project_id, "--output", "json"] - # bws child intentionally receives the access token; exact preservation - # (BWS_SERVER_URL manual overrides etc. must survive untouched). - from tools.environments.local import build_subprocess_env - env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False) + # bws child intentionally receives the access token. Under a profile-local + # fetch it must not inherit sibling credentials from process-global env. + source_env = get_source_environment() + if source_env is os.environ: + from tools.environments.local import build_subprocess_env + + env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False) + else: + env = dict(source_env) env["BWS_ACCESS_TOKEN"] = access_token # Make sure we're not echoing telemetry / colour codes into json. env.setdefault("NO_COLOR", "1") @@ -908,7 +914,7 @@ class BitwardenSource(SecretSource): result = FetchResult() access_token_env = str(cfg.get("access_token_env") or "BWS_ACCESS_TOKEN") - access_token = os.environ.get(access_token_env, "").strip() + access_token = get_source_environment().get(access_token_env, "").strip() if not access_token: result.error = ( f"secrets.bitwarden.enabled is true but {access_token_env} is " diff --git a/agent/secret_sources/command.py b/agent/secret_sources/command.py index 9bce311ba9..0831648418 100644 --- a/agent/secret_sources/command.py +++ b/agent/secret_sources/command.py @@ -44,6 +44,7 @@ from typing import Dict, Optional # Reuse the exact result shape the bitwarden source returns so # hermes_cli.env_loader can consume both providers identically. from agent.secret_sources.base import ErrorKind, SecretSource +from agent.secret_sources.base import get_source_environment from agent.secret_sources.bitwarden import FetchResult __all__ = [ @@ -181,8 +182,17 @@ def _run_helper( # User-configured secret-helper command: runs with the user's full shell # env by design (it may need any credential to resolve the secret). - from tools.environments.local import build_subprocess_env - env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False) + source_env = get_source_environment() + if source_env is os.environ: + # Legacy single-profile startup intentionally preserves the existing + # helper contract, which may rely on the user's full environment. + from tools.environments.local import build_subprocess_env + env = build_subprocess_env(scrub_secrets=False, inherit_profile_home=False) + else: + # A multiplex profile must never inherit sibling secrets from the + # process-global environment. hydrate_profile_secret_sources seeds + # only global-safe values plus this profile's own .env. + env = dict(source_env) env["HERMES_SECRET_KEY"] = secret_key try: diff --git a/agent/secret_sources/onepassword.py b/agent/secret_sources/onepassword.py index d8216dfcbd..3aef02df5b 100644 --- a/agent/secret_sources/onepassword.py +++ b/agent/secret_sources/onepassword.py @@ -55,6 +55,7 @@ from agent.secret_sources._cache import ( is_valid_env_name, ) from agent.secret_sources.base import ErrorKind, SecretSource +from agent.secret_sources.base import get_source_environment logger = logging.getLogger(__name__) @@ -182,15 +183,16 @@ def _auth_fingerprint(token_env: str) -> str: previous identity is never served under a new one. Never logged or displayed; the raw token never leaves this hash. """ + source_env = get_source_environment() parts: List[str] = [ - f"token={os.environ.get(token_env, '')}", - f"account={os.environ.get('OP_ACCOUNT', '')}", - f"connect_host={os.environ.get('OP_CONNECT_HOST', '')}", - f"connect_token={os.environ.get('OP_CONNECT_TOKEN', '')}", + f"token={source_env.get(token_env, '')}", + f"account={source_env.get('OP_ACCOUNT', '')}", + f"connect_host={source_env.get('OP_CONNECT_HOST', '')}", + f"connect_token={source_env.get('OP_CONNECT_TOKEN', '')}", ] - for key in sorted(os.environ): + for key in sorted(source_env): if key.startswith("OP_SESSION_"): - parts.append(f"{key}={os.environ[key]}") + parts.append(f"{key}={source_env[key]}") material = "\n".join(parts) return hashlib.sha256(material.encode("utf-8")).hexdigest()[:16] @@ -238,13 +240,14 @@ def _scrub(text: str) -> str: def _op_child_env(token_value: str) -> Dict[str, str]: """Build a minimal allowlisted environment for the ``op`` child process.""" + source_env = get_source_environment() env: Dict[str, str] = {} for key in _OP_ENV_ALLOWLIST: - val = os.environ.get(key) + val = source_env.get(key) if val is not None: env[key] = val # Desktop / interactive session credentials. - for key, val in os.environ.items(): + for key, val in source_env.items(): if key.startswith("OP_SESSION_"): env[key] = val # `op` reads OP_SERVICE_ACCOUNT_TOKEN regardless of which env var the user @@ -340,7 +343,7 @@ def fetch_onepassword_secrets( if not valid: return {}, warnings - token_value = os.environ.get(token_env, "").strip() + token_value = get_source_environment().get(token_env, "").strip() cache_key: _CacheKey = ( _auth_fingerprint(token_env), account or "", diff --git a/agent/secret_sources/registry.py b/agent/secret_sources/registry.py index f81b8a35dd..216db858ba 100644 --- a/agent/secret_sources/registry.py +++ b/agent/secret_sources/registry.py @@ -32,7 +32,7 @@ import logging import os from dataclasses import dataclass, field from pathlib import Path -from typing import Dict, List, Optional +from typing import Dict, List, MutableMapping, Optional from agent.secret_sources.base import ( SECRET_SOURCE_API_VERSION, @@ -40,6 +40,8 @@ from agent.secret_sources.base import ( FetchResult, SecretSource, is_valid_env_name, + reset_source_environment, + set_source_environment, ) logger = logging.getLogger(__name__) @@ -196,7 +198,8 @@ def _reset_registry_for_tests() -> None: def _fetch_with_timeout( - source: SecretSource, cfg: dict, home_path: Path + source: SecretSource, cfg: dict, home_path: Path, + environ: MutableMapping[str, str], ) -> FetchResult: """Run source.fetch() under a wall-clock budget; never raises. @@ -211,7 +214,14 @@ def _fetch_with_timeout( max_workers=1, thread_name_prefix=f"secret-src-{source.name}" ) try: - future = executor.submit(source.fetch, cfg, home_path) + def _fetch() -> FetchResult: + token = set_source_environment(environ) + try: + return source.fetch(cfg, home_path) + finally: + reset_source_environment(token) + + future = executor.submit(_fetch) try: result = future.result(timeout=timeout) except concurrent.futures.TimeoutError: @@ -321,7 +331,7 @@ def _profile_alias_target(var: str, profile: str) -> Optional[str]: def apply_all(secrets_cfg: dict, home_path: Path, - environ: Optional[Dict[str, str]] = None) -> ApplyReport: + environ: Optional[MutableMapping[str, str]] = None) -> ApplyReport: """Fetch from every enabled source and apply the merged result to env. ``environ`` defaults to ``os.environ``; injectable for tests. @@ -376,7 +386,7 @@ def apply_all(secrets_cfg: dict, home_path: Path, for source in ordered: cfg = secrets_cfg.get(source.name) cfg = cfg if isinstance(cfg, dict) else {} - result = _fetch_with_timeout(source, cfg, home_path) + result = _fetch_with_timeout(source, cfg, home_path, env) fetches.append((source, cfg, result)) try: for var in source.protected_env_vars(cfg): diff --git a/agent/session_activity.py b/agent/session_activity.py new file mode 100644 index 0000000000..243f30a5a4 --- /dev/null +++ b/agent/session_activity.py @@ -0,0 +1,106 @@ +"""Shared session activity observation contract (#72016 / #72039). + +Observation-only: timestamp + bounded description/provenance. +Notification, timeout, kill, and retry policy stay in their own components. +Consumers distinguish work (API / tool / compacting / stalled) from the +description text itself — there is no separate phase enum. + +Provenance is a small closed enum of *noun* sources (where the stamp came +from). The default agent activity clock (``_touch_activity``) stamps +``unknown`` unless a caller passes an explicit ``provenance=``; named +values are for special writers. +""" + +from __future__ import annotations + +from enum import Enum +from typing import Any, Mapping, Optional + +ACTIVITY_DESCRIPTION_MAX = 120 + +# Durable SessionDB activity heartbeat cadence (seconds between writes per +# session). Contract: MUST stay >= 30s — the SessionDB write path is +# contended (deadline/patience retry, compression-lock patience), and the +# heartbeat is an observation-only projection that never justifies extra +# write pressure. This cadence is deliberately a code constant, independent +# of any compression.* or agent.* config, so no configuration can turn the +# heartbeat into a high-frequency writer. Matches the kanban auto-heartbeat +# cadence. force_persist (terminal stamps) is the only bypass. +SESSION_ACTIVITY_HEARTBEAT_MIN_INTERVAL_SECONDS = 60.0 + + +class ActivityProvenance(str, Enum): + """Where a durable/in-memory activity stamp came from.""" + + UNKNOWN = "unknown" + # Compression writers (#72424 / activity contract): heartbeat, host timeout, cooldown. + AGENT_COMPRESSION = "agent.compression" + AGENT_COMPRESSION_TIMEOUT = "agent.compression_timeout" + AGENT_COMPRESSION_COOLDOWN = "agent.compression_cooldown" + + +def bound_activity_description(description: Optional[str]) -> str: + """Clamp free-form activity text to the shared description budget.""" + text = (description or "").strip() + if len(text) <= ACTIVITY_DESCRIPTION_MAX: + return text + return text[: ACTIVITY_DESCRIPTION_MAX - 1] + "…" + + +def normalize_activity_provenance( + provenance: Optional[ActivityProvenance | str], +) -> ActivityProvenance: + """Return a known provenance, or ``UNKNOWN`` when unset/unrecognized.""" + if isinstance(provenance, ActivityProvenance): + return provenance + value = (provenance or "").strip() + try: + return ActivityProvenance(value) + except ValueError: + return ActivityProvenance.UNKNOWN + + +def reset_session_activity_persist_window(agent: Any) -> None: + """Clear the agent's durable SessionDB activity persist rate-limit window. + + The next ``_touch_activity`` / ``_persist_session_activity_if_due`` will + write through even if a stamp landed within the last 60s. Used for + terminal compression labels that must not stay stuck on mid-compress + text (e.g. "context compression in progress" after /compress). + """ + try: + agent._session_activity_last_persist_mono = 0.0 + except Exception: + pass + + +def build_activity_snapshot( + *, + last_activity_at: Optional[float], + last_activity_description: Optional[str], + last_activity_provenance: Optional[ActivityProvenance | str] = None, + now: Optional[float] = None, + extra: Optional[Mapping[str, Any]] = None, +) -> dict[str, Any]: + """Build the shared activity snapshot (plus optional caller extras).""" + import time as _time + + when = float(last_activity_at) if last_activity_at is not None else None + clock = float(now if now is not None else _time.time()) + desc = bound_activity_description(last_activity_description) + prov = normalize_activity_provenance(last_activity_provenance) + elapsed = round(clock - when, 1) if when is not None else None + snap: dict[str, Any] = { + "last_activity_at": when, + "last_activity_description": desc, + "last_activity_provenance": prov.value, + "seconds_since_activity": elapsed, + # Short aliases used by existing gateway/delegate readers. + "last_activity_ts": when, + "last_activity_desc": desc, + "description": desc, + "provenance": prov.value, + } + if extra: + snap.update(dict(extra)) + return snap diff --git a/agent/shell_hooks.py b/agent/shell_hooks.py index 5699e4d9eb..db44ddab2b 100644 --- a/agent/shell_hooks.py +++ b/agent/shell_hooks.py @@ -319,8 +319,9 @@ def _parse_hooks_block(hooks_cfg: Any) -> List[ShellHookSpec]: for event_name, entries in hooks_cfg.items(): # Reserved sub-keys that aren't event names — skip silently. These # are config sub-sections nested under `hooks:` for related - # functionality (e.g. output-spill budgets). - if event_name in ("output_spill",): + # functionality (e.g. output-spill budgets, outbound webhooks — + # the latter parsed by agent/outbound_webhooks.py). + if event_name in ("output_spill", "outbound"): continue if event_name not in VALID_HOOKS: suggestion = difflib.get_close_matches( diff --git a/agent/subagent_lifecycle.py b/agent/subagent_lifecycle.py index bbe63ad093..55e110aae5 100644 --- a/agent/subagent_lifecycle.py +++ b/agent/subagent_lifecycle.py @@ -21,6 +21,7 @@ from contextlib import contextmanager from concurrent.futures import Future, TimeoutError from typing import Any, Callable, Mapping, Optional +from agent.interrupt_compat import request_hard_interrupt PUBLIC_CONTRACT_VERSION = 1 _MAX_GOAL_CHARS = 16_000 @@ -300,16 +301,22 @@ class SubagentLifecycleService: agent = record.agent record.state = SubagentState.CANCEL_REQUESTED record.updated_at = time.time() - if agent is None or not hasattr(agent, "interrupt"): + if agent is None: return SubagentCancelResult( False, unsupported=True, state=SubagentState.CANCEL_REQUESTED ) try: - agent.interrupt(f"Lifecycle cancellation requested: {reason[:500]}") + accepted = request_hard_interrupt( + agent, f"Lifecycle cancellation requested: {reason[:500]}" + ) except Exception: return SubagentCancelResult( False, unsupported=True, state=SubagentState.CANCEL_REQUESTED ) + if not accepted: + return SubagentCancelResult( + False, unsupported=True, state=SubagentState.CANCEL_REQUESTED + ) return SubagentCancelResult(True, state=SubagentState.CANCEL_REQUESTED) def result(self, handle: SubagentHandle) -> SubagentResult: diff --git a/agent/tool_dispatch_helpers.py b/agent/tool_dispatch_helpers.py index 07b2c2e65e..f7f003f24a 100644 --- a/agent/tool_dispatch_helpers.py +++ b/agent/tool_dispatch_helpers.py @@ -4,9 +4,10 @@ Pure module-level utilities extracted from ``run_agent.py``: * ``_is_destructive_command`` — terminal-command heuristic used to gate parallel batch dispatch. -* ``_should_parallelize_tool_batch`` / ``_extract_parallel_scope_path`` / - ``_paths_overlap`` — the rules engine deciding when a multi-tool batch - can run concurrently. +* ``_should_parallelize_tool_batch`` / ``_extract_parallel_scope_paths`` / + ``_extract_parallel_scope_path`` / ``_paths_overlap`` — the rules engine + deciding when a multi-tool batch can run concurrently (V4A patch scope + uses patch-body file headers, not a decoy ``path=``). * ``_is_multimodal_tool_result`` / ``_multimodal_text_summary`` / ``_append_subdir_hint_to_multimodal`` — envelope helpers for the ``{"_multimodal": True, "content": [...], "text_summary": ...}`` dict @@ -57,8 +58,17 @@ _PARALLEL_SAFE_TOOLS = frozenset({ "web_search", }) +# Filesystem tools whose parallel admission is decided by path overlap. +# Readers may share a subtree with other readers; a writer conflicts with +# ANY overlapping reservation (reader or writer). This is what keeps a +# batched ``search_files``/``read_file`` from observing pre-mutation file +# state when the model batches it alongside the ``patch``/``write_file`` +# it depends on (the classic same-block write→read race). +_PATH_SCOPED_READERS = frozenset({"read_file", "search_files"}) +_PATH_SCOPED_WRITERS = frozenset({"write_file", "patch"}) + # File tools can run concurrently when they target independent paths. -_PATH_SCOPED_TOOLS = frozenset({"read_file", "write_file", "patch"}) +_PATH_SCOPED_TOOLS = _PATH_SCOPED_READERS | _PATH_SCOPED_WRITERS # Patterns that indicate a terminal command may modify/delete files. _DESTRUCTIVE_PATTERNS = re.compile( @@ -116,10 +126,18 @@ def _plan_tool_batch_segments(tool_calls, *, execution_cwd: Optional[Path] = Non * ``_NEVER_PARALLEL_TOOLS`` (interactive tools) → barrier. * Unparseable / non-dict arguments → barrier. - * Path-scoped tools (``read_file``/``write_file``/``patch``) join a - parallel run only when their target path does not overlap another - path already reserved in the same run; an overlap closes the run so - the conflicting call starts a NEW run after the first completes. + * Path-scoped tools (``read_file``/``search_files``/``write_file``/ + ``patch``) join a parallel run only when their target path(s) do not + CONFLICT with a path already reserved in the same run. Reservations + carry a reader/writer role: reader↔reader overlap is harmless (two + reads of the same file commute) and stays parallel; any overlap + involving a writer closes the run so the conflicting call starts a + NEW run after the first completes. ``search_files`` reserves its + search root (default ``.``) as a reader — a search batched after a + write into the searched subtree is ordered behind that write instead + of racing it. For V4A ``patch(mode="patch")`` the reserved paths are + the file headers in the patch body, not a possibly-stale ``path=`` + argument. * Anything not in ``_PARALLEL_SAFE_TOOLS`` and not an opted-in MCP tool → barrier. @@ -129,7 +147,8 @@ def _plan_tool_batch_segments(tool_calls, *, execution_cwd: Optional[Path] = Non """ segments: list[list] = [] # [kind, calls] pairs, normalized to tuples on return current: list = [] - reserved_paths: list[Path] = [] + # (canonical_path, is_writer) reservations for the current parallel run. + reserved_paths: list[tuple[Path, bool]] = [] def _close_parallel() -> None: nonlocal current, reserved_paths @@ -173,15 +192,25 @@ def _plan_tool_batch_segments(tool_calls, *, execution_cwd: Optional[Path] = Non continue if tool_name in _PATH_SCOPED_TOOLS: - scoped_path = _extract_parallel_scope_path(tool_name, function_args, execution_cwd=execution_cwd) - if scoped_path is None: + scoped_paths = _extract_parallel_scope_paths( + tool_name, function_args, execution_cwd=execution_cwd + ) + if not scoped_paths: _add_sequential(tool_call) continue - if any(_paths_overlap(scoped_path, existing) for existing in reserved_paths): + is_writer = tool_name in _PATH_SCOPED_WRITERS + if any( + (is_writer or existing_is_writer) + and _paths_overlap(scoped_path, existing) + for scoped_path in scoped_paths + for existing, existing_is_writer in reserved_paths + ): # Same-subtree conflict inside this run: close it so this # call starts a fresh run AFTER the conflicting one lands. + # Reader↔reader overlap never conflicts — concurrent reads + # of the same subtree commute. _close_parallel() - reserved_paths.append(scoped_path) + reserved_paths.extend((p, is_writer) for p in scoped_paths) current.append(tool_call) continue @@ -233,33 +262,77 @@ def _canonical_path(raw_path: str, execution_cwd: Optional[Path] = None) -> Path return Path(resolved) -def _extract_parallel_scope_path( +def _extract_parallel_scope_paths( tool_name: str, function_args: dict, execution_cwd: Optional[Path] = None, -) -> Optional[Path]: - """Return the canonical file target for path-scoped tools. +) -> List[Path]: + """Return every canonical path this call reserves for overlap checks. *execution_cwd* should be the working directory that the tool will actually use at runtime. When omitted the process cwd is used, which may differ from the tool execution environment on some platforms (e.g. WSL, sandboxed sub-processes). + + For ``patch`` in V4A ``mode=patch``, scope comes from patch-body + ``*** Update/Add/Delete/Move File:`` headers (not a possibly-decoy + ``path=``). An empty result means the planner cannot determine the + scope and must treat the call as a sequential barrier. """ if tool_name not in _PATH_SCOPED_TOOLS: - return None + return [] - raw_path = function_args.get("path") - if not isinstance(raw_path, str) or not raw_path.strip(): - return None + raw_paths: List[str] = [] + if tool_name == "patch" and (function_args.get("mode") or "replace") == "patch": + raw_paths.extend(_extract_file_mutation_targets(tool_name, function_args)) + else: + raw_path = function_args.get("path") + if isinstance(raw_path, str) and raw_path.strip(): + raw_paths.append(raw_path) + elif tool_name == "search_files": + # ``search_files`` defaults its search root to the cwd when + # ``path`` is omitted — reserve that root rather than falling + # back to a sequential barrier (an empty result here would + # demote every bare search to a barrier and destroy read + # parallelism). + raw_paths.append(".") - return _canonical_path(raw_path, execution_cwd) + scoped: List[Path] = [] + seen: set[str] = set() + for raw in raw_paths: + if not isinstance(raw, str) or not raw.strip(): + continue + canonical = _canonical_path(raw, execution_cwd) + key = str(canonical) + if key in seen: + continue + seen.add(key) + scoped.append(canonical) + return scoped + + +def _extract_parallel_scope_path( + tool_name: str, + function_args: dict, + execution_cwd: Optional[Path] = None, +) -> Optional[Path]: + """Return the primary canonical file target for path-scoped tools. + + Thin view over ``_extract_parallel_scope_paths`` kept for callers/tests + that only need a single representative path. For multi-file V4A + patches this is the first header target. + """ + scoped = _extract_parallel_scope_paths( + tool_name, function_args, execution_cwd=execution_cwd + ) + return scoped[0] if scoped else None def _paths_overlap(left: Path, right: Path) -> bool: """Return True when two paths may refer to the same subtree. Both *left* and *right* must already be canonical (as returned by - ``_extract_parallel_scope_path`` / ``_canonical_path``) so that + ``_extract_parallel_scope_paths`` / ``_canonical_path``) so that symlink aliases and case differences are already normalised. """ left_parts = left.parts @@ -354,8 +427,10 @@ def _extract_file_mutation_targets(tool_name: str, args: Dict[str, Any]) -> List if not isinstance(body, str) or not body: return [] paths: List[str] = [] + # ``\s*`` (not ``\s+``) after ``***`` matches patch_parser / file_tools: + # they accept ``***Update File:`` with no space after the asterisks. for _m in re.finditer( - r'^\*\*\*\s+(?:Update|Add|Delete)\s+File:\s*(.+)$', + r'^\*\*\*\s*(?:Update|Add|Delete)\s+File:\s*(.+)$', body, re.MULTILINE, ): @@ -363,7 +438,7 @@ def _extract_file_mutation_targets(tool_name: str, args: Dict[str, Any]) -> List if p: paths.append(p) for _m in re.finditer( - r'^\*\*\*\s+Move\s+File:\s*(.+?)\s*->\s*(.+)$', + r'^\*\*\*\s*Move\s+File:\s*(.+?)\s*->\s*(.+)$', body, re.MULTILINE, ): @@ -634,6 +709,8 @@ __all__ = [ "_NEVER_PARALLEL_TOOLS", "_PARALLEL_SAFE_TOOLS", "_PATH_SCOPED_TOOLS", + "_PATH_SCOPED_READERS", + "_PATH_SCOPED_WRITERS", "_DESTRUCTIVE_PATTERNS", "_REDIRECT_OVERWRITE", "_is_destructive_command", @@ -641,6 +718,7 @@ __all__ = [ "_should_parallelize_tool_batch", "_canonical_path", "_extract_parallel_scope_path", + "_extract_parallel_scope_paths", "_paths_overlap", "_is_multimodal_tool_result", "_multimodal_text_summary", diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 23b1a95535..ee2daa07c4 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -1236,8 +1236,8 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe logging.debug("file-mutation verifier record failed: %s", _ver_err) if agent.verbose_logging: - logging.debug(f"Tool {function_name} completed in {tool_duration:.2f}s") - logging.debug(f"Tool result ({len(function_result)} chars): {function_result}") + logging.debug("Tool %s completed in %.2fs", function_name, tool_duration) + logging.debug("Tool result (%d chars): %s", len(function_result), function_result) agent._current_tool = None _status_suffix = " (error)" if is_error else "" @@ -1296,7 +1296,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe result=display_function_result, ) except Exception as cb_err: - logging.debug(f"Tool progress callback error: {cb_err}") + logging.debug("Tool progress callback error: %s", cb_err) # Print cute message per tool if agent._should_emit_quiet_tool_messages(): @@ -1320,7 +1320,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe tc.id, name, display_args, display_function_result, ) except Exception as cb_err: - logging.debug(f"Tool complete callback error: {cb_err}") + logging.debug("Tool complete callback error: %s", cb_err) if ( risk_metadata is not None @@ -1339,12 +1339,10 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe except Exception as cb_err: logging.debug("Tool output risk callback error: %s", cb_err) - # ── Per-tool /steer drain ─────────────────────────────────── - # Same as the sequential path: drain between each collected - # result so the steer lands as early as possible. - agent._apply_pending_steer_to_tool_results(messages, 1) - # ── Per-turn aggregate budget enforcement ───────────────────────── + # Keep /steer pending until the final post-budget drain below. The model + # cannot observe a partial batch, while an early drain can be discarded + # when aggregate budget enforcement replaces that tool result. num_tools = len(parsed_calls) if finalize and num_tools > 0: turn_tool_msgs = messages[-num_tools:] @@ -1359,6 +1357,26 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe +def _append_cancelled_tool_results(messages: list, tool_calls, *, reason: str) -> None: + """Append a cancelled ``tool`` result for each call in ``tool_calls``. + + Used when a hard interrupt (KeyboardInterrupt / BaseException) aborts the + sequential executor mid-batch. Without this, the loop re-raises leaving the + assistant tool-call turn with no matching tool results — a message-role + alternation violation that malforms the next provider request. Mirrors the + cooperative-interrupt skip block and the concurrent path, both of which + already emit a result for every call_id. + """ + for tc in tool_calls: + name = getattr(getattr(tc, "function", None), "name", "") or "tool" + messages.append(make_tool_result_message( + name, + f"[Tool execution cancelled — {name} was skipped due to {reason}]", + getattr(tc, "id", "") or "", + effect_disposition="none", + )) + + def execute_tool_calls_sequential(agent, assistant_message, messages: list, effective_task_id: str, api_call_count: int = 0, *, finalize: bool = True) -> None: """Execute tool calls sequentially (original behavior). Used for single calls or interactive tools. @@ -1439,7 +1457,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe stage=f"invalid tool arguments {function_name}", ): return - agent._apply_pending_steer_to_tool_results(messages, 1) continue # Tool Search unwrap — see execute_tool_calls_concurrent for full @@ -1799,6 +1816,14 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe agent.interrupt("keyboard interrupt") except Exception: pass + # Emit a tool result for THIS call and every remaining call in + # the batch before re-raising, so the assistant tool-call turn + # is never left without matching tool results (alternation). + _append_cancelled_tool_results( + messages, + assistant_message.tool_calls[i - 1:], + reason="keyboard interrupt", + ) raise except Exception as tool_error: function_result = f"Error executing tool '{function_name}': {tool_error}" @@ -1868,6 +1893,13 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe agent.interrupt("keyboard interrupt") except Exception: pass + # Emit a tool result for THIS call and every remaining call in + # the batch before re-raising (see interactive branch above). + _append_cancelled_tool_results( + messages, + assistant_message.tool_calls[i - 1:], + reason="keyboard interrupt", + ) raise except Exception as tool_error: function_result = f"Error executing tool '{function_name}': {tool_error}" @@ -1944,9 +1976,9 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe agent._touch_activity(f"tool completed: {function_name} ({tool_duration:.1f}s){_status_suffix}") if agent.verbose_logging: - logging.debug(f"Tool {function_name} completed in {tool_duration:.2f}s") + logging.debug("Tool %s completed in %.2fs", function_name, tool_duration) _log_result = _multimodal_text_summary(function_result) - logging.debug(f"Tool result ({len(_log_result)} chars): {_log_result}") + logging.debug("Tool result (%d chars): %s", len(_log_result), _log_result) display_function_result = function_result function_result = maybe_persist_tool_result( @@ -1988,7 +2020,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe result=display_function_result, ) except Exception as cb_err: - logging.debug(f"Tool progress callback error: {cb_err}") + logging.debug("Tool progress callback error: %s", cb_err) if not _execution_blocked and agent.tool_complete_callback: try: @@ -2003,7 +2035,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe display_function_result, ) except Exception as cb_err: - logging.debug(f"Tool complete callback error: {cb_err}") + logging.debug("Tool complete callback error: %s", cb_err) if ( risk_metadata is not None @@ -2022,12 +2054,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe except Exception as cb_err: logging.debug("Tool output risk callback error: %s", cb_err) - # ── Per-tool /steer drain ─────────────────────────────────── - # Drain pending steer BETWEEN individual tool calls so the - # injection lands as soon as a tool finishes — not after the - # entire batch. The model sees it on the next API iteration. - agent._apply_pending_steer_to_tool_results(messages, 1) - if not agent.quiet_mode and getattr(agent, "tool_progress_mode", "all") != "off": if agent.verbose_logging: print(f" ✅ Tool {i} completed in {tool_duration:.2f}s") @@ -2057,6 +2083,9 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe break # ── Per-turn aggregate budget enforcement ───────────────────────── + # Keep /steer pending until the final post-budget drain below. The model + # only receives this batch after all calls finish, and an early drain can + # be discarded when aggregate budget enforcement replaces a tool result. num_tools_seq = len(assistant_message.tool_calls) if finalize and num_tools_seq > 0: enforce_turn_budget(messages[-num_tools_seq:], env=get_active_env(effective_task_id), config=_tool_budget) diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 15dd3409e3..de71c2b500 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -544,10 +544,13 @@ class ResponsesApiTransport(ProviderTransport): *, allow_stream: bool = False, is_github_responses: bool = False, + sanitize_harmony_tokens: bool = False, ) -> dict: """Validate and sanitize Codex API kwargs before the call. Normalizes input items, strips unsupported fields, validates structure. + ``sanitize_harmony_tokens`` is enabled only for the ChatGPT Codex + backend, which rejects literal reserved Harmony wire tokens in text. """ from agent.codex_responses_adapter import _preflight_codex_api_kwargs @@ -555,6 +558,7 @@ class ResponsesApiTransport(ProviderTransport): api_kwargs, allow_stream=allow_stream, is_github_responses=is_github_responses, + sanitize_harmony_tokens=sanitize_harmony_tokens, ) if "prompt_cache_key" in normalized: bounded = _bounded_prompt_cache_key(normalized["prompt_cache_key"]) diff --git a/agent/turn_finalizer.py b/agent/turn_finalizer.py index 86f3f50992..d4d6a23e78 100644 --- a/agent/turn_finalizer.py +++ b/agent/turn_finalizer.py @@ -339,6 +339,12 @@ def finalize_turn( # otherwise ``/resume`` reloads ``content=""`` and the bug # resurfaces cross-session. _tail.pop("_db_persisted", None) + # The bounded flush-scan cursor (run_agent.py) skips the + # identity-matched prefix of its previous snapshot on the + # assumption that no live dict loses the marker in place — + # this pop is the one place that does. Invalidate it so the + # filled row is re-examined instead of skipped. + agent._db_flush_scan_prefix = None # The model has completed its request, so replace API-local # voice/model/skill guidance with the clean user input before writing the @@ -378,6 +384,18 @@ def finalize_turn( ): _before = len(messages) _compacted = _compressor._micro_compact(messages) + # Micro-compaction defrag rewrites the newest MICRO + # marker's content and pops _db_persisted from the live + # dict in place — the sibling of the pop site above. The + # compressor has no agent reference, so it raises a flag + # for us to invalidate the bounded flush-scan cursor; + # otherwise the rewritten marker row is identity-skipped + # and the stale summary persists to state.db. + if getattr( + _compressor, "_flush_scan_cursor_invalidated", False + ): + _compressor._flush_scan_cursor_invalidated = False + agent._db_flush_scan_prefix = None if isinstance(_compacted, list) and _compacted: messages[:] = _compacted _after = len(messages) @@ -647,6 +665,13 @@ def finalize_turn( } if agent._tool_guardrail_halt_decision is not None: result["guardrail"] = agent._tool_guardrail_halt_decision.to_metadata() + # Persistence failures already set failed=True + an explanation in + # final_response; also stamp `error` so gateway surfaces status="error" + # (and desktop can toast disk-full) instead of a quiet complete frame. + if failed and str(_turn_exit_reason) == "session_persistence_failed": + result["error"] = final_response or ( + "session storage could not be written — free disk space and try again" + ) # Surface any post-loop cleanup failures so the caller can distinguish a # clean turn from one whose trajectory/session/resource teardown raised # (the response is still returned either way — #8049). diff --git a/agent/turn_retry_state.py b/agent/turn_retry_state.py index 59e343bfed..d73fe5b6bf 100644 --- a/agent/turn_retry_state.py +++ b/agent/turn_retry_state.py @@ -45,6 +45,14 @@ class TurnRetryState: nous_auth_retry_attempted: bool = False nous_paid_entitlement_refresh_attempted: bool = False copilot_auth_retry_attempted: bool = False + # Copilot surfaces a stale/degraded credential as a 400 + # ``model_not_available_for_integrator`` / ``model_not_supported`` instead + # of a clean 401 (e.g. a raw OAuth token seeded when the token exchange + # degraded at startup, routing the request to the restricted + # ``copilot-language-server`` integrator). Guard a single-shot forced + # re-exchange + client rebuild for that case, separate from the 401 guard + # so both can fire within one attempt if needed. + copilot_stale_cred_retry_attempted: bool = False vertex_auth_retry_attempted: bool = False # ── Format / payload recovery guards ───────────────────────────────── diff --git a/apps/bootstrap-installer/package.json b/apps/bootstrap-installer/package.json index 32c3e7f8c3..5e72fb283f 100644 --- a/apps/bootstrap-installer/package.json +++ b/apps/bootstrap-installer/package.json @@ -19,41 +19,33 @@ "fix": "npm run lint:fix" }, "dependencies": { - "@nous-research/ui": "0.16.0", - "@tailwindcss/vite": "^4.2.4", - "@tailwindcss/typography": "^0.5.19", - "@tauri-apps/api": "^2.0.0", - "@tauri-apps/plugin-dialog": "^2.0.0", - "@tauri-apps/plugin-opener": "^2.0.0", - "@tauri-apps/plugin-process": "^2.0.0", - "@tauri-apps/plugin-shell": "^2.0.0", - "@vscode/codicons": "^0.0.45", - "class-variance-authority": "^0.7.1", - "clsx": "^2.1.1", - "katex": "^0.16.45", - "lucide-react": "^0.577.0", - "nanostores": "^1.3.0", - "radix-ui": "^1.4.3", - "react": "^19.2.4", - "react-dom": "^19.2.4", - "tailwind-merge": "^3.5.0", - "tailwindcss": "^4.2.1", - "tw-shimmer": "^0.4.11" + "@nous-research/ui": "0.18.2", + "@tailwindcss/typography": "0.5.20", + "@tailwindcss/vite": "4.3.3", + "@tauri-apps/api": "2.11.1", + "@tauri-apps/plugin-dialog": "2.7.1", + "@tauri-apps/plugin-opener": "2.5.4", + "@tauri-apps/plugin-process": "2.3.1", + "@tauri-apps/plugin-shell": "2.3.5", + "@vscode/codicons": "0.0.45", + "class-variance-authority": "0.7.1", + "clsx": "2.1.1", + "katex": "0.16.47", + "lucide-react": "0.577.0", + "nanostores": "1.4.0", + "radix-ui": "1.6.7", + "react": "19.2.7", + "react-dom": "19.2.7", + "tailwind-merge": "3.6.0", + "tailwindcss": "4.3.3", + "tw-shimmer": "0.4.12" }, "devDependencies": { - "@eslint/js": "^9.39.4", - "@tauri-apps/cli": "^2.0.0", - "@types/react": "^19.2.14", - "@types/react-dom": "^19.2.3", - "@vitejs/plugin-react": "^6.0.2", - "eslint": "^9.39.4", - "eslint-plugin-perfectionist": "^5.9.0", - "eslint-plugin-react": "^7.37.5", - "eslint-plugin-react-hooks": "^7.1.1", - "eslint-plugin-unused-imports": "^4.4.1", - "globals": "^17.4.0", - "typescript": "^6.0.3", - "typescript-eslint": "^8.56.1", - "vite": "^8.0.16" + "@tauri-apps/cli": "2.11.4", + "@types/react": "19.2.17", + "@types/react-dom": "19.2.3", + "@vitejs/plugin-react": "6.0.3", + "typescript": "6.0.3", + "vite": "8.2.0" } } diff --git a/apps/bootstrap-installer/src-tauri/src/paths.rs b/apps/bootstrap-installer/src-tauri/src/paths.rs index 0eec8ccd31..3a7b1b0dbf 100644 --- a/apps/bootstrap-installer/src-tauri/src/paths.rs +++ b/apps/bootstrap-installer/src-tauri/src/paths.rs @@ -98,6 +98,12 @@ pub fn update_in_progress_marker() -> PathBuf { /// that path), where copying onto ourselves would be a Windows sharing /// violation. Best-effort: a failure here must not fail the install, so the /// caller logs and continues. +/// +/// NOTE: because of that no-op, a user's staged installer is only ever written +/// by a full install/repair. Every later `--update` runs the ORIGINAL binary, +/// so an installer-protocol change can strand the whole installed base on a +/// binary that predates it (see `restage_from_checkout`, which repairs this +/// from the freshly-updated checkout). pub fn copy_self_to_hermes_home() -> std::io::Result<()> { let src = std::env::current_exe()?; let dest = installer_dest(); diff --git a/apps/desktop/assets/icon.ico b/apps/desktop/assets/icon.ico index 1dcf20d8c3..b9de6754c2 100644 Binary files a/apps/desktop/assets/icon.ico and b/apps/desktop/assets/icon.ico differ diff --git a/apps/desktop/electron/backend-env.test.ts b/apps/desktop/electron/backend-env.test.ts index a92ce6e062..986f164f9e 100644 --- a/apps/desktop/electron/backend-env.test.ts +++ b/apps/desktop/electron/backend-env.test.ts @@ -7,6 +7,7 @@ import { appendUniquePathEntries, buildDesktopBackendEnv, buildDesktopBackendPath, + hermesManagedNodePathEntries, normalizeHermesHomeRoot, pathEnvKey, POSIX_SANE_PATH_ENTRIES @@ -22,8 +23,12 @@ test('desktop backend PATH adds Hermes-managed bins and missing POSIX sane entri }) const entries = result.split(':') - assert.equal(entries[0], '/Users/test/.hermes/node/bin') - assert.equal(entries[1], '/Users/test/.hermes/hermes-agent/venv/bin') + // Both managed-Node layouts lead, POSIX-native shape first, then the venv. + assert.deepEqual(entries.slice(0, 3), [ + '/Users/test/.hermes/node/bin', + '/Users/test/.hermes/node', + '/Users/test/.hermes/hermes-agent/venv/bin' + ]) assert.ok(entries.includes('/opt/homebrew/bin'), 'Apple Silicon Homebrew bin is added') assert.ok(entries.includes('/opt/homebrew/sbin'), 'Apple Silicon Homebrew sbin is added') assert.ok(entries.includes('/usr/local/sbin'), 'missing standard sbin is added') @@ -33,6 +38,56 @@ test('desktop backend PATH adds Hermes-managed bins and missing POSIX sane entri } }) +test('managed Node dirs lead with the platform-native layout but always offer both', () => { + const posix = hermesManagedNodePathEntries('/Users/test/.hermes', { + platform: 'darwin', + pathModule: path.posix + }) + + const windows = hermesManagedNodePathEntries('C:\\Users\\test\\AppData\\Local\\hermes', { + platform: 'win32', + pathModule: path.win32 + }) + + // install.sh uses node/bin; install.ps1 unpacks node.exe into node\ itself. + // Both shapes are always emitted so migrated installs keep resolving. + assert.deepEqual(posix, ['/Users/test/.hermes/node/bin', '/Users/test/.hermes/node']) + assert.deepEqual(windows, [ + 'C:\\Users\\test\\AppData\\Local\\hermes\\node', + 'C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin' + ]) +}) + +test('managed Node dirs are empty without a Hermes home', () => { + assert.deepEqual(hermesManagedNodePathEntries(undefined, { platform: 'darwin', pathModule: path.posix }), []) + assert.deepEqual(hermesManagedNodePathEntries('', { platform: 'win32', pathModule: path.win32 }), []) +}) + +test('every managed Node dir outranks the inherited PATH on both platforms', () => { + for (const [platform, pathModule, home, inherited, delimiter] of [ + ['darwin', path.posix, '/Users/test/.hermes', '/usr/local/bin:/usr/bin', ':'], + ['win32', path.win32, 'C:\\hermes', 'C:\\Program Files\\nodejs;C:\\Windows\\System32', ';'] + ] as const) { + const entries = buildDesktopBackendPath({ + hermesHome: home, + venvRoot: null, + currentPath: inherited, + platform, + pathModule + }).split(delimiter) + + const managed = hermesManagedNodePathEntries(home, { platform, pathModule }) + const firstInherited = Math.min(...inherited.split(delimiter).map(entry => entries.indexOf(entry))) + + for (const dir of managed) { + assert.ok( + entries.indexOf(dir) >= 0 && entries.indexOf(dir) < firstInherited, + `${dir} must precede the inherited PATH on ${platform}` + ) + } + } +}) + test('desktop backend PATH preserves first occurrence and avoids duplicates', () => { const result = buildDesktopBackendPath({ hermesHome: '/Users/test/.hermes', @@ -64,7 +119,11 @@ test('buildDesktopBackendEnv extends PYTHONPATH and backend PATH together', () = }) assert.equal(env.PYTHONPATH, '/repo/hermes-agent:/existing/pythonpath') - assert.ok(env.PATH.startsWith('/Users/test/.hermes/node/bin:/Users/test/.hermes/hermes-agent/venv/bin:')) + assert.ok( + env.PATH.startsWith( + '/Users/test/.hermes/node/bin:/Users/test/.hermes/node:/Users/test/.hermes/hermes-agent/venv/bin:' + ) + ) assert.ok(env.PATH.includes('/opt/homebrew/bin')) }) @@ -115,7 +174,13 @@ test('Windows PATH casing and delimiter are preserved without POSIX sane entries assert.equal(pathEnvKey({ Path: 'x' }, 'win32'), 'Path') assert.equal(env.PATH, undefined) - assert.ok(env.Path.startsWith('C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin;')) + // Windows leads with the portable layout (install.ps1 unpacks node.exe + // straight into node\, no bin\), then the POSIX shape for migrated installs. + assert.ok( + env.Path.startsWith( + 'C:\\Users\\test\\AppData\\Local\\hermes\\node;C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin;' + ) + ) assert.ok(env.Path.includes('\\venv\\Scripts;')) assert.ok(env.Path.includes(';C:\\Windows\\System32;C:\\Windows')) assert.equal(env.Path.includes('/opt/homebrew/bin'), false) diff --git a/apps/desktop/electron/backend-env.ts b/apps/desktop/electron/backend-env.ts index 3db4a19d03..24d6928bad 100644 --- a/apps/desktop/electron/backend-env.ts +++ b/apps/desktop/electron/backend-env.ts @@ -60,6 +60,34 @@ function appendUniquePathEntries(entries, { delimiter = path.delimiter } = {}) { return ordered.join(delimiter) } +/** + * Hermes-managed Node.js directories, in preferred lookup order. + * + * There are two on-disk layouts. `scripts/install.ps1` unpacks portable Node + * straight into `%LOCALAPPDATA%\hermes\node` (node.exe at the root, no `bin\`); + * `scripts/install.sh` and the node-bootstrap helper use the POSIX + * `$HERMES_HOME/node/bin`. Emit BOTH on every platform so mixed and migrated + * installs resolve, leading with the layout native to the current platform. + * + * This is the single source of truth for the ordering rule on the Node side — + * `main.ts` imports it rather than keeping its own copy. Mirrors + * `iter_hermes_node_dirs()` in hermes_constants.py, which the Electron main + * process cannot import. + */ +function hermesManagedNodePathEntries( + hermesHome, + { platform = process.platform, pathModule = pathModuleForPlatform(platform) }: any = {} +) { + if (!hermesHome) { + return [] + } + + const root = pathModule.join(hermesHome, 'node') + const bin = pathModule.join(root, 'bin') + + return platform === 'win32' ? [root, bin] : [bin, root] +} + function buildDesktopBackendPath({ hermesHome, venvRoot, @@ -68,11 +96,11 @@ function buildDesktopBackendPath({ pathModule = pathModuleForPlatform(platform) }: any = {}) { const delimiter = delimiterForPlatform(platform) - const hermesNodeBin = hermesHome ? pathModule.join(hermesHome, 'node', 'bin') : null + const hermesNodeDirs = hermesManagedNodePathEntries(hermesHome, { platform, pathModule }) const venvBin = venvRoot ? pathModule.join(venvRoot, platform === 'win32' ? 'Scripts' : 'bin') : null const saneEntries = platform === 'win32' ? [] : POSIX_SANE_PATH_ENTRIES - return appendUniquePathEntries([hermesNodeBin, venvBin, currentPath, saneEntries], { delimiter }) + return appendUniquePathEntries([hermesNodeDirs, venvBin, currentPath, saneEntries], { delimiter }) } function normalizeHermesHomeRoot(hermesHome, { pathModule = pathModuleForPlatform(process.platform) }: any = {}) { @@ -126,6 +154,7 @@ export { buildDesktopBackendEnv, buildDesktopBackendPath, delimiterForPlatform, + hermesManagedNodePathEntries, normalizeHermesHomeRoot, pathEnvKey, POSIX_SANE_PATH_ENTRIES diff --git a/apps/desktop/electron/bootstrap-repair-guard.test.ts b/apps/desktop/electron/bootstrap-repair-guard.test.ts new file mode 100644 index 0000000000..e1a7a1ecf3 --- /dev/null +++ b/apps/desktop/electron/bootstrap-repair-guard.test.ts @@ -0,0 +1,109 @@ +import assert from 'node:assert/strict' + +import { test } from 'vitest' + +import { decideBootstrapRepair } from './bootstrap-repair-guard' + +test('first soft attempt with alive backend returns soft restart', () => { + const decision = decideBootstrapRepair({ + attempt: 1, + primaryBackendAlive: true + }) + + assert.equal(decision.hardReinstall, false) + assert.equal(decision.attempt, 1) + assert.match(decision.reason, /still alive/) + assert.match(decision.reason, /1\/3/) +}) + +test('first attempt with dead backend still returns soft restart', () => { + const decision = decideBootstrapRepair({ + attempt: 1, + primaryBackendAlive: false + }) + + assert.equal(decision.hardReinstall, false) + assert.match(decision.reason, /has exited/) +}) + +test('soft restart budget exhausts at maxSoftAttempts+1 and escalates', () => { + const decision = decideBootstrapRepair({ + attempt: 4, + maxSoftAttempts: 3, + primaryBackendAlive: true + }) + + assert.equal(decision.hardReinstall, true) + assert.equal(decision.attempt, 4) + assert.match(decision.reason, /exceeds soft-restart budget/) +}) + +test('attempt exactly at maxSoftAttempts is still soft', () => { + const decision = decideBootstrapRepair({ + attempt: 3, + maxSoftAttempts: 3, + primaryBackendAlive: true + }) + + assert.equal(decision.hardReinstall, false) + assert.equal(decision.attempt, 3) +}) + +test('custom maxSoftAttempts is honored', () => { + const soft = decideBootstrapRepair({ + attempt: 5, + maxSoftAttempts: 10, + primaryBackendAlive: true + }) + + assert.equal(soft.hardReinstall, false) + + const hard = decideBootstrapRepair({ + attempt: 11, + maxSoftAttempts: 10, + primaryBackendAlive: false + }) + + assert.equal(hard.hardReinstall, true) +}) + +test('default maxSoftAttempts is 3', () => { + // Probe the default indirectly: attempt 4 with no override must escalate. + const decision = decideBootstrapRepair({ + attempt: 4, + primaryBackendAlive: true + }) + + assert.equal(decision.hardReinstall, true) +}) + +test('fractional or zero attempts are clamped to 1', () => { + const zeroDecision = decideBootstrapRepair({ + attempt: 0, + primaryBackendAlive: true + }) + + assert.equal(zeroDecision.attempt, 1) + assert.equal(zeroDecision.hardReinstall, false) + + const fractionalDecision = decideBootstrapRepair({ + attempt: 2.7, + primaryBackendAlive: true + }) + + assert.equal(fractionalDecision.attempt, 2) + assert.equal(fractionalDecision.hardReinstall, false) +}) + +test('alive=false on a high attempt number still escalates (defense in depth)', () => { + // A dead backend should normally be handled by the renderer before it + // reaches the repair path, but if it does reach us with a high attempt + // count we still escalate — never silently keep soft-restarting. + const decision = decideBootstrapRepair({ + attempt: 5, + maxSoftAttempts: 3, + primaryBackendAlive: false + }) + + assert.equal(decision.hardReinstall, true) +}) diff --git a/apps/desktop/electron/bootstrap-repair-guard.ts b/apps/desktop/electron/bootstrap-repair-guard.ts new file mode 100644 index 0000000000..f971bcccfe --- /dev/null +++ b/apps/desktop/electron/bootstrap-repair-guard.ts @@ -0,0 +1,121 @@ +/** + * Repair-loop guard for the desktop bootstrap. + * + * Why this exists + * ─────────────── + * Hermes desktop can request a "repair" of its bundled backend when the + * renderer observes a transient backend failure (see issue #74874). The + * classic failure fingerprint: + * + * 1. Backend Python process hits a transient GIL stall (e.g. heavy + * import, MCP discovery, a long-running agent turn). + * 2. The renderer's WebSocket can't deliver the `gateway.ready` frame + * in time and treats the socket as dead. + * 3. Renderer calls `hermes:bootstrap:repair`. + * 4. Bootstrap unconditionally force-reinstalls the venv, restarting + * the backend — which stalls again for the same reason. + * 5. Renderer reports dead backend → another repair → infinite loop. + * + * The desktop should distinguish: + * - "the venv/install is genuinely broken" → hard reinstall is correct + * - "the runtime is healthy but temporarily stalled" → restart only, + * NOT a destructive reinstall that drops the venv + * + * What this module does + * ───────────────────── + * A pure decision helper. Given the current repair attempt count and a + * hint about whether the live backend process still looks alive, return + * whether the next repair should: + * - `hardReinstall: true` → run the installer, recreate the venv + * - `hardReinstall: false` → restart the existing backend, keep the venv + * + * Cap on soft restarts is bounded so an actually-corrupted install still + * eventually escalates to a hard reinstall after repeated stalls — the + * guard prevents the *unbounded* reinstall loop, not all reinstalls. + * + * The module is intentionally pure (no I/O, no logging, no global state) + * so it is unit-testable in isolation. Wiring into `main.ts` lives there. + */ + +export type RepairDecision = + | { + /** Run the installer (recreate venv). Caller bypasses the active runtime. */ + hardReinstall: true + /** Human-readable rationale for the desktop log. */ + reason: string + /** 1-indexed repair attempt number for diagnostics. */ + attempt: number + } + | { + /** Skip the installer; restart the existing backend only. */ + hardReinstall: false + reason: string + attempt: number + } + +export type RepairDecisionInput = { + /** + * 1-indexed count of how many repair attempts have happened in this + * failure episode. The first repair is `attempt === 1`; a successful + * boot resets the counter (see `main.ts`'s bootstrap completion path). + */ + attempt: number + /** + * Soft-restart budget before escalation to a hard reinstall. Defaults + * to 3: three "just restart" attempts, then a real reinstall. Bounded + * so a corrupt install still gets fixed; high enough that a GIL + * stall no longer loops the user into a 30-minute reinstall cycle. + */ + maxSoftAttempts?: number + /** + * Whether the live backend process (the one we are about to tear down + * to honour the repair request) still looks alive. A process whose + * `exitCode !== null` or `signalCode !== null` has actually exited; + * a process with both null is either still running or stalled — and a + * stall is exactly the case the soft-restart path is for. + */ + primaryBackendAlive: boolean +} + +/** + * Decide the next repair action. + * + * Decision matrix: + * attempt ≤ maxSoftAttempts AND alive → soft restart (don't reinstall) + * attempt ≤ maxSoftAttempts AND dead → soft restart (process exited, + * but we don't yet trust that + * the install is corrupt; restart + * once to confirm) + * attempt > maxSoftAttempts → hard reinstall (give up on the + * current install) + * + * "Alive" being true does NOT force a soft restart on every call: the + * attempt counter still increments, so an actually-broken install that + * keeps respawning a child but never announces READY still escalates + * after `maxSoftAttempts` cycles. + */ +export function decideBootstrapRepair(input: RepairDecisionInput): RepairDecision { + const maxSoftAttempts = input.maxSoftAttempts ?? 3 + const attempt = Math.max(1, Math.floor(input.attempt)) + const alive = Boolean(input.primaryBackendAlive) + + if (attempt > maxSoftAttempts) { + return { + hardReinstall: true, + attempt, + reason: + `repair attempt ${attempt} exceeds soft-restart budget ` + `(${maxSoftAttempts}); escalating to hard reinstall` + } + } + + return { + hardReinstall: false, + attempt, + reason: alive + ? `repair attempt ${attempt}/${maxSoftAttempts}: primary backend process ` + + `still alive (likely transient stall, see #74874); restarting only, ` + + `skipping installer` + : `repair attempt ${attempt}/${maxSoftAttempts}: primary backend process ` + + `has exited; restarting before escalating to reinstall` + } +} diff --git a/apps/desktop/electron/connection-config.test.ts b/apps/desktop/electron/connection-config.test.ts index 5bae2e8e70..a306bec5ca 100644 --- a/apps/desktop/electron/connection-config.test.ts +++ b/apps/desktop/electron/connection-config.test.ts @@ -133,16 +133,47 @@ test('profileRemoteOverride tolerates a missing/!object profiles map', () => { assert.equal(profileRemoteOverride(null, 'coder'), null) }) -test('SSH remains separate from URL-shaped remote modes', () => { +test('SSH remains separate from URL-shaped remote modes and preserves an explicit remote profile', () => { assert.equal(modeIsRemoteLike('ssh'), false) - const config = { profiles: { coder: { mode: 'ssh', host: 'alice@box:2222', keyPath: '/key' } } } + + const config = { + profiles: { coder: { mode: 'ssh', host: 'alice@box:2222', keyPath: '/key', remoteProfile: 'default' } } + } + assert.equal(profileRemoteOverride(config, 'coder'), null) + assert.deepEqual(profileSshOverride(config, 'coder'), { mode: 'ssh', host: 'box', user: 'alice', port: 2222, - keyPath: '/key' + keyPath: '/key', + remoteProfile: 'default' + }) +}) + +test('normalizeSshConfig rejects unsafe remote profile mappings', () => { + assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'writer_2' }), { + mode: 'ssh', + host: 'box', + remoteProfile: 'writer_2' + }) + assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'bad profile' }), { + mode: 'ssh', + host: 'box' + }) + assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: '' }), { + mode: 'ssh', + host: 'box' + }) + assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'root' }), { + mode: 'ssh', + host: 'box' + }) + assert.deepEqual(normalizeSshConfig({ mode: 'ssh', host: 'box', remoteProfile: 'default' }), { + mode: 'ssh', + host: 'box', + remoteProfile: 'default' }) }) @@ -350,6 +381,19 @@ test('normalizeRemoteBaseUrl rejects garbage', () => { assert.throws(() => normalizeRemoteBaseUrl('not a url'), /not valid/) }) +test('normalizeRemoteBaseUrl auto-prepends http:// for scheme-less host:port input', () => { + assert.equal(normalizeRemoteBaseUrl('100.64.0.1:9119'), 'http://100.64.0.1:9119') + assert.equal(normalizeRemoteBaseUrl('mini.tailnet-1234.ts.net:9119'), 'http://mini.tailnet-1234.ts.net:9119') + assert.equal(normalizeRemoteBaseUrl('localhost:9119'), 'http://localhost:9119') + assert.equal(normalizeRemoteBaseUrl('gw.example.com'), 'http://gw.example.com') + assert.equal(normalizeRemoteBaseUrl('gw.example.com/hermes/'), 'http://gw.example.com/hermes') +}) + +test('normalizeRemoteBaseUrl still rejects explicit non-http(s) schemes after scheme-less handling', () => { + assert.throws(() => normalizeRemoteBaseUrl('ws://host:9119'), /http:\/\/ or https:\/\//) + assert.throws(() => normalizeRemoteBaseUrl('ftp://host:21'), /http:\/\/ or https:\/\//) +}) + // --- buildGatewayWsUrl (token) --- test('buildGatewayWsUrl uses wss for https and bakes the token', () => { diff --git a/apps/desktop/electron/connection-config.ts b/apps/desktop/electron/connection-config.ts index c15644b112..4644008d48 100644 --- a/apps/desktop/electron/connection-config.ts +++ b/apps/desktop/electron/connection-config.ts @@ -45,14 +45,27 @@ const RT_COOKIE_VARIANTS = ['__Host-hermes_session_rt', '__Secure-hermes_session // cookies above. `privy-token` is the access token (the required signal); // variants cover the secured-prefix forms and the older `privy-session` name. const PRIVY_SESSION_COOKIE_VARIANTS = ['__Host-privy-token', '__Secure-privy-token', 'privy-token', 'privy-session'] +// Keep this aligned with hermes_cli.profiles.validate_profile_name(). `default` +// is the built-in root alias; these names cannot be created as profiles. +const RESERVED_REMOTE_PROFILES = new Set(['hermes', 'test', 'tmp', 'root', 'sudo']) function normalizeRemoteBaseUrl(rawUrl) { - const value = String(rawUrl || '').trim() + let value = String(rawUrl || '').trim() if (!value) { throw new Error('Remote gateway URL is required.') } + // Users routinely paste scheme-less "host:port" (a Tailscale IP, a LAN + // hostname). Without this, `new URL('100.64.0.1:9119')` either throws or — + // worse — parses `host:` as the protocol and produces a baffling + // "must be http:// or https://, got myhost:" error. Only a real + // `scheme://` prefix opts out, so explicit non-http schemes (ftp://, + // file://) still reach the protocol check below and get rejected. + if (!/^[a-z][a-z0-9+.-]*:\/\//i.test(value)) { + value = `http://${value}` + } + let parsed try { @@ -271,6 +284,16 @@ function normalizeSshConfig(entry) { out.remoteHermesPath = remoteHermesPath } + // A Desktop profile can be a local routing label rather than the profile + // name used by the remote Hermes installation. Preserve an explicit mapping + // when it is a valid Hermes profile identifier; otherwise fall back to the + // historical same-name behavior in the caller. + const remoteProfile = String(entry.remoteProfile || '').trim() + + if (/^[a-z0-9][a-z0-9_-]{0,63}$/.test(remoteProfile) && !RESERVED_REMOTE_PROFILES.has(remoteProfile)) { + out.remoteProfile = remoteProfile + } + return out } diff --git a/apps/desktop/electron/git-review-ops.test.ts b/apps/desktop/electron/git-review-ops.test.ts index 3d102c082d..9880b53ab3 100644 --- a/apps/desktop/electron/git-review-ops.test.ts +++ b/apps/desktop/electron/git-review-ops.test.ts @@ -6,7 +6,7 @@ import path from 'node:path' import { afterEach, test } from 'vitest' -import { gitFor, repoStatus, resolveRenamePath } from './git-review-ops' +import { gitFor, repoStatus, resolveRenamePath, REVIEW_FILE_CAP, reviewList } from './git-review-ops' const tempDirs: string[] = [] @@ -87,3 +87,33 @@ test('repoStatus reports an untracked directory without recursively listing its ['generated/'] ) }) + +test('reviewList reports an untracked directory without recursively listing its contents', async () => { + const dir = makeRepo() + const nested = path.join(dir, 'browser-profile', 'Default', 'Cache') + + fs.mkdirSync(nested, { recursive: true }) + + for (let i = 0; i < 20; i++) { + fs.writeFileSync(path.join(nested, `cache-${i}.bin`), 'generated\n') + } + + const result = await reviewList(dir, 'uncommitted', null, 'git') + + assert.deepEqual( + result.files.map(file => file.path), + ['browser-profile/'] + ) +}) + +test('reviewList caps the file payload returned to the renderer', async () => { + const dir = makeRepo() + + for (let i = 0; i < REVIEW_FILE_CAP + 10; i++) { + fs.writeFileSync(path.join(dir, `untracked-${String(i).padStart(4, '0')}.txt`), 'generated\n') + } + + const result = await reviewList(dir, 'uncommitted', null, 'git') + + assert.equal(result.files.length, REVIEW_FILE_CAP) +}) diff --git a/apps/desktop/electron/git-review-ops.ts b/apps/desktop/electron/git-review-ops.ts index 5fcb41fd0a..a44be7d3fe 100644 --- a/apps/desktop/electron/git-review-ops.ts +++ b/apps/desktop/electron/git-review-ops.ts @@ -14,6 +14,7 @@ import { resolveRequestedPathForIpc } from './hardening' const COMMIT_CONTEXT_DIFF_MAX_CHARS = 120_000 const COMMIT_CONTEXT_UNTRACKED_MAX = 80 +const REVIEW_FILE_CAP = 2_000 const UNTRACKED_LINE_COUNT_CONCURRENCY = 16 const UNTRACKED_LINE_COUNT_MAX_BYTES = 1024 * 1024 @@ -253,7 +254,7 @@ async function reviewList(repoPath, scope, baseRef, gitBin) { const range = scope === 'branch' ? `${base}...HEAD` : base const summary = await git.diffSummary([range]) - const files = summary.files.map(file => ({ + const files = summary.files.slice(0, REVIEW_FILE_CAP).map(file => ({ path: resolveRenamePath(file.file), added: 'insertions' in file ? file.insertions : 0, removed: 'deletions' in file ? file.deletions : 0, @@ -262,12 +263,22 @@ async function reviewList(repoPath, scope, baseRef, gitBin) { })) // "Last turn" also surfaces files created since the baseline (untracked). - if (scope === 'lastTurn') { - const status = await git.status() + if (scope === 'lastTurn' && files.length < REVIEW_FILE_CAP) { + // Keep untracked directories compact. A recursive status can produce + // hundreds of thousands of rows for browser profiles, generated + // artifacts, or dependency trees before the response reaches the + // renderer. + const status = await git.status(['--untracked-files=normal']) + const knownPaths = new Set(files.map(file => file.path)) for (const path of status.not_added) { - if (!files.some(f => f.path === path)) { + if (files.length >= REVIEW_FILE_CAP) { + break + } + + if (!knownPaths.has(path)) { files.push({ path, added: 0, removed: 0, status: '?', staged: false }) + knownPaths.add(path) } } } @@ -280,7 +291,10 @@ async function reviewList(repoPath, scope, baseRef, gitBin) { // Default: uncommitted (staged + unstaged + untracked), one row per path. const [status, staged, unstaged] = await Promise.all([ - git.status(), + // `normal` reports an untracked directory as one row instead of walking + // every descendant. The result is also capped before per-file stat/read + // work and before crossing the Electron IPC boundary. + git.status(['--untracked-files=normal']), git.diffSummary(['--cached']), git.diffSummary([]) ]) @@ -288,7 +302,7 @@ async function reviewList(repoPath, scope, baseRef, gitBin) { const stagedCounts = countsByPath(staged) const unstagedCounts = countsByPath(unstaged) - const files = status.files.map(file => { + const files = status.files.slice(0, REVIEW_FILE_CAP).map(file => { const filePath = resolveRenamePath(file.path) const sc = stagedCounts.get(filePath) || { added: 0, removed: 0 } const uc = unstagedCounts.get(filePath) || { added: 0, removed: 0 } @@ -696,6 +710,7 @@ export { gitFor, repoStatus, resolveRenamePath, + REVIEW_FILE_CAP, reviewCommit, reviewCommitContext, reviewCreatePr, diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 9df9c4be79..834a67ece3 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -35,7 +35,7 @@ import { classifyActiveRuntime } from './active-runtime-state' import { stopBackendChild as stopBackendChildImpl } from './backend-child' import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command' import { createBackendConnectionState } from './backend-connection-state' -import { buildDesktopBackendEnv, normalizeHermesHomeRoot } from './backend-env' +import { buildDesktopBackendEnv, hermesManagedNodePathEntries, normalizeHermesHomeRoot } from './backend-env' import { isReauthRequiredError, waitForHermesReady } from './backend-health' import { canImportHermesCli, @@ -47,6 +47,7 @@ import { import { waitForDashboardPortAnnouncement } from './backend-ready' import { shouldLatchBackendStartFailure, shouldLatchRemoteReauthFailure } from './backend-start-failure' import { detectRemoteDisplay, isWindowsBinaryPathInWsl, isWslEnvironment } from './bootstrap-platform' +import { decideBootstrapRepair } from './bootstrap-repair-guard' import { runBootstrap } from './bootstrap-runner' import { applyConnectionChange, resolveTerminalConnection } from './connection-apply' import { @@ -188,7 +189,7 @@ import { createStreamThrottle } from './stream-throttle' import { nativeOverlayWidth as computeNativeOverlayWidth, macTitleBarOverlayHeight } from './titlebar-overlay-width' import { resolveBehindCount, shouldCountCommits } from './update-count' import { waitForUpdateClearance } from './update-gate' -import { readLiveUpdateMarker, writeUpdateMarker } from './update-marker' +import { readLiveUpdateMarker, updateHandoffConflict, writeUpdateMarker } from './update-marker' import { runRebuildWithRetry } from './update-rebuild' import { buildRelaunchScript, @@ -200,9 +201,14 @@ import { sandboxPreflight } from './update-relaunch' import { isOfficialSshRemote, OFFICIAL_REPO_HTTPS_URL } from './update-remote' -import { spawnUpdaterProcess } from './updater-process' +import { + resolveStagedUpdaterBinary, + spawnUpdaterProcess, + stagedUpdaterSupportsPrewrittenMarker +} from './updater-process' import { formatBlockerMessage, formatProbeFailedMessage, scanVenvBlockers } from './venv-blocker-scan' import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace' +import { createWakeIndicatorWindowController } from './wake-indicator-window' import { computeWindowOptions, debounce, @@ -572,19 +578,10 @@ function resolveHermesHome() { const HERMES_HOME = resolveHermesHome() -function hermesManagedNodePathEntries() { - // NOTE: keep this ordering in sync with iter_hermes_node_dirs() in - // hermes_constants.py — this Node main process cannot import the Python - // module, so the platform-ordering rule is mirrored here. - const root = path.join(HERMES_HOME, 'node') - const bin = path.join(root, 'bin') - const entries = IS_WINDOWS ? [root, bin] : [bin, root] - - return entries.filter(directoryExists) -} - function pathWithHermesManagedNode(...entries) { - return [...hermesManagedNodePathEntries(), ...entries, process.env.PATH].filter(Boolean).join(path.delimiter) + const managed = hermesManagedNodePathEntries(HERMES_HOME).filter(directoryExists) + + return [...managed, ...entries, process.env.PATH].filter(Boolean).join(path.delimiter) } // ACTIVE_HERMES_ROOT — the canonical mutable Hermes install. Same path @@ -680,7 +677,15 @@ const WINDOW_BUTTON_POSITION = { // (pure + unit-testable); computeNativeOverlayWidth() applies it per platform. // It's only the pre-layout fallback — the renderer measures the exact overlay // width live via the Window Controls Overlay API. +// The apple-touch PNG bakes in the macOS-style ~10% margin, which is correct +// for the dock but renders visibly smaller than neighboring taskbar icons on +// Windows, where icons are full-bleed. Windows prefers the full-bleed +// assets/icon.ico (shipped to resources/ via extraResources) and only falls +// back to the padded PNG if the ico is missing. const APP_ICON_PATHS = [ + ...(IS_WINDOWS + ? [path.join(process.resourcesPath ?? '', 'icon.ico'), path.join(APP_ROOT, 'assets', 'icon.ico')] + : []), path.join(APP_ROOT, 'public', 'apple-touch-icon.png'), path.join(APP_ROOT, 'dist', 'apple-touch-icon.png'), path.join(unpackedPathFor(APP_ROOT), 'dist', 'apple-touch-icon.png') @@ -1099,6 +1104,14 @@ let bootstrapAbortController = null // repair can force the installer without destroying provenance about how the // install was created. Cleared once the reinstall is under way. let bootstrapRepairRequested = false +// Counter for in-flight repair attempts. Reset on a clean boot completion +// (see runBootstrap -> ensureRuntime resolve path). Each successive repair +// in the same failure episode increments this; once it crosses +// MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS the guard escalates from "soft restart" +// to "hard reinstall" so a transient backend stall (issue #74874) stops +// looping the user through a destructive venv reinstall. +let bootstrapRepairAttempt = 0 +const MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS = 3 let connectionConfigCache = null let connectionConfigCacheMtime = null const hermesLog = [] @@ -2602,18 +2615,14 @@ let isQuittingForHandoff = false let quitPromptOpen = false let quitConfirmedWithActiveWork = false -// Resolve the staged updater binary. The Tauri installer copies itself to -// HERMES_HOME/hermes-setup.exe on a successful install (see -// apps/bootstrap-installer paths::copy_self_to_hermes_home). That binary owns -// ALL repo mutation — running `hermes update` + rebuilding the desktop — so -// the desktop never touches its own bits while running. Returns null when the -// updater isn't staged (e.g. a dev/source run that never went through the -// installer); callers degrade gracefully. +// Resolve the staged updater binary the desktop may hand an update to. On +// Windows that binary owns ALL repo mutation — running `hermes update` + +// rebuilding the desktop — so the desktop never touches its own bits while +// running. macOS/Linux stage the same binary but deliberately do not use it; +// see resolveStagedUpdaterBinary for the policy and for #74836. Returns null +// whenever no hand-off applies; callers degrade gracefully. function resolveUpdaterBinary() { - const name = IS_WINDOWS ? 'hermes-setup.exe' : 'hermes-setup' - const candidate = path.join(HERMES_HOME, name) - - return fileExists(candidate) ? candidate : null + return resolveStagedUpdaterBinary(HERMES_HOME, { fileExists, isWindows: IS_WINDOWS }) } function repairMacUpdaterHelper(updater) { @@ -2841,12 +2850,13 @@ async function applyUpdates(opts = {}) { const updater = resolveUpdaterBinary() if (!updater && !IS_WINDOWS) { - // macOS/Linux drag-install: no staged Tauri hermes-setup. Unlike Windows - // (where a venv-shim file lock forces the quit→hand-off→rebuild dance), - // there's no mandatory file locking here, so the desktop can drive the - // whole update itself: `hermes update` (backend) + `hermes desktop - // --build-only` (OS-aware GUI rebuild), then swap the running .app bundle - // with the freshly built one and relaunch. + // macOS/Linux: never hand off, staged hermes-setup or not — the resolver + // returns null there by policy. Unlike Windows (where a venv-shim file + // lock forces the quit→hand-off→rebuild dance), there's no mandatory file + // locking here, so the desktop can drive the whole update itself: + // `hermes update` (backend) + `hermes desktop --build-only` (OS-aware GUI + // rebuild), then swap the running .app bundle with the freshly built one + // and relaunch. return await applyUpdatesPosixInApp(opts) } @@ -2884,6 +2894,19 @@ async function applyUpdates(opts = {}) { return { ok: true, manual: true, command, hermesRoot: updateRoot } } + const handoffConflict = updateHandoffConflict(HERMES_HOME) + + if (handoffConflict) { + // A different updater already owns the marker — most often a previous + // "Update" click whose updater is still alive and parked mid-run. + // Spawning another here would overwrite its claim and let two updaters + // mutate the checkout at once (#75778); refuse instead. + rememberLog(`[updates] refusing hand-off: ${handoffConflict.message}`) + emitUpdateProgress({ stage: 'error', message: handoffConflict.message, percent: null }) + + return { ok: false, error: 'update-already-running', message: handoffConflict.message } + } + emitUpdateProgress({ stage: 'restart', message: @@ -2984,8 +3007,20 @@ async function applyUpdates(opts = {}) { // the venv. By writing the marker ourselves the renderer's // waitForUpdateToFinish() gate sees a live update and parks instead. // The updater overwrites this with its own PID later; same format. - if (Number.isInteger(child.pid)) { + // + // SKIPPED for pre-#74782 staged updaters: those have no self-PID + // exclusion, so they read this very marker as a foreign live owner and + // abort with "Another Hermes update is already running (PID )" — + // an unbreakable loop, because the update that would replace the stale + // binary is the one being refused. Losing the anti-respawn hardening is + // strictly better than never updating again, and the updater still writes + // its own marker moments later. + if (Number.isInteger(child.pid) && stagedUpdaterSupportsPrewrittenMarker(updater)) { writeUpdateMarker(HERMES_HOME, child.pid) + } else if (Number.isInteger(child.pid)) { + rememberLog( + `[updates] skipping marker pre-write: staged updater predates self-adopt (${updater}); it would refuse its own claim` + ) } rememberLog(`[updates] launched updater: ${updater} ${updaterArgs.join(' ')}; exiting desktop to release venv shim`) @@ -3017,6 +3052,24 @@ async function handOffWindowsBootstrapRecovery(reason) { return false } + const handoffConflict = updateHandoffConflict(HERMES_HOME) + + if (handoffConflict) { + // Same hazard as applyUpdates (#75778): a live foreign updater already + // owns the marker. Spawning another here would overwrite its claim and + // race a second updater over the same install tree. The live updater + // is already working on this exact install and will restart us when + // it finishes, so treat this the same as a successful hand-off instead + // of clobbering it with our own. + rememberLog(`[bootstrap] refusing recovery hand-off: ${handoffConflict.message}`) + isQuittingForHandoff = true + setTimeout(() => { + app.quit() + }, UPDATE_HANDOFF_DWELL_MS) + + return true + } + const updateRoot = resolveUpdateRoot() const { branch: configuredBranch } = readDesktopUpdateConfig() @@ -3054,9 +3107,15 @@ async function handOffWindowsBootstrapRecovery(reason) { // Same marker pre-write as applyUpdates — see comment there. The recovery // hand-off has the same window where the renderer can respawn a backend - // before the updater writes its own marker. - if (Number.isInteger(child.pid)) { + // before the updater writes its own marker, and the same stale-updater + // exclusion: a pre-#74782 binary would refuse its own pre-written claim and + // strand the very recovery meant to heal the install. + if (Number.isInteger(child.pid) && stagedUpdaterSupportsPrewrittenMarker(updater)) { writeUpdateMarker(HERMES_HOME, child.pid) + } else if (Number.isInteger(child.pid)) { + rememberLog( + `[bootstrap] skipping marker pre-write: staged updater predates self-adopt (${updater}); it would refuse its own claim` + ) } rememberLog( @@ -4040,6 +4099,7 @@ async function ensureRuntime(backend) { // The repair request has been honoured by reaching the installer; clear it // so a later boot isn't forced through bootstrap again. bootstrapRepairRequested = false + bootstrapRepairAttempt = 0 const bootstrapResult = await runBootstrap({ installStamp: backend.installStamp, @@ -6949,6 +7009,7 @@ async function sanitizeDesktopConnectionConfig(config = readDesktopConnectionCon sshPort: (ssh || savedSsh)?.port || null, sshKeyPath: (ssh || savedSsh)?.keyPath || '', sshRemoteHermesPath: (ssh || savedSsh)?.remoteHermesPath || '', + sshRemoteProfile: (ssh || savedSsh)?.remoteProfile || '', // The env override only forces the global/primary connection; a per-profile // scope is never overridden by HERMES_DESKTOP_REMOTE_URL. envOverride @@ -7081,7 +7142,8 @@ function buildSshBlock(input: any, existingBlock: any = {}) { user: input.sshUser ?? existingBlock.user, port: input.sshPort ?? existingBlock.port, keyPath: input.sshKeyPath ?? existingBlock.keyPath, - remoteHermesPath: input.sshRemoteHermesPath ?? existingBlock.remoteHermesPath + remoteHermesPath: input.sshRemoteHermesPath ?? existingBlock.remoteHermesPath, + remoteProfile: input.sshRemoteProfile ?? existingBlock.remoteProfile }) if (!merged) { @@ -7370,7 +7432,7 @@ async function bootstrapSshConnectionInner(profile, sshConfig, reuseToken, sourc const lifecycle = platform.os === 'Windows' ? connectWindowsRemote : remoteLifecycle.connect result = await lifecycle({ ssh, - profile: connectionScopeKey(profile) || '', + profile: sshConfig.remoteProfile || connectionScopeKey(profile) || '', remoteHermesPath: sshConfig.remoteHermesPath || '', ownershipId: sshOwnershipKey(profile), reuseToken: reuseToken || '', @@ -8550,6 +8612,13 @@ async function startHermes() { error: null }) + // A successful boot (including a soft restart that the repair-guard + // chose over a hard reinstall, see #74874) means any in-flight repair + // attempt counter has been honoured — reset it so the next genuine + // failure starts fresh from attempt 1 instead of inheriting the + // accumulated count of the resolved episode. + bootstrapRepairAttempt = 0 + return { baseUrl, mode: 'local', @@ -8804,6 +8873,17 @@ function createInstanceWindow() { return win } +// A macOS-only ambient wake cue. It is deliberately a gateway-less helper +// window: the active renderer owns voice state and sends only the visual phase. +const wakeIndicatorController = createWakeIndicatorWindowController({ + devServer: DEV_SERVER, + isMac: IS_MAC, + loadWindowUrl, + preloadPath: PRELOAD_PATH, + rendererIndex: resolveRendererIndex, + wireWindow: window => wireCommonWindowHandlers(window, zoomWiringForWindowKind('wakeIndicator')) +}) + // The pet overlay: a single transparent, frameless, always-on-top window that // hosts ONLY the floating mascot. Shift-clicking the in-window pet "pops it out" // here so it can leave the app's bounds and stay visible while Hermes is @@ -9264,6 +9344,7 @@ function createWindow() { const createdMainWindow = mainWindow mainWindow.on('closed', () => { closePetOverlay() + wakeIndicatorController.close() if (mainWindow === createdMainWindow) { mainWindow = null @@ -9464,6 +9545,10 @@ ipcMain.handle('hermes:window:openInstance', async () => { return { ok: true } }) +ipcMain.handle('hermes:wake-indicator:get', () => wakeIndicatorController.getState()) +ipcMain.on('hermes:wake-indicator:set', (_event, state) => { + wakeIndicatorController.setState(state) +}) // --- Text size (zoom) ------------------------------------------------------- // The settings UI drives the same clamped zoom scale as the Ctrl/Cmd @@ -9631,9 +9716,41 @@ ipcMain.handle('hermes:bootstrap:repair', async () => { // transient backend errors on a perfectly healthy install, and deleting the // marker in that case stranded the app in first-run setup with no way back // (#72166). The explicit flag carries the intent instead. - rememberLog('[bootstrap] repair requested by renderer; forcing reinstall + clearing latched failure') + bootstrapRepairAttempt += 1 - bootstrapRepairRequested = true + // Probe the live backend process so the guard can distinguish "venv is + // genuinely broken" (force reinstall) from "backend is just transiently + // stalled under GIL pressure" (#74874 — `event loop stalled` followed by + // `ws ready frame send failed`, then renderer keeps reporting dead). + const primaryProc = backendConnectionState.getProcess() + + const primaryBackendAlive = Boolean( + primaryProc && + (primaryProc as { exitCode?: number | null }).exitCode === null && + (primaryProc as { signalCode?: string | null }).signalCode === null + ) + + const repairDecision = decideBootstrapRepair({ + attempt: bootstrapRepairAttempt, + maxSoftAttempts: MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS, + primaryBackendAlive + }) + + rememberLog( + `[bootstrap] repair requested by renderer; forcing reinstall + clearing latched failure ` + + `(attempt=${repairDecision.attempt}/${MAX_BOOTSTRAP_REPAIR_SOFT_ATTEMPTS}, ` + + `primaryBackendAlive=${primaryBackendAlive}, ` + + `hardReinstall=${repairDecision.hardReinstall}): ${repairDecision.reason}` + ) + + // The guard may decide the install is healthy enough that a restart + // (without touching the venv) is the right answer. Translate that into + // the existing flag: if the guard said "soft restart", we skip the + // "bypass active runtime" path inside startHermes() and fall through + // to the normal restart branch, which just kills the current child + // and respawns it against the same venv. See #74874 — this is what + // breaks the infinite reinstall loop the user hit. + bootstrapRepairRequested = repairDecision.hardReinstall bootstrapFailure = null backendStartFailure = null remoteReauthFailure = null @@ -11711,6 +11828,17 @@ app.whenReady().then(() => { // it without the renderer visiting Settings. A failed registration is logged // here and surfaced in Settings via the IPC state (never silent). applyQuickEntrySettings(readQuickEntrySettings()) + + if (IS_MAC) { + const reposition = () => wakeIndicatorController.reposition() + + screen.on('display-added', reposition) + + screen.on('display-metrics-changed', reposition) + + screen.on('display-removed', reposition) + } + createWindow() // Win/Linux cold start: the launching hermes:// URL is in our own argv. @@ -11839,6 +11967,7 @@ app.on('before-quit', event => { // The always-on-top overlay isn't a "real" app window; close it so a stray // pet can't keep the process alive or float over a quit app. closePetOverlay() + wakeIndicatorController.close() // Same for the Quick Entry composer — and release its global accelerator so a // quitting Hermes never keeps another app's chord hostage. diff --git a/apps/desktop/electron/native-oauth.test.ts b/apps/desktop/electron/native-oauth.test.ts index bf58ca996e..bc341f33ef 100644 --- a/apps/desktop/electron/native-oauth.test.ts +++ b/apps/desktop/electron/native-oauth.test.ts @@ -207,10 +207,7 @@ test('parseTokenResponse cannot read a persisted set (the reload bug #73271)', ( }) test('parseStoredTokenSet rejects a non-normalized server response', () => { - assert.throws( - () => parseStoredTokenSet({ access_token: 'AT-server' }), - /missing accessToken/i - ) + assert.throws(() => parseStoredTokenSet({ access_token: 'AT-server' }), /missing accessToken/i) }) // --- refresh timing --- diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index 8fb6d97c3f..11483d9ce8 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -8,6 +8,16 @@ contextBridge.exposeInMainWorld('hermesDesktop', { openSessionWindow: (sessionId, opts) => ipcRenderer.invoke('hermes:window:openSession', sessionId, opts), openWindow: () => ipcRenderer.invoke('hermes:window:openInstance'), claimAmbientCue: key => ipcRenderer.invoke('hermes:ambient:claim', key), + wakeIndicator: { + getState: () => ipcRenderer.invoke('hermes:wake-indicator:get'), + setState: state => ipcRenderer.send('hermes:wake-indicator:set', state), + onState: callback => { + const listener = (_event, state) => callback(state) + ipcRenderer.on('hermes:wake-indicator:state', listener) + + return () => ipcRenderer.removeListener('hermes:wake-indicator:state', listener) + } + }, petOverlay: { // Main renderer → main process: window lifecycle + drag. `request` is // `{ bounds, screen }`; resolves with the screen bounds it actually used. diff --git a/apps/desktop/electron/primary-backend-startup.ts b/apps/desktop/electron/primary-backend-startup.ts index edbe5167b0..b6e2ba8718 100644 --- a/apps/desktop/electron/primary-backend-startup.ts +++ b/apps/desktop/electron/primary-backend-startup.ts @@ -10,8 +10,7 @@ export interface PrimaryBackendStartupOptions = - | { kind: 'local'; backend: RuntimeBackend } - | { kind: 'remote'; connection: Connection } + { kind: 'local'; backend: RuntimeBackend } | { kind: 'remote'; connection: Connection } export class FirstRunSetupResetError extends Error { readonly firstRunSetupReset = true diff --git a/apps/desktop/electron/quick-entry.ts b/apps/desktop/electron/quick-entry.ts index db2975696b..d8233c217e 100644 --- a/apps/desktop/electron/quick-entry.ts +++ b/apps/desktop/electron/quick-entry.ts @@ -111,12 +111,7 @@ const ACCELERATOR_PUNCTUATION = new Set([ /** Why a shortcut string was rejected. The renderer maps these to copy. */ export type QuickEntryShortcutError = - | 'empty' - | 'invalid-key' - | 'invalid-modifier' - | 'no-key' - | 'no-modifier' - | 'reserved' + 'empty' | 'invalid-key' | 'invalid-modifier' | 'no-key' | 'no-modifier' | 'reserved' export type QuickEntryShortcutParse = { ok: false; reason: QuickEntryShortcutError } | { accelerator: string; ok: true } diff --git a/apps/desktop/electron/remote-lifecycle.test.ts b/apps/desktop/electron/remote-lifecycle.test.ts index b421b0ef0c..3855cdab22 100644 --- a/apps/desktop/electron/remote-lifecycle.test.ts +++ b/apps/desktop/electron/remote-lifecycle.test.ts @@ -2,6 +2,7 @@ import assert from 'node:assert/strict' import { test } from 'vitest' +import { profileSshOverride } from './connection-config' import { buildSpawnCommand, cleanupStale, @@ -420,6 +421,51 @@ test('connect() spawns fresh when there is no lockfile, adopts the served token' assert.equal(result.tokenFingerprint, fingerprintToken('the-served-token')) }) +test('managed SSH maps a local scope to a different non-default remote profile', async () => { + const localScope = 'work' + + const sshConfig = profileSshOverride( + { + profiles: { + [localScope]: { + mode: 'ssh', + host: 'remote-box', + remoteProfile: 'writer_2' + } + } + }, + localScope + ) + + assert.equal(sshConfig?.remoteProfile, 'writer_2') + + const ssh = fakeSsh([ + [/uname/, 'Linux\nx86_64'], + [/\[ -x/, 'OK'], + [/cat .*lock\.json/, ''], + [/grep -q ssh-session-token-file/, 'YES\n'], + [/python3 -c/, ''], + [/printf '%s\\n'/, ''], + [/setsid/, '778\n'], + [/kill -0 778/, 'ALIVE'], + [/cat .*\.log/, 'HERMES_BACKEND_READY port=52000\n'] + ]) + + await connect( + connectDeps(ssh, { + profile: sshConfig?.remoteProfile, + adoptServedToken: async () => 'mapped-profile-token' + }) + ) + + const spawn = ssh.calls.find(command => /setsid|nohup/.test(command)) || '' + assert.match(spawn, /--profile\b/) + assert.ok(spawn.includes('writer_2')) + assert.match(spawn, /serve\s+--isolated/) + assert.match(spawn, /\.hermes\/desktop-ssh\/[0-9a-f]{32}\/[0-9a-f]{16}\.token/) + assert.ok(!spawn.includes(' work'), 'the local Desktop scope must not become the remote profile') +}) + test('connect() reuses a healthy dashboard when fingerprint + probe pass', async () => { const reuseToken = 'stored-token' const lock = ownedLock({ tokenFingerprint: fingerprintToken(reuseToken) }) @@ -440,6 +486,36 @@ test('connect() reuses a healthy dashboard when fingerprint + probe pass', async assert.ok(!ssh.calls.some(c => /setsid/.test(c)), 'reuse path must not spawn a new dashboard') }) +test('connect() respawns when the requested remote profile differs from the lockfile profile', async () => { + const reuseToken = 'stored-token' + const lock = ownedLock({ profile: 'desktop-work', tokenFingerprint: fingerprintToken(reuseToken) }) + + const ssh = fakeSsh([ + [/uname/, 'Linux\nx86_64'], + [/\[ -x/, 'OK'], + [/cat .*lock\.json/, JSON.stringify(lock)], + [/kill -0 333/, 'ALIVE'], + [/print\("OWNED"/, 'OWNED\n'], + [/kill 333/, ''], + [/--version/, 'Hermes Agent v0.18.2\n'], + [/grep -q ssh-session-token-file/, 'YES\n'], + [/python3 -c/, ''], + [/setsid/, '890\n'], + [/kill -0 890/, 'ALIVE'], + [/cat .*\.log/, 'HERMES_DASHBOARD_READY port=52050\n'] + ]) + + const result = await connect( + connectDeps(ssh, { profile: 'default', reuseToken, adoptServedToken: async () => 'fresh' }) + ) + + assert.equal(result.reused, false) + assert.ok( + ssh.calls.some(c => /setsid/.test(c)), + 'profile mismatch must spawn a fresh dashboard' + ) +}) + test('connect() respawns when the lockfile hermesPath differs from the resolved path', async () => { const reuseToken = 'stored-token' const lock = ownedLock({ hermesPath: '/old/stale/hermes', tokenFingerprint: fingerprintToken(reuseToken) }) diff --git a/apps/desktop/electron/remote-lifecycle.ts b/apps/desktop/electron/remote-lifecycle.ts index 705c80000b..db4d5b20dd 100644 --- a/apps/desktop/electron/remote-lifecycle.ts +++ b/apps/desktop/electron/remote-lifecycle.ts @@ -713,6 +713,7 @@ async function connect(deps) { pidAlive && owned && lock.port > 0 && + lock.profile === profile && Boolean(reuseToken) && lock.tokenFingerprint === fingerprintToken(reuseToken) && lock.hermesPath === hermesPath && diff --git a/apps/desktop/electron/ssh-bootstrap-coordinator.test.ts b/apps/desktop/electron/ssh-bootstrap-coordinator.test.ts index cf5c24e972..d3ffb99e0e 100644 --- a/apps/desktop/electron/ssh-bootstrap-coordinator.test.ts +++ b/apps/desktop/electron/ssh-bootstrap-coordinator.test.ts @@ -28,6 +28,7 @@ test('sshConfigFingerprint covers scope and every connection field', () => { port: 2222, keyPath: '/other', remoteHermesPath: '/other-hermes', + remoteProfile: 'default', effectiveConfigFingerprint: 'changed-config' })) { assert.notEqual(base, sshConfigFingerprint('', { ...config, [field]: value })) diff --git a/apps/desktop/electron/ssh-bootstrap-coordinator.ts b/apps/desktop/electron/ssh-bootstrap-coordinator.ts index 827e9a4e7a..c38e0473aa 100644 --- a/apps/desktop/electron/ssh-bootstrap-coordinator.ts +++ b/apps/desktop/electron/ssh-bootstrap-coordinator.ts @@ -8,6 +8,7 @@ function sshConfigFingerprint(scope, config) { config.port, config.keyPath, config.remoteHermesPath, + config.remoteProfile, config.effectiveConfigFingerprint ] diff --git a/apps/desktop/electron/update-marker.test.ts b/apps/desktop/electron/update-marker.test.ts index 3da49396cc..0fb1142b4c 100644 --- a/apps/desktop/electron/update-marker.test.ts +++ b/apps/desktop/electron/update-marker.test.ts @@ -24,6 +24,7 @@ import { markerPath, readLiveUpdateMarker, UPDATE_MARKER_MAX_AGE_MS, + updateHandoffConflict, writeUpdateMarker } from './update-marker' @@ -128,3 +129,51 @@ test('writeUpdateMarker + dead pid => self-heals on read', () => { assert.equal(res, null, 'a dead-pid marker from writeUpdateMarker self-heals') assert.ok(!fs.existsSync(markerPath(home)), 'marker file is pruned') }) + +// --------------------------------------------------------------------------- +// updateHandoffConflict (#75778) +// +// A retried "Update" click must not spawn a second updater over a still-live +// one — writeUpdateMarker unconditionally overwrites the marker, so an +// unchecked hand-off clobbers the original updater's claim while it is still +// alive and mutating the checkout. +// --------------------------------------------------------------------------- + +test('no marker => hand-off is not blocked', () => { + const home = tmpHome('conflict-none') + assert.equal(updateHandoffConflict(home, { kill: ALIVE }), null) +}) + +test('a different live updater already owns the marker => hand-off is blocked', () => { + const home = tmpHome('conflict-live') + const now = 1_000_000_000_000 + writeMarker(home, 1010, Math.floor(now / 1000) - 6) // 6s old + const conflict = updateHandoffConflict(home, { kill: ALIVE, now: () => now }) + assert.ok(conflict, 'a live foreign updater must block a new hand-off') + assert.equal(conflict.pid, 1010) + assert.match(conflict.message, /already running/) + assert.match(conflict.message, /PID 1010/) + assert.match(conflict.message, /6s/) +}) + +test('a dead-pid marker does not block a hand-off (self-heals)', () => { + const home = tmpHome('conflict-dead') + writeMarker(home, 999999, Math.floor(Date.now() / 1000)) + assert.equal(updateHandoffConflict(home, { kill: DEAD }), null) +}) + +test('an expired marker does not block a hand-off (self-heals)', () => { + const home = tmpHome('conflict-expired') + const now = 1_000_000_000_000 + writeMarker(home, 1010, Math.floor((now - UPDATE_MARKER_MAX_AGE_MS - 60_000) / 1000)) + assert.equal(updateHandoffConflict(home, { kill: ALIVE, now: () => now }), null) +}) + +test('minutes-scale elapsed time is formatted as "Nm Ss"', () => { + const home = tmpHome('conflict-minutes') + const now = 1_000_000_000_000 + writeMarker(home, 1010, Math.floor(now / 1000) - 125) // 2m 5s old + const conflict = updateHandoffConflict(home, { kill: ALIVE, now: () => now }) + assert.ok(conflict) + assert.match(conflict.message, /2m 5s/) +}) diff --git a/apps/desktop/electron/update-marker.ts b/apps/desktop/electron/update-marker.ts index 543fce0451..ee50b52e21 100644 --- a/apps/desktop/electron/update-marker.ts +++ b/apps/desktop/electron/update-marker.ts @@ -136,3 +136,47 @@ export function writeUpdateMarker(hermesHome, pid, { now = Date.now } = {}) { // updater will write its own when it reaches run_update. } } + +/** + * Whether a NEW updater hand-off must be refused because a different, + * already-alive updater currently owns the marker (#75778). + * + * `writeUpdateMarker` unconditionally overwrites the marker file. Called + * before every hand-off with no conflict check, a user who clicks "Update" + * again while a prior updater is still parked mid-run (e.g. "waiting for + * Hermes to exit…") clobbers that still-running updater's claim: the + * retry's pre-write now names the NEW child, so the OLD process — alive + * and mutating the checkout — is no longer recorded as the owner. A second + * live updater can then run over the same tree unrecorded, the exact + * two-updaters-at-once hazard `UpdateMarkerGuard` in the Rust updater + * exists to prevent (apps/bootstrap-installer/src-tauri/src/update.rs). + * + * Returns the live foreign owner (with a ready-to-show message) when the + * hand-off must be refused, or `null` when it's safe to spawn — no marker, + * or the existing one is stale/dead and self-heals via + * `readLiveUpdateMarker`. + */ +export function updateHandoffConflict( + hermesHome, + opts: { + now?: () => number + maxAgeMs?: number + kill?: typeof process.kill + } = {} +) { + const owner = readLiveUpdateMarker(hermesHome, opts) + + if (!owner) { + return null + } + + const mins = Math.floor(owner.ageMs / 60_000) + const secs = Math.floor((owner.ageMs % 60_000) / 1000) + const elapsed = mins > 0 ? `${mins}m ${secs}s` : `${secs}s` + + return { + pid: owner.pid, + ageMs: owner.ageMs, + message: `An update is already running (PID ${owner.pid}, started ${elapsed} ago). Wait for it to finish, then try again.` + } +} diff --git a/apps/desktop/electron/updater-process.test.ts b/apps/desktop/electron/updater-process.test.ts index ce61855dff..e781fe3efa 100644 --- a/apps/desktop/electron/updater-process.test.ts +++ b/apps/desktop/electron/updater-process.test.ts @@ -1,9 +1,68 @@ import assert from 'node:assert/strict' import type { SpawnOptions } from 'node:child_process' +import path from 'node:path' import { test } from 'vitest' -import { spawnUpdaterProcess } from './updater-process' +import { + MARKER_SELF_ADOPT_EPOCH_MS, + resolveStagedUpdaterBinary, + spawnUpdaterProcess, + stagedUpdaterSupportsPrewrittenMarker +} from './updater-process' + +const DAY_MS = 24 * 60 * 60 * 1000 + +test('stagedUpdaterSupportsPrewrittenMarker rejects installers predating the self-adopt fix', () => { + // The real-world trap: an installer staged at first install months ago, never + // refreshed because copy_self_to_hermes_home no-ops during --update. + assert.equal( + stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', { + stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS - 60 * DAY_MS + }), + false + ) +}) + +test('stagedUpdaterSupportsPrewrittenMarker accepts installers from the fix onward', () => { + assert.equal( + stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', { + stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS + }), + true + ) + assert.equal( + stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', { + stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS + 30 * DAY_MS + }), + true + ) +}) + +test('stagedUpdaterSupportsPrewrittenMarker treats an unreadable mtime as unsupported', () => { + // Bias toward the path that can always make progress: a skipped pre-write + // loses anti-respawn hardening, a wedged updater can never update again. + assert.equal( + stagedUpdaterSupportsPrewrittenMarker('C:\\Hermes\\hermes-setup.exe', { + stagedMtimeMs: () => null + }), + false + ) +}) + +test('resolveStagedUpdaterBinary still returns a stale staged updater on Windows', () => { + // Staleness gates only the marker PRE-WRITE, never the hand-off itself: + // the stale binary is the only updater these users have, and it works fine + // once it is allowed to write its own claim. + assert.equal( + resolveStagedUpdaterBinary('C:\\Hermes', { + fileExists: () => true, + isWindows: true, + stagedMtimeMs: () => MARKER_SELF_ADOPT_EPOCH_MS - 60 * DAY_MS + }), + path.join('C:\\Hermes', 'hermes-setup.exe') + ) +}) test('spawnUpdaterProcess hides the updater console and detaches the child on Windows', () => { const calls: Array<{ args: string[]; command: string; options: SpawnOptions }> = [] @@ -60,3 +119,49 @@ test('spawnUpdaterProcess preserves updater options off Windows', () => { assert.deepEqual(capturedOptions, { detached: true, stdio: 'ignore' }) }) + +test('resolveStagedUpdaterBinary hands Windows the staged installer it finds', () => { + const home = 'C:\\Users\\hermes\\AppData\\Local\\hermes' + const staged = path.join(home, 'hermes-setup.exe') + const probed: string[] = [] + + const resolved = resolveStagedUpdaterBinary(home, { + fileExists: candidate => { + probed.push(candidate) + + return candidate === staged + }, + isWindows: true + }) + + assert.equal(resolved, staged) + assert.deepEqual(probed, [staged]) +}) + +test('resolveStagedUpdaterBinary returns null off Windows even when hermes-setup is staged (#74836)', () => { + const home = '/Users/hermes/.hermes' + let probes = 0 + + const resolved = resolveStagedUpdaterBinary(home, { + // The installer stages hermes-setup on macOS/Linux too, so "it exists" is + // the normal case — and precisely the one that must not win. + fileExists: () => { + probes += 1 + + return true + }, + isWindows: false + }) + + assert.equal(resolved, null) + assert.equal(probes, 0) +}) + +test('resolveStagedUpdaterBinary returns null on Windows when nothing is staged', () => { + const resolved = resolveStagedUpdaterBinary('C:\\Users\\hermes\\AppData\\Local\\hermes', { + fileExists: () => false, + isWindows: true + }) + + assert.equal(resolved, null) +}) diff --git a/apps/desktop/electron/updater-process.ts b/apps/desktop/electron/updater-process.ts index d0706a6e51..97b1d9651a 100644 --- a/apps/desktop/electron/updater-process.ts +++ b/apps/desktop/electron/updater-process.ts @@ -1,4 +1,6 @@ import { spawn, type SpawnOptions } from 'node:child_process' +import { statSync } from 'node:fs' +import path from 'node:path' import { hiddenWindowsChildOptions } from './windows-child-options' @@ -7,6 +9,109 @@ export interface UpdaterChild { unref: () => void } +export interface ResolveStagedUpdaterBinaryDeps { + isWindows?: boolean + fileExists?: (candidate: string) => boolean + stagedMtimeMs?: (candidate: string) => number | null +} + +/** + * Staged installers older than this have no self-PID exclusion in + * `UpdateMarkerGuard::acquire` and will refuse an update whose marker was + * pre-written on their behalf. + * + * The self-adopt fix landed in #74782 / 160586ff8 (2026-07-30 17:57 +0700). + * We compare against the start of 2026-07-31 UTC so the boundary is + * unambiguous for binaries staged that same day. + */ +export const MARKER_SELF_ADOPT_EPOCH_MS = Date.UTC(2026, 6, 31) + +function stagedFileExists(candidate: string): boolean { + try { + return statSync(candidate).isFile() + } catch { + return false + } +} + +function stagedFileMtimeMs(candidate: string): number | null { + try { + return statSync(candidate).mtimeMs + } catch { + return null + } +} + +/** + * Decide which staged installer binary — if any — may be handed an update. + * + * The Tauri installer self-copies into HERMES_HOME on *every* platform + * (`hermes-setup.exe` on Windows, `hermes-setup` elsewhere — see + * apps/bootstrap-installer `paths::installer_dest` and + * `bootstrap::copy_self_to_hermes_home`), so finding that binary on macOS or + * Linux is expected, not leftover junk. + * + * Handing an update to it is nonetheless a Windows-only policy. Windows needs + * the quit -> hand-off -> rebuild dance because a venv shim file lock keeps the + * running desktop from rewriting its own bits; macOS and Linux have no such + * lock and update in place through applyUpdatesPosixInApp(). Off Windows the + * hand-off therefore buys nothing and costs a great deal: a staged binary older + * than the hand-off protocol holds the update marker, spawns `hermes update`, + * and that child refuses its own parent — wedging the in-app Update button for + * good, with no route (update, re-download, reinstall) to a newer binary + * (#74836). Returning null off Windows is what routes those platforms to the + * in-app updater. + * + * Null on Windows too when nothing is staged (a dev/source run, or a CLI + * install that never went through the installer); callers degrade gracefully. + */ +export function resolveStagedUpdaterBinary( + hermesHome: string, + deps: ResolveStagedUpdaterBinaryDeps = {} +): string | null { + const isWindows = deps.isWindows ?? process.platform === 'win32' + + if (!isWindows) { + return null + } + + const fileExists = deps.fileExists ?? stagedFileExists + const candidate = path.join(hermesHome, 'hermes-setup.exe') + + return fileExists(candidate) ? candidate : null +} + +/** + * True when the staged installer is new enough to survive a pre-written marker. + * + * `copy_self_to_hermes_home` deliberately no-ops during `--update` + * (apps/bootstrap-installer/src-tauri/src/paths.rs), so the binary staged by a + * user's ORIGINAL install orchestrates every later update — forever. Installers + * predating #74782 have no self-PID exclusion in `UpdateMarkerGuard::acquire`, + * so when the desktop pre-writes the marker naming that very updater, the + * updater reads its own claim as a foreign live owner and aborts with + * "Another Hermes update is already running (PID , started 1s ago)" — + * the observed infinite "Install didn't finish" loop. Skipping the pre-write + * for those binaries lets them acquire cleanly and run `hermes update`, which + * pulls the permanent fixes. See shouldPrewriteUpdateMarker. + * + * We cannot ask the binary its version without executing it, so use its mtime: + * the installer is written to HERMES_HOME at install/repair time, making mtime + * a faithful stamp of which installer generation produced it. + * + * Unreadable mtime counts as UNSUPPORTED — the pre-write is a best-effort + * hardening, while a wedged updater is unrecoverable, so we bias toward the + * path that can always make progress. + */ +export function stagedUpdaterSupportsPrewrittenMarker( + candidate: string, + deps: ResolveStagedUpdaterBinaryDeps = {} +): boolean { + const mtimeMs = (deps.stagedMtimeMs ?? stagedFileMtimeMs)(candidate) + + return typeof mtimeMs === 'number' && Number.isFinite(mtimeMs) && mtimeMs >= MARKER_SELF_ADOPT_EPOCH_MS +} + export interface SpawnUpdaterProcessDeps { isWindows?: boolean spawnProcess?: (command: string, args: string[], options: SpawnOptions) => UpdaterChild diff --git a/apps/desktop/electron/wake-indicator-window.ts b/apps/desktop/electron/wake-indicator-window.ts new file mode 100644 index 0000000000..7ef676a43f --- /dev/null +++ b/apps/desktop/electron/wake-indicator-window.ts @@ -0,0 +1,174 @@ +import { pathToFileURL } from 'node:url' + +import { BrowserWindow, screen } from 'electron' + +import { + normalizeWakeIndicatorState, + selectWakeIndicatorDisplay, + WAKE_INDICATOR_FADE_MS, + type WakeIndicatorState, + wakeIndicatorWindowBounds +} from './wake-indicator' + +interface WakeIndicatorWindowOptions { + devServer?: string + isMac: boolean + loadWindowUrl: (window: BrowserWindow, url: string, label: string) => void + preloadPath: string + rendererIndex: () => string + wireWindow: (window: BrowserWindow) => void +} + +export function createWakeIndicatorWindowController({ + devServer, + isMac, + loadWindowUrl, + preloadPath, + rendererIndex, + wireWindow +}: WakeIndicatorWindowOptions) { + let hideTimer: NodeJS.Timeout | null = null + let state: WakeIndicatorState = 'hidden' + let window: BrowserWindow | null = null + + const url = () => { + if (devServer) { + return `${devServer.endsWith('/') ? devServer.slice(0, -1) : devServer}/?win=wake#/` + } + + return `${pathToFileURL(rendererIndex()).toString()}?win=wake#/` + } + + const selectedDisplay = () => selectWakeIndicatorDisplay(screen.getAllDisplays(), screen.getPrimaryDisplay()) + + const reposition = () => { + if (!window || window.isDestroyed()) { + return + } + + window.setBounds(wakeIndicatorWindowBounds(selectedDisplay())) + } + + const sendState = () => { + if (!window || window.isDestroyed()) { + return + } + + window.webContents.send('hermes:wake-indicator:state', state) + } + + const spawn = () => { + const next = new BrowserWindow({ + ...wakeIndicatorWindowBounds(selectedDisplay()), + alwaysOnTop: true, + backgroundColor: '#00000000', + focusable: false, + frame: false, + fullscreenable: false, + hasShadow: false, + hiddenInMissionControl: true, + maximizable: false, + minimizable: false, + movable: false, + resizable: false, + show: false, + skipTaskbar: false, + transparent: true, + type: 'panel', + webPreferences: { + backgroundThrottling: false, + contextIsolation: true, + devTools: true, + nodeIntegration: false, + preload: preloadPath, + sandbox: true + } + }) + + next.setAlwaysOnTop(true, 'floating') + next.setHiddenInMissionControl?.(true) + next.setIgnoreMouseEvents(true, { forward: true }) + + try { + next.setVisibleOnAllWorkspaces(true, { + skipTransformProcessType: true, + visibleOnFullScreen: true + }) + } catch { + // Best effort on older Electron/macOS combinations. + } + + wireWindow(next) + + next.webContents.on('did-finish-load', sendState) + next.once('ready-to-show', () => { + if (!next.isDestroyed() && state !== 'hidden') { + next.showInactive() + } + }) + next.on('closed', () => { + if (window === next) { + window = null + } + }) + + loadWindowUrl(next, url(), 'Wake indicator') + + return next + } + + const setState = (value: unknown) => { + if (!isMac) { + return + } + + state = normalizeWakeIndicatorState(value) + + if (hideTimer) { + clearTimeout(hideTimer) + hideTimer = null + } + + if (state === 'hidden') { + sendState() + hideTimer = setTimeout(() => { + hideTimer = null + + if (state === 'hidden' && window && !window.isDestroyed()) { + window.hide() + } + }, WAKE_INDICATOR_FADE_MS) + + return + } + + if (!window || window.isDestroyed()) { + window = spawn() + } else { + reposition() + sendState() + window.showInactive() + } + } + + const close = () => { + if (hideTimer) { + clearTimeout(hideTimer) + hideTimer = null + } + + if (window && !window.isDestroyed()) { + window.close() + } + + window = null + state = 'hidden' + } + + return { + close, + getState: () => state, + reposition, + setState + } +} diff --git a/apps/desktop/electron/wake-indicator.test.ts b/apps/desktop/electron/wake-indicator.test.ts new file mode 100644 index 0000000000..433a9ca235 --- /dev/null +++ b/apps/desktop/electron/wake-indicator.test.ts @@ -0,0 +1,143 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +import { normalizeWakeIndicatorState, selectWakeIndicatorDisplay, wakeIndicatorWindowBounds } from './wake-indicator' +import { createWakeIndicatorWindowController } from './wake-indicator-window' + +const electronMock = vi.hoisted(() => { + type Listener = () => void + + class FakeBrowserWindow { + private destroyed = false + private readonly listeners = new Map() + + readonly close = vi.fn(() => { + this.destroyed = true + this.emit('closed') + }) + readonly hide = vi.fn() + readonly setAlwaysOnTop = vi.fn() + readonly setBounds = vi.fn() + readonly setHiddenInMissionControl = vi.fn() + readonly setIgnoreMouseEvents = vi.fn() + readonly setVisibleOnAllWorkspaces = vi.fn() + readonly showInactive = vi.fn() + readonly webContents = { + on: vi.fn(), + send: vi.fn() + } + + constructor(readonly options: unknown) { + windows.push(this) + } + + isDestroyed() { + return this.destroyed + } + + on(event: string, listener: Listener) { + const listeners = this.listeners.get(event) ?? [] + listeners.push(listener) + this.listeners.set(event, listeners) + + return this + } + + once(event: string, listener: Listener) { + return this.on(event, listener) + } + + private emit(event: string) { + for (const listener of this.listeners.get(event) ?? []) { + listener() + } + } + } + + const windows: FakeBrowserWindow[] = [] + + const display = { + bounds: { height: 982, width: 1512, x: 0, y: 0 }, + id: 'internal', + internal: true + } + + return { + BrowserWindow: FakeBrowserWindow, + screen: { + getAllDisplays: () => [display], + getPrimaryDisplay: () => display + }, + windows + } +}) + +vi.mock('electron', () => ({ + BrowserWindow: electronMock.BrowserWindow, + screen: electronMock.screen +})) + +beforeEach(() => { + electronMock.windows.length = 0 +}) + +describe('wake indicator window', () => { + it('centers the helper window at the top of the selected display', () => { + expect( + wakeIndicatorWindowBounds({ + bounds: { height: 982, width: 1512, x: -120, y: 40 } + }) + ).toEqual({ + height: 52, + width: 176, + x: 548, + y: 40 + }) + }) + + it('prefers the internal display and falls back to the primary display', () => { + const primary = { + bounds: { height: 1080, width: 1920, x: 0, y: 0 }, + id: 'primary', + internal: false + } + + const internal = { + bounds: { height: 982, width: 1512, x: 1920, y: 0 }, + id: 'internal', + internal: true + } + + expect(selectWakeIndicatorDisplay([primary, internal], primary)).toBe(internal) + + expect(selectWakeIndicatorDisplay([primary], primary)).toBe(primary) + }) + + it('rejects unknown renderer states', () => { + expect(normalizeWakeIndicatorState('capturing')).toBe('capturing') + expect(normalizeWakeIndicatorState('other')).toBe('hidden') + expect(normalizeWakeIndicatorState(null)).toBe('hidden') + }) +}) + +describe('wake indicator window controller', () => { + it('closes an active helper window and resets its state', () => { + const controller = createWakeIndicatorWindowController({ + isMac: true, + loadWindowUrl: vi.fn(), + preloadPath: '/tmp/preload.cjs', + rendererIndex: () => '/tmp/index.html', + wireWindow: vi.fn() + }) + + controller.setState('detected') + + expect(electronMock.windows).toHaveLength(1) + expect(controller.getState()).toBe('detected') + + const [window] = electronMock.windows + controller.close() + + expect(window.close).toHaveBeenCalledOnce() + expect(controller.getState()).toBe('hidden') + }) +}) diff --git a/apps/desktop/electron/wake-indicator.ts b/apps/desktop/electron/wake-indicator.ts new file mode 100644 index 0000000000..bf95672645 --- /dev/null +++ b/apps/desktop/electron/wake-indicator.ts @@ -0,0 +1,34 @@ +export const WAKE_INDICATOR_WINDOW_WIDTH = 176 +export const WAKE_INDICATOR_WINDOW_HEIGHT = 52 +export const WAKE_INDICATOR_FADE_MS = 500 + +export const WAKE_INDICATOR_STATES = ['hidden', 'detected', 'capturing'] as const + +export type WakeIndicatorState = (typeof WAKE_INDICATOR_STATES)[number] + +interface DisplayLike { + bounds: { + height: number + width: number + x: number + y: number + } + internal?: boolean +} + +export function normalizeWakeIndicatorState(value: unknown): WakeIndicatorState { + return WAKE_INDICATOR_STATES.includes(value as WakeIndicatorState) ? (value as WakeIndicatorState) : 'hidden' +} + +export function selectWakeIndicatorDisplay(displays: T[], primary: T): T { + return displays.find(display => display.internal === true) ?? primary +} + +export function wakeIndicatorWindowBounds(display: DisplayLike) { + return { + height: WAKE_INDICATOR_WINDOW_HEIGHT, + width: WAKE_INDICATOR_WINDOW_WIDTH, + x: Math.round(display.bounds.x + (display.bounds.width - WAKE_INDICATOR_WINDOW_WIDTH) / 2), + y: Math.round(display.bounds.y) + } +} diff --git a/apps/desktop/electron/windows-remote-lifecycle.test.ts b/apps/desktop/electron/windows-remote-lifecycle.test.ts index 56cfb4d45d..b63d9c69bf 100644 --- a/apps/desktop/electron/windows-remote-lifecycle.test.ts +++ b/apps/desktop/electron/windows-remote-lifecycle.test.ts @@ -1,4 +1,5 @@ import assert from 'node:assert/strict' +import crypto from 'node:crypto' import { test } from 'vitest' @@ -9,6 +10,7 @@ import { helperCommand, powerShellCommand, psLiteral, + reusableWindowsLock, validLock } from './windows-remote-lifecycle' @@ -114,6 +116,31 @@ test('Windows lock validation is scoped and exact', () => { assert.equal(validLock({ ...lock, port: -1 }, ownershipId), false) }) +test('Windows SSH reuse requires the requested remote profile to match the lock', () => { + const token = 'stored-token' + + const lock = { + schemaVersion: 2, + protocolVersion: 1, + ownershipId, + spawnNonce: '0123456789abcdef', + pid: 10, + creationTimeNs: '1784219690452757504', + port: 1234, + profile: 'default', + tokenFingerprint: crypto.createHash('sha256').update(token).digest('hex').slice(0, 32), + hermesPath: 'C:\\h\\hermes.exe', + hermesHome: 'C:\\h' + } + + const state = { alive: true, owned: true } + const runtime = { hermesPath: lock.hermesPath, hermesHome: lock.hermesHome } + + assert.equal(reusableWindowsLock(lock, state, 'default', token, runtime), true) + assert.equal(reusableWindowsLock(lock, state, 'desktop-work', token, runtime), false) + assert.equal(reusableWindowsLock({ ...lock, profile: '' }, state, '', token, runtime), true) +}) + test('Windows integrated terminal uses encoded PowerShell and preserves cwd as literal data', () => { const command = buildWindowsInteractiveCommand("C:\\Users\\O'Brien\\repo") const script = Buffer.from(command.split(' ').pop()!, 'base64').toString('utf16le') diff --git a/apps/desktop/electron/windows-remote-lifecycle.ts b/apps/desktop/electron/windows-remote-lifecycle.ts index ff03d6a8c7..5d96853a1a 100644 --- a/apps/desktop/electron/windows-remote-lifecycle.ts +++ b/apps/desktop/electron/windows-remote-lifecycle.ts @@ -149,6 +149,19 @@ function validLock(lock, ownershipId) { ) } +function reusableWindowsLock(lock, state, profile, reuseToken, runtime) { + return Boolean( + state.alive && + state.owned && + lock.port > 0 && + lock.profile === profile && + reuseToken && + lock.tokenFingerprint === fingerprintToken(reuseToken) && + lock.hermesPath === runtime.hermesPath && + lock.hermesHome === runtime.hermesHome + ) +} + function assertCurrent(signal) { if (signal?.aborted) { const error: any = new Error('SSH bootstrap was cancelled.') @@ -299,14 +312,7 @@ async function connectWindowsRemote(deps) { throw error } - const reusable = - state.alive && - state.owned && - lock.port > 0 && - Boolean(reuseToken) && - lock.tokenFingerprint === fingerprintToken(reuseToken) && - lock.hermesPath === runtime.hermesPath && - lock.hermesHome === runtime.hermesHome + const reusable = reusableWindowsLock(lock, state, profile, reuseToken, runtime) if (reusable) { const localPort = await pickLocalPort() @@ -451,5 +457,6 @@ export { powerShellCommand, probeWindowsRemote, psLiteral, + reusableWindowsLock, validLock } diff --git a/apps/desktop/electron/zoom.test.ts b/apps/desktop/electron/zoom.test.ts index 7fbb291a24..c288018979 100644 --- a/apps/desktop/electron/zoom.test.ts +++ b/apps/desktop/electron/zoom.test.ts @@ -162,8 +162,8 @@ test('installZoomReassertOnWindowEvents skips destroyed windows', () => { assert.equal(calls, 0) }) -// Zoom-wiring contract: chat windows keep global UI zoom, the pet overlay -// opts out. Tested via the extracted config — no source-text regex. +// Zoom-wiring contract: chat windows keep global UI zoom while fixed-size +// helper windows opt out. Tested via the extracted config — no source-text regex. test('chat windows opt into zoom', () => { assert.deepEqual(zoomWiringForWindowKind('chat'), { zoom: true }) }) @@ -172,6 +172,10 @@ test('pet overlay opts out of zoom', () => { assert.deepEqual(zoomWiringForWindowKind('petOverlay'), { zoom: false }) }) +test('wake indicator opts out of zoom', () => { + assert.deepEqual(zoomWiringForWindowKind('wakeIndicator'), { zoom: false }) +}) + test('unknown window kinds default to chat (zoom enabled)', () => { assert.deepEqual(zoomWiringForWindowKind('unknown'), { zoom: true }) assert.deepEqual(zoomWiringForWindowKind(undefined), { zoom: true }) diff --git a/apps/desktop/electron/zoom.ts b/apps/desktop/electron/zoom.ts index ee3d576488..4a6d0db6bb 100644 --- a/apps/desktop/electron/zoom.ts +++ b/apps/desktop/electron/zoom.ts @@ -108,7 +108,8 @@ export function installZoomReassertOnWindowEvents(win, reassert, platform = proc export const ZOOM_WINDOW_CONFIG = { chat: { zoom: true }, petOverlay: { zoom: false }, - quickEntry: { zoom: false } + quickEntry: { zoom: false }, + wakeIndicator: { zoom: false } } as const export function zoomWiringForWindowKind(kind) { diff --git a/apps/desktop/package.json b/apps/desktop/package.json index 8a3f76e9f4..1a0f100075 100644 --- a/apps/desktop/package.json +++ b/apps/desktop/package.json @@ -8,13 +8,13 @@ "type": "module", "main": "dist/electron-main.mjs", "engines": { - "node": "^20.19.0 || >=22.12.0" + "node": ">=22.22.0" }, "scripts": { "clean": "npm run clean:e2e && npm run clean:renderer && npm run clean:electron", - "clean:e2e":"tsc --build tsconfig.e2e.json --clean", - "clean:renderer":"tsc --build tsconfig.json --clean ", - "clean:electron":"tsc --build tsconfig.electron.json --clean", + "clean:e2e": "tsc --build tsconfig.e2e.json --clean", + "clean:renderer": "tsc --build tsconfig.json --clean ", + "clean:electron": "tsc --build tsconfig.electron.json --clean", "dev": "concurrently -k \"npm:dev:renderer\" \"npm:dev:electron\"", "dev:fake-boot": "cross-env HERMES_DESKTOP_BOOT_FAKE=1 HERMES_DESKTOP_BOOT_FAKE_STEP_MS=650 npm run dev", "dev:mock": "node scripts/dev-mock.mjs", @@ -64,110 +64,101 @@ "test:e2e:update-snapshots": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list --update-snapshots" }, "dependencies": { - "@assistant-ui/react": "^0.14.23", - "@assistant-ui/react-streamdown": "^0.3.4", - "@audiowave/react": "^0.6.2", - "@chenglou/pretext": "^0.0.6", - "@codemirror/commands": "^6.10.4", - "@codemirror/language": "^6.12.4", - "@codemirror/language-data": "^6.5.2", - "@codemirror/state": "^6.7.0", - "@codemirror/view": "^6.43.3", - "@dnd-kit/core": "^6.3.1", - "@dnd-kit/sortable": "^10.0.0", - "@dnd-kit/utilities": "^3.2.2", + "@assistant-ui/core": "0.2.23", + "@assistant-ui/react": "0.14.24", + "@assistant-ui/react-streamdown": "0.3.5", + "@audiowave/react": "0.6.2", + "@chenglou/pretext": "0.0.6", + "@codemirror/commands": "6.10.4", + "@codemirror/language": "6.12.4", + "@codemirror/language-data": "6.5.2", + "@codemirror/state": "6.7.1", + "@codemirror/view": "6.43.6", + "@dnd-kit/core": "6.3.1", + "@dnd-kit/sortable": "10.0.0", + "@dnd-kit/utilities": "3.2.2", "@hermes/shared": "file:../shared", - "@icons-pack/react-simple-icons": "=13.11.1", - "@lezer/highlight": "^1.2.3", - "@nanostores/react": "^1.1.0", - "@nous-research/ui": "^0.13.0", - "@radix-ui/react-slot": "^1.2.4", - "@streamdown/code": "^1.1.1", - "@tabler/icons-react": "^3.41.1", - "@tailwindcss/typography": "^0.5.19", - "@tailwindcss/vite": "^4.2.4", - "@tanstack/react-query": "^5.100.6", - "@tanstack/react-virtual": "^3.13.24", - "@vscode/codicons": "^0.0.45", - "@xterm/addon-fit": "^0.11.0", - "@xterm/addon-serialize": "^0.14.0", - "@xterm/addon-unicode11": "^0.9.0", - "@xterm/addon-web-links": "^0.12.0", - "@xterm/addon-webgl": "^0.19.0", - "@xterm/xterm": "^6.0.0", - "class-variance-authority": "^0.7.1", - "clsx": "^2.1.1", - "cmdk": "^1.1.1", - "d3-force": "^3.0.0", - "dnd-core": "^14.0.1", - "dompurify": "^3.4.11", - "emojibase-data": "^16.0.3", - "fflate": "^0.8.3", - "frimousse": "^0.3.0", - "hast-util-from-html-isomorphic": "^2.0.0", - "hast-util-to-text": "^4.0.2", - "ignore": "^7.0.5", - "katex": "^0.16.45", - "mermaid": "^11.15.0", - "motion": "^12.38.0", - "nanostores": "^1.3.0", + "@icons-pack/react-simple-icons": "13.11.1", + "@lezer/highlight": "1.2.3", + "@nanostores/react": "1.1.0", + "@nous-research/ui": "0.18.2", + "@streamdown/code": "1.1.1", + "@tabler/icons-react": "3.44.0", + "@tailwindcss/typography": "0.5.20", + "@tailwindcss/vite": "4.3.3", + "@tanstack/react-query": "5.101.2", + "@tanstack/react-virtual": "3.14.6", + "@vscode/codicons": "0.0.45", + "@xterm/addon-fit": "0.11.0", + "@xterm/addon-serialize": "0.14.0", + "@xterm/addon-unicode11": "0.9.0", + "@xterm/addon-web-links": "0.12.0", + "@xterm/addon-webgl": "0.19.0", + "@xterm/xterm": "6.0.0", + "class-variance-authority": "0.7.1", + "clsx": "2.1.1", + "cmdk": "1.1.1", + "d3-force": "3.0.0", + "dnd-core": "14.0.1", + "dompurify": "3.4.12", + "emojibase-data": "16.0.3", + "fflate": "0.8.3", + "frimousse": "0.3.0", + "hast-util-from-html-isomorphic": "2.0.0", + "hast-util-to-text": "4.0.2", + "ignore": "7.0.6", + "katex": "0.16.47", + "mermaid": "11.16.0", + "motion": "12.42.2", + "nanostores": "1.4.0", "node-pty": "1.1.0", - "radix-ui": "^1.6.5", - "react": "^19.2.5", - "react-arborist": "^3.5.0", - "react-dnd-html5-backend": "^14.0.3", - "react-dom": "^19.2.5", - "react-router-dom": "^7.17.0", - "react-shiki": "^0.9.3", - "remark-math": "^6.0.0", - "remend": "^1.3.0", - "shiki": "^4.0.2", - "simple-git": "^3.36.0", - "streamdown": "^2.5.0", - "tailwind-merge": "^3.5.0", - "tailwindcss": "^4.2.4", - "tw-shimmer": "^0.4.11", - "unicode-animations": "^1.0.3", - "unified": "^11.0.5", - "unist-util-visit-parents": "^6.0.2", - "use-stick-to-bottom": "^1.1.6", - "vfile": "^6.0.3", - "web-haptics": "^0.0.6" + "radix-ui": "1.6.7", + "react": "19.2.7", + "react-arborist": "3.13.2", + "react-dnd-html5-backend": "14.1.0", + "react-dom": "19.2.7", + "react-router": "8.3.0", + "react-shiki": "0.9.3", + "remark-math": "6.0.0", + "remend": "1.3.0", + "shiki": "4.3.1", + "simple-git": "3.36.0", + "streamdown": "2.5.0", + "tailwind-merge": "3.6.0", + "tailwindcss": "4.3.3", + "tw-shimmer": "0.4.12", + "unicode-animations": "1.0.3", + "unified": "11.0.5", + "unist-util-visit-parents": "6.0.2", + "use-stick-to-bottom": "1.1.6", + "vfile": "6.0.3", + "web-haptics": "0.0.6" }, "devDependencies": { - "@electron/rebuild": "^4.0.6", - "@eslint/js": "^9.39.4", - "@playwright/test": "=1.58.2", - "@testing-library/dom": "^10.4.0", - "@testing-library/react": "^16.3.2", - "@types/d3-force": "^3.0.10", - "@types/hast": "^3.0.4", - "@types/node": "^22.20.0", - "@types/react": "^19.2.14", - "@types/react-dom": "^19.2.3", - "@typescript-eslint/eslint-plugin": "^8.59.1", - "@typescript-eslint/parser": "^8.59.1", - "@vitejs/plugin-react": "^6.0.1", + "@electron/rebuild": "4.2.0", + "@playwright/test": "1.58.2", + "@testing-library/dom": "10.4.1", + "@testing-library/react": "16.3.2", + "@types/d3-force": "3.0.10", + "@types/hast": "3.0.5", + "@types/node": "22.20.1", + "@types/react": "19.2.17", + "@types/react-dom": "19.2.3", + "@vitejs/plugin-react": "6.0.3", "bippy": "0.5.43", - "concurrently": "^10.0.3", - "cross-env": "^10.1.0", + "concurrently": "10.0.4", + "cross-env": "10.1.0", "electron": "40.10.2", - "electron-builder": "^26.8.1", - "esbuild": "^0.28.1", - "eslint": "^9.39.4", - "eslint-plugin-perfectionist": "^5.9.0", - "eslint-plugin-react": "^7.37.5", - "eslint-plugin-react-hooks": "^7.1.1", - "eslint-plugin-unused-imports": "^4.4.1", - "globals": "^16.5.0", - "jsdom": "^29.1.1", - "prettier": "^3.8.3", - "rcedit": "^5.0.2", - "tsx": "^4.22.4", - "typescript": "^6.0.3", - "vite": "^8.0.10", - "vitest": "^4.1.5", - "wait-on": "^9.0.5" + "electron-builder": "26.15.3", + "esbuild": "0.28.1", + "jsdom": "29.1.1", + "prettier": "3.9.5", + "rcedit": "5.0.2", + "tsx": "4.23.1", + "typescript": "6.0.3", + "vite": "8.2.0", + "vitest": "4.1.10", + "wait-on": "9.0.10" }, "build": { "electronVersion": "40.10.2", diff --git a/apps/desktop/src/app/artifacts/index.tsx b/apps/desktop/src/app/artifacts/index.tsx index 2482405fb7..83f4686359 100644 --- a/apps/desktop/src/app/artifacts/index.tsx +++ b/apps/desktop/src/app/artifacts/index.tsx @@ -1,6 +1,6 @@ import type * as React from 'react' import { memo, useCallback, useEffect, useMemo, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { ZoomableImage } from '@/components/chat/zoomable-image' import { PageLoader } from '@/components/page-loader' diff --git a/apps/desktop/src/app/chat/close-tab.test.ts b/apps/desktop/src/app/chat/close-tab.test.ts index 0326efebea..65642734bc 100644 --- a/apps/desktop/src/app/chat/close-tab.test.ts +++ b/apps/desktop/src/app/chat/close-tab.test.ts @@ -1,12 +1,14 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const closeFocusedSessionTab = vi.fn(() => false) +const closeFocusedToolTab = vi.fn(() => false) const nextSessionTileForWorkspace = vi.fn<() => null | string>(() => null) const closeSessionTile = vi.fn() const requestFreshSession = vi.fn() vi.mock('@/components/pane-shell/tree/store', () => ({ - closeFocusedSessionTab: () => closeFocusedSessionTab() + closeFocusedSessionTab: () => closeFocusedSessionTab(), + closeFocusedToolTab: () => closeFocusedToolTab() })) vi.mock('@/store/session-states', () => ({ @@ -51,6 +53,7 @@ beforeEach(() => { $activeSessionId.set(null) $workspaceIsPage.set(false) closeFocusedSessionTab.mockReturnValue(false) + closeFocusedToolTab.mockReturnValue(false) nextSessionTileForWorkspace.mockReturnValue(null) vi.clearAllMocks() }) @@ -135,4 +138,13 @@ describe('closeWorkspaceTab', () => { expect(closeActiveTab(vi.fn())).toBe(true) expect(requestFreshSession).toHaveBeenCalledTimes(1) }) + + it('a focused tool panel (terminal / logs) claims ⌘W before main empties', () => { + loadedMainOnly() + closeFocusedToolTab.mockReturnValue(true) + + expect(closeActiveTab(vi.fn())).toBe(true) + // The logs/terminal tab closed — main keeps its loaded chat. + expect(requestFreshSession).not.toHaveBeenCalled() + }) }) diff --git a/apps/desktop/src/app/chat/close-tab.ts b/apps/desktop/src/app/chat/close-tab.ts index 0d75455bd8..29284c919d 100644 --- a/apps/desktop/src/app/chat/close-tab.ts +++ b/apps/desktop/src/app/chat/close-tab.ts @@ -1,7 +1,7 @@ import { mainChatOccupied } from '@/app/open-session' import { closeActiveTerminal } from '@/app/right-sidebar/terminal/terminals' import { $workspaceIsPage } from '@/app/routes' -import { closeFocusedSessionTab } from '@/components/pane-shell/tree/store' +import { closeFocusedSessionTab, closeFocusedToolTab } from '@/components/pane-shell/tree/store' import { isFocusWithin } from '@/lib/keybinds/combo' import { $previewTabs, closeActiveRightRailTab } from '@/store/preview' import { requestFreshSession } from '@/store/profile' @@ -56,12 +56,13 @@ export function closeWorkspaceTab(loadSessionIntoWorkspace?: (storedSessionId: s * 1. a focused terminal → its active terminal tab, * 2. right-rail tabs (live preview and/or file peeks), * 3. the FOCUSED chat zone → its active tab (a session tile stacked into it). - * 4. the workspace tab itself — see `closeWorkspaceTab`. + * 4. a focused TOOL PANEL zone (terminal / logs) → its active tab. + * 5. the workspace tab itself — see `closeWorkspaceTab`. * Returns false when nothing closes, so ⌘W is a no-op — it never closes the * window. Shared by the keyboard path (Win/Linux) and the macOS * menu-accelerator IPC. * - * Steps 3-4 follow the same focused zone ⌘1…⌘9 indexes, so a second chat zone + * Steps 3-5 follow the same focused zone ⌘1…⌘9 indexes, so a second chat zone * with its own tab strip closes ITS tab instead of main's. */ export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: string) => void): boolean { @@ -84,5 +85,11 @@ export function closeActiveTab(loadSessionIntoWorkspace?: (storedSessionId: stri return true } + // A tool panel zone hosts no chat strip, so the chat rung skips it — but its + // tabs close like any other. Without this ⌘W was dead over terminal / logs. + if (closeFocusedToolTab()) { + return true + } + return closeWorkspaceTab(loadSessionIntoWorkspace) } diff --git a/apps/desktop/src/app/chat/composer/directive-actions.test.tsx b/apps/desktop/src/app/chat/composer/directive-actions.test.tsx new file mode 100644 index 0000000000..4c827fe9a1 --- /dev/null +++ b/apps/desktop/src/app/chat/composer/directive-actions.test.tsx @@ -0,0 +1,150 @@ +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { I18nProvider } from '@/i18n' + +import { ComposerDirectiveActions } from './directive-actions' +import { refChipElement } from './rich-editor' + +const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] } + +const openSession = vi.fn() + +vi.mock('@/app/open-session', () => ({ openSession: (...args: unknown[]) => openSession(...args) })) + +/** A live contenteditable holding real chips, with the watcher bound to it — + * the same pair both composers mount. */ +function mountEditor(chips: { kind: string; value: string }[]) { + const editor = document.createElement('div') + + editor.contentEditable = 'true' + editor.append(...chips.map(chip => refChipElement(chip.kind, `\`${chip.value}\``))) + document.body.append(editor) + + render( + + + + ) + + return editor +} + +function chips(editor: HTMLElement, kind: string) { + return Array.from(editor.querySelectorAll(`[data-ref-kind="${kind}"]`)) +} + +function hover(node: Element) { + fireEvent.pointerOver(node, { bubbles: true }) +} + +/** The reference the visible action pill points at, or null when there is none. */ +function pillValue() { + return document.querySelector('[data-slot="composer-directive-action"]')?.getAttribute('data-value') ?? null +} + +afterEach(() => { + cleanup() + document.body.replaceChildren() + delete desktopWindow.hermesDesktop + openSession.mockReset() + vi.useRealTimers() +}) + +describe('ComposerDirectiveActions', () => { + it('offers an action for a hovered actionable chip', () => { + const editor = mountEditor([{ kind: 'url', value: 'https://example.com/docs' }]) + + expect(pillValue()).toBeNull() + + hover(chips(editor, 'url')[0]!) + + expect(pillValue()).toBe('https://example.com/docs') + }) + + it('opens a url externally rather than navigating the app', () => { + const openExternal = vi.fn().mockResolvedValue(undefined) + + desktopWindow.hermesDesktop = { openExternal } as unknown as Window['hermesDesktop'] + + const editor = mountEditor([{ kind: 'url', value: 'https://example.com/docs' }]) + + hover(chips(editor, 'url')[0]!) + fireEvent.click(screen.getByRole('button')) + + expect(openExternal).toHaveBeenCalledWith('https://example.com/docs') + expect(pillValue()).toBeNull() + }) + + it('runs the kind-specific action — a session chip opens the session', async () => { + const editor = mountEditor([{ kind: 'session', value: 'default/20260722_204335_d62c16' }]) + + hover(chips(editor, 'session')[0]!) + fireEvent.click(screen.getByRole('button')) + // openSessionRef lazy-imports the navigator, so the call lands a tick later. + await vi.waitFor(() => + expect(openSession).toHaveBeenCalledWith('20260722_204335_d62c16', expect.any(Function), 'tab') + ) + }) + + it('leaves kinds with no action alone', () => { + const editor = mountEditor([{ kind: 'file', value: 'src/main.tsx' }]) + + hover(chips(editor, 'file')[0]!) + + expect(pillValue()).toBeNull() + }) + + it('follows the pointer from one chip to the next', () => { + const editor = mountEditor([ + { kind: 'url', value: 'https://one.example' }, + { kind: 'url', value: 'https://two.example' } + ]) + + const [first, second] = chips(editor, 'url') + + hover(first!) + + expect(pillValue()).toBe('https://one.example') + + hover(second!) + + expect(pillValue()).toBe('https://two.example') + }) + + it('keeps the pill up while the pointer crosses onto it', () => { + vi.useFakeTimers() + + const editor = mountEditor([{ kind: 'url', value: 'https://example.com' }]) + const chip = chips(editor, 'url')[0]! + + hover(chip) + fireEvent.pointerOut(chip, { relatedTarget: document.body }) + fireEvent.mouseEnter(screen.getByRole('button').parentElement!) + vi.advanceTimersByTime(500) + + expect(pillValue()).toBe('https://example.com') + }) + + it('binds to the document so a late-attached editor still gets the affordance', () => { + // The edit composer's editor isn't reliably in the DOM when the effect + // first runs; a document listener that reads the editor lazily works + // regardless — this is the whole reason it binds to document, not editor. + const editor = document.createElement('div') + + editor.contentEditable = 'true' + editor.append(refChipElement('url', '`https://late.example`')) + + render( + + + + ) + + // Editor attached AFTER mount. + document.body.append(editor) + hover(editor.querySelector('[data-ref-kind="url"]')!) + + expect(pillValue()).toBe('https://late.example') + }) +}) diff --git a/apps/desktop/src/app/chat/composer/directive-actions.tsx b/apps/desktop/src/app/chat/composer/directive-actions.tsx new file mode 100644 index 0000000000..b18ade2aac --- /dev/null +++ b/apps/desktop/src/app/chat/composer/directive-actions.tsx @@ -0,0 +1,154 @@ +/** + * Hover actions for directive chips in a composer. + * + * A directive chip (`@url:`, `@session:`, …) reads as the thing it points at + * and is coloured like one, but a composer is an editor — a click inside the + * contenteditable only places the caret, so there's no way to *act* on the + * reference. Instead, hovering a chip whose kind has an action floats a small + * pill above it that runs it. + * + * The kind → action table (`DIRECTIVE_ACTIONS`) lives in `directive-text`, so + * it is shared with the sent-message chip: one entry lights up both surfaces. + */ +import { type RefObject, useCallback, useEffect, useRef, useState } from 'react' +import { createPortal } from 'react-dom' + +import { DIRECTIVE_ACTIONS, type DirectiveAction } from '@/components/assistant-ui/directive-text' +import { composerFloatingPill } from '@/components/chat/composer-dock' +import { Codicon } from '@/components/ui/codicon' +import { useI18n } from '@/i18n' +import { cn } from '@/lib/utils' + +/** Moving between the chip and the pill crosses a gap where neither is hovered. + * Short enough that it still reads as instant on the way out. */ +const HIDE_DELAY_MS = 120 + +/** The actionable directive chip under `target` that also belongs to `editor`, + * if there is one. */ +function actionableChipAt(target: EventTarget | null, editor: HTMLElement): HTMLElement | null { + const chip = target instanceof Element ? target.closest('[data-ref-kind]') : null + const kind = chip?.dataset.refKind + + return chip && kind && chip.dataset.refId && editor.contains(chip) && DIRECTIVE_ACTIONS[kind] ? chip : null +} + +interface Anchor { + action: DirectiveAction + chip: HTMLElement + left: number + top: number + value: string +} + +function anchorFor(chip: HTMLElement): Anchor | null { + const value = chip.dataset.refId + const action = chip.dataset.refKind ? DIRECTIVE_ACTIONS[chip.dataset.refKind] : undefined + + if (!value || !action || !chip.isConnected) { + return null + } + + const rect = chip.getBoundingClientRect() + + return { action, chip, left: rect.left, top: rect.top, value } +} + +/** + * Renders the action pill for whichever actionable chip in `editorRef` is + * hovered. + * + * Listeners bind to `document`, not the editor, so mount timing can't strand + * them: the edit composer's contenteditable isn't reliably attached when this + * effect first runs, and a document listener that reads the editor lazily works + * regardless. Each instance filters to its own editor, so the docked and edit + * composers never show two pills for one chip. + */ +export function ComposerDirectiveActions({ editorRef }: { editorRef: RefObject }) { + const { t } = useI18n() + const [anchor, setAnchor] = useState(null) + const hideTimerRef = useRef(undefined) + + const cancelHide = useCallback(() => { + window.clearTimeout(hideTimerRef.current) + }, []) + + const hideSoon = useCallback(() => { + cancelHide() + hideTimerRef.current = window.setTimeout(() => setAnchor(null), HIDE_DELAY_MS) + }, [cancelHide]) + + useEffect(() => { + const onPointerOver = (event: PointerEvent) => { + const editor = editorRef.current + const chip = editor && actionableChipAt(event.target, editor) + + if (!chip) { + return + } + + cancelHide() + setAnchor(current => (current?.chip === chip ? current : anchorFor(chip))) + } + + const onPointerOut = (event: PointerEvent) => { + const editor = editorRef.current + const chip = editor && actionableChipAt(event.target, editor) + + // A move within the same chip (its icon → its label) is not a leave. + if (chip && editor && chip === actionableChipAt(event.relatedTarget, editor)) { + return + } + + hideSoon() + } + + // The chip can move or vanish under a parked pointer: the editor scrolls, + // the window resizes, or the user deletes the reference the pill points at. + const reanchor = () => setAnchor(current => (current ? anchorFor(current.chip) : null)) + + document.addEventListener('pointerover', onPointerOver) + document.addEventListener('pointerout', onPointerOut) + window.addEventListener('scroll', reanchor, true) + window.addEventListener('resize', reanchor) + + return () => { + document.removeEventListener('pointerover', onPointerOver) + document.removeEventListener('pointerout', onPointerOut) + window.removeEventListener('scroll', reanchor, true) + window.removeEventListener('resize', reanchor) + window.clearTimeout(hideTimerRef.current) + } + }, [cancelHide, editorRef, hideSoon]) + + if (!anchor) { + return null + } + + return createPortal( +
+ +
, + document.body + ) +} diff --git a/apps/desktop/src/app/chat/composer/empty-composer.test.ts b/apps/desktop/src/app/chat/composer/empty-composer.test.ts index fe8305d2c4..dcf3398e6b 100644 --- a/apps/desktop/src/app/chat/composer/empty-composer.test.ts +++ b/apps/desktop/src/app/chat/composer/empty-composer.test.ts @@ -1,7 +1,9 @@ import { describe, expect, it } from 'vitest' import { + beginComposerComposition, composerPlainText, + deleteChipBeforeCaret, normalizeComposerEditorDom, renderComposerContents, RICH_INPUT_SLOT @@ -136,4 +138,133 @@ describe('an emptied composer shows its placeholder again', () => { expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true) }) + + // Chromium leaves zero-length text nodes behind whenever an edit lands next + // to a contenteditable=false chip. They render as nothing, so an editor + // holding only those is empty to the user — counting them as contents left + // the placeholder hidden under a composer that looked blank. + it('advertises emptiness for an editor holding only zero-length text nodes', () => { + const el = editor() + + el.append(document.createTextNode(''), document.createTextNode('')) + normalizeComposerEditorDom(el) + + expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true) + }) + + it('does not advertise emptiness while real text sits beside that litter', () => { + const el = editor() + + el.append(document.createTextNode(''), document.createTextNode('one'), document.createTextNode('')) + normalizeComposerEditorDom(el) + + expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false) + }) + + // Input events are skipped for the duration of an IME composition, so nothing + // else clears the marker until it ends — the hint would sit behind the + // hiragana the user is composing (#75960). + it('hides the placeholder before IME preedit text starts', () => { + const el = emptied() + + beginComposerComposition(el) + + expect(el.matches(PLACEHOLDER_SHOWS)).toBe(false) + }) + + it('brings the placeholder back when composition ends with nothing committed', () => { + const el = emptied() + + beginComposerComposition(el) + normalizeComposerEditorDom(el) + + expect(el.matches(PLACEHOLDER_SHOWS)).toBe(true) + }) +}) + +/** A directive chip, as `refChipElement` builds it. */ +function chip(): HTMLSpanElement { + const el = document.createElement('span') + + el.contentEditable = 'false' + el.dataset.refText = '@folder:`apps/desktop/`' + el.append(document.createTextNode('apps/desktop/')) + + return el +} + +function caretAt(node: Node, offset: number) { + const range = document.createRange() + + range.setStart(node, offset) + range.collapse(true) + + const selection = window.getSelection() + + selection?.removeAllRanges() + selection?.addRange(range) +} + +/** Committing a completion empties the typed token's text node instead of + * removing it, and `Range.insertNode` splits the line around the caret — so a + * freshly-chipped directive sits between zero-length text nodes. Backspace has + * to see past them or the chip can't be deleted at all. */ +describe('backspace deletes a chip surrounded by Chromium litter', () => { + it('deletes the chip when the caret sits in a zero-length text node after it', () => { + const el = editor() + + el.append(document.createTextNode(''), chip(), document.createTextNode('')) + caretAt(el.childNodes[2] as Node, 0) + + expect(deleteChipBeforeCaret(el)).toBe(true) + expect(el.querySelector('[data-ref-text]')).toBeNull() + }) + + it('deletes the chip when the caret is past a zero-length text node at editor level', () => { + const el = editor() + + el.append(chip(), document.createTextNode('')) + caretAt(el, 2) + + expect(deleteChipBeforeCaret(el)).toBe(true) + expect(el.querySelector('[data-ref-text]')).toBeNull() + }) + + it('still swallows the auto-inserted trailing space through that litter', () => { + const el = editor() + + el.append(chip(), document.createTextNode(''), document.createTextNode(' ')) + caretAt(el, 2) + + expect(deleteChipBeforeCaret(el)).toBe(true) + expect(composerPlainText(el)).toBe('') + }) + + it('keeps real following text when it deletes the chip', () => { + const el = editor() + + el.append(chip(), document.createTextNode(''), document.createTextNode(' and this')) + caretAt(el, 2) + + expect(deleteChipBeforeCaret(el)).toBe(true) + expect(composerPlainText(el)).toBe('and this') + }) + + it('leaves plain text to the native backspace', () => { + const el = editor() + + el.append(document.createTextNode('hello')) + caretAt(el.firstChild as Node, 5) + + expect(deleteChipBeforeCaret(el)).toBe(false) + }) + + it('sweeps the litter out of the editor when it normalizes', () => { + const el = editor() + + el.append(document.createTextNode(''), chip(), document.createTextNode('')) + normalizeComposerEditorDom(el) + + expect(Array.from(el.childNodes).map(node => node.nodeName)).toEqual(['SPAN']) + }) }) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts b/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts index 49e0ba5ce7..449471dc07 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts @@ -4,6 +4,7 @@ import { useCallback, useEffect, useRef, useState } from 'react' import { useI18n } from '@/i18n' import { chatMessageText, collectUnspokenTurnSpeech } from '@/lib/chat-messages' import { triggerHaptic } from '@/lib/haptics' +import { clearWakeIndicator, syncWakeIndicatorWithVoice } from '@/lib/wake-indicator' import { $voiceConversationStartRequest, takeVoiceConversationStart } from '@/store/composer' import { resetBrowseState } from '@/store/composer-input-history' import { $gateway } from '@/store/gateway' @@ -62,6 +63,7 @@ export function useComposerVoice({ const { $messages } = useComposerScope() const [voiceConversationActive, setVoiceConversationActive] = useState(false) const lastSpokenIdRef = useRef(null) + const ownsWakeIndicatorRef = useRef(false) const voiceStartRequest = useStore($voiceConversationStartRequest) const { dictate, voiceActivityState, voiceStatus } = useVoiceRecorder({ @@ -150,6 +152,26 @@ export function useComposerVoice({ beforeMicOpen: () => wakePauseBarrierRef.current ?? undefined }) + // eslint-disable-next-line no-restricted-syntax -- ownership token used only by unmount cleanup + useEffect(() => { + if (target !== 'main') { + return + } + + if (syncWakeIndicatorWithVoice(voiceConversationActive, conversation.status)) { + ownsWakeIndicatorRef.current = voiceConversationActive + } + }, [conversation.status, target, voiceConversationActive]) + + useEffect( + () => () => { + if (ownsWakeIndicatorRef.current) { + clearWakeIndicator() + } + }, + [] + ) + // The `composer.voice` hotkey (Ctrl+B) toggles the conversation. Starting // with STT unconfigured lets the conversation surface its own "configure // speech-to-text" notice rather than silently no-opping. diff --git a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts index 5f4be4b637..55978c4284 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts @@ -287,8 +287,7 @@ export function useVoiceConversation({ // start — don't auto-start the next sentence, the user chose to stop. const stoppedByUser = stoppedDuringSetup || - (speechStartSequenceRef.current > 0 && - $voicePlayback.get().sequence > speechStartSequenceRef.current) + (speechStartSequenceRef.current > 0 && $voicePlayback.get().sequence > speechStartSequenceRef.current) speechStartSequenceRef.current = 0 diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index 7471775922..3767a4ad4e 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -32,6 +32,7 @@ import { import { ContextMenu } from './context-menu' import { COMPOSER_AREAS, runComposerMiddleware } from './contrib' import { ComposerControls } from './controls' +import { ComposerDirectiveActions } from './directive-actions' import { COMPOSER_DROP_ACTIVE_CLASS, COMPOSER_DROP_FADE_CLASS } from './drop-affordance' import { markActiveComposer } from './focus' import { HelpHint } from './help-hint' @@ -57,7 +58,7 @@ import { ActionBadges } from './micro-actions' import { chipTypedPathOnSpace, pathifyRefs } from './path-refs' import { QueuePanel } from './queue-panel' import { - COMPOSER_PLACEHOLDER_CLASS, + beginComposerComposition, composerPlainText, deleteChipBeforeCaret, deleteSelectionInEditor, @@ -947,7 +948,6 @@ export function ChatBar({ autoCorrect="off" className={cn( 'min-h-[1.625rem] min-h-(--composer-input-min-height) max-h-(--composer-input-max-height) cursor-text overflow-y-auto whitespace-pre-wrap break-words [overflow-wrap:anywhere] bg-transparent pb-1 pr-1 pt-1 leading-normal text-foreground outline-none disabled:cursor-not-allowed', - COMPOSER_PLACEHOLDER_CLASS, '**:data-ref-text:cursor-default', stacked && 'pl-3', stacked ? 'w-full' : 'min-w-(--composer-input-inline-min-width) flex-1' @@ -969,8 +969,13 @@ export function ChatBar({ // until an unrelated edit forces a sync (#39614). flushEditorToDraft(event.currentTarget) }} - onCompositionStart={() => { + onCompositionStart={event => { composingRef.current = true + + // Input events are skipped for the rest of the composition, so + // nothing else would clear the empty marker until it ends — and the + // hint would sit behind the preedit text the whole time (#75960). + beginComposerComposition(event.currentTarget) }} onDragOver={handleInputDragOver} onDrop={handleInputDrop} @@ -985,6 +990,7 @@ export function ChatBar({ spellCheck={false} suppressContentEditableWarning /> + {/* assistant-ui requires ComposerPrimitive.Input somewhere in the tree so the composer-state binding (text + IME + paste + form-submit hookup) wires up. We render the real input UI ourselves above via the diff --git a/apps/desktop/src/app/chat/composer/micro-actions.tsx b/apps/desktop/src/app/chat/composer/micro-actions.tsx index 6f8cc77ad5..65e3729ead 100644 --- a/apps/desktop/src/app/chat/composer/micro-actions.tsx +++ b/apps/desktop/src/app/chat/composer/micro-actions.tsx @@ -1,5 +1,6 @@ import { memo, useState } from 'react' +import { composerFloatingPill } from '@/components/chat/composer-dock' import { Codicon } from '@/components/ui/codicon' import { useSessionSlice } from '@/lib/use-session-slice' import { cn } from '@/lib/utils' @@ -7,11 +8,9 @@ import { $composerActionsBySession, type ComposerAction } from '@/store/composer import { notifyError } from '@/store/notifications' /** - * Floating pill — the treatment the thread's jump/approval button uses for a - * control that sits over scrolling content: full radius, hairline border, the - * shared composer fill behind a blur so thread text never bleeds through. - * Sized against the composer's own control height so a row of pills lines up - * with the chrome it floats above. + * Floating pill — the shared treatment for a control that sits over the + * composer (`composerFloatingPill`), plus this strip's own width cap and + * disabled state. * * NEVER `pointer-events-none`, not even when disabled. The pop-out drag region * is an `absolute` sibling behind these pills, so a pill that stops taking @@ -19,10 +18,8 @@ import { notifyError } from '@/store/notifications' * becomes a grab handle that floats the composer. */ const PILL = cn( - 'inline-flex h-(--composer-control-size) max-w-56 shrink-0 cursor-pointer items-center gap-1.5 rounded-full px-2.5', - 'border border-border/65 bg-(--composer-fill) backdrop-blur-[0.75rem] [-webkit-backdrop-filter:blur(0.75rem)]', - 'text-xs font-normal text-(--ui-text-secondary) transition-colors', - 'hover:bg-(--chrome-action-hover) hover:text-foreground', + composerFloatingPill, + 'max-w-56', 'disabled:cursor-default disabled:opacity-50 disabled:hover:bg-(--composer-fill)', 'focus-visible:outline-none focus-visible:ring-[0.1875rem] focus-visible:ring-ring/50' ) diff --git a/apps/desktop/src/app/chat/composer/rich-editor.ts b/apps/desktop/src/app/chat/composer/rich-editor.ts index 0c9c709887..a7d8e04067 100644 --- a/apps/desktop/src/app/chat/composer/rich-editor.ts +++ b/apps/desktop/src/app/chat/composer/rich-editor.ts @@ -21,28 +21,67 @@ import { slashCommandMatches, type SlashCommandScanOptions } from './slash-refs' export const RICH_INPUT_SLOT = 'composer-rich-input' -/** Paints `data-placeholder` while the editor is empty. +/** Chromium's litter: editing beside a `contenteditable=false` chip splits the + * line and leaves zero-length text nodes behind. They render as nothing and + * serialize as nothing, so no reader of the editor should count them. */ +function isEmptyTextNode(node: ChildNode | null): boolean { + return node?.nodeType === Node.TEXT_NODE && !node.textContent +} + +/** The node before `node`, stepping over that litter. */ +function meaningfulPreviousSibling(node: ChildNode | null): ChildNode | null { + let prev = node?.previousSibling ?? null + + while (isEmptyTextNode(prev)) { + prev = prev?.previousSibling ?? null + } + + return prev +} + +/** The node after `node`, stepping over that litter. */ +function meaningfulNextSibling(node: ChildNode | null): ChildNode | null { + let next = node?.nextSibling ?? null + + while (isEmptyTextNode(next)) { + next = next?.nextSibling ?? null + } + + return next +} + +/** Keep the `data-empty` marker the placeholder paints on in step with the + * editor root's contents. * * `:empty` can't be the whole test: a cleared editor keeps a scaffolding
* so the contenteditable doesn't collapse, and that break makes `:empty` * false. Nor can CSS infer it on its own — a text node is invisible to * selectors, so `one
` and a lone `
` are the same shape, and - * `:has(> br:only-child)` would paint the placeholder straight over the - * user's text. The code that empties the editor is what knows, so it marks it. + * `:has(> br:only-child)` would paint the placeholder over the user's text. + * The code that empties the editor is what knows, so it marks it. * - * @see markEditorEmptiness */ -export const COMPOSER_PLACEHOLDER_CLASS = - '[&:is(:empty,[data-empty])]:before:content-[attr(data-placeholder)] [&:is(:empty,[data-empty])]:before:text-muted-foreground/60' - -/** Keep that marker in step with the editor root's contents. */ + * Zero-length text nodes don't count as contents. Chromium leaves them behind + * whenever an edit lands next to a `contenteditable=false` chip, and counting + * them left an editor the user had emptied looking occupied. */ export function markEditorEmptiness(editor: HTMLElement) { - if (editor.childNodes.length === 0) { + if (Array.from(editor.childNodes).every(isEmptyTextNode)) { editor.dataset.empty = '' } else { delete editor.dataset.empty } } +/** Drop the marker as IME composition starts, before any preedit text lands. + * + * Input events during composition are deliberately skipped (they carry + * uncommitted preedit text), so nothing else clears the marker until + * `compositionend` — and the hint would otherwise sit behind the hiragana the + * user is composing. `normalizeComposerEditorDom` restores it if composition + * ends with nothing committed. */ +export function beginComposerComposition(editor: HTMLElement) { + delete editor.dataset.empty +} + /** @see referenceRe — the shared pattern every surface recognises a reference * with. Module-level `/g` regexes carry `lastIndex`, so call sites reset it. */ export const REF_RE = referenceRe() @@ -407,7 +446,14 @@ export function replaceBeforeCaret(editor: HTMLElement, length: number, fragment /** Backspace at a collapsed caret immediately after a chip: delete the chip AND * the single trailing space we auto-insert after it, atomically — so removing a * directive never strands an orphaned space (the contenteditable-driven cleanup - * was unreliable). Returns whether it ran. */ + * was unreliable). Returns whether it ran. + * + * "Immediately after" has to be read through Chromium's litter. Committing a + * completion empties the typed token's text node rather than removing it, and + * `Range.insertNode` splits around the caret, so the chip routinely sits + * between zero-length text nodes. Reading those as content made the caret look + * like it was after plain text; the delete declined and Chromium's own + * backspace bounced between the leftovers instead of removing the chip. */ export function deleteChipBeforeCaret(editor: HTMLElement): boolean { const hit = composerSelectionRange(editor) @@ -419,16 +465,20 @@ export function deleteChipBeforeCaret(editor: HTMLElement): boolean { let chip: ChildNode | null = null if (startContainer === editor) { - chip = startOffset > 0 ? editor.childNodes[startOffset - 1] : null + chip = startOffset > 0 ? (editor.childNodes[startOffset - 1] ?? null) : null + + if (isEmptyTextNode(chip)) { + chip = meaningfulPreviousSibling(chip) + } } else if (startContainer.nodeType === Node.TEXT_NODE && startOffset === 0) { - chip = startContainer.previousSibling + chip = meaningfulPreviousSibling(startContainer as ChildNode) } if (chip?.nodeType !== Node.ELEMENT_NODE || !(chip as HTMLElement).dataset.refText) { return false } - const after = chip.nextSibling + const after = meaningfulNextSibling(chip) chip.remove() // Drop the auto-inserted trailing space; keep any real following text. @@ -666,6 +716,14 @@ function isBlankNode(node: ChildNode | null): boolean { * rendering emits (we use text nodes +
+ chips). Real
line breaks * (Shift+Enter, which sit after actual text) are preserved. */ export function normalizeComposerEditorDom(editor: HTMLElement) { + // Chromium's zero-length text nodes first: every check below reads siblings, + // and litter between them makes a chip look like it has text either side. + for (const child of Array.from(editor.childNodes)) { + if (isEmptyTextNode(child)) { + child.remove() + } + } + // A trailing block wrapper holding only a break/whitespace is the phantom // "new line" Chromium adds after a chip on backspace — drop it. const tailBlock = editor.lastChild as HTMLElement | null diff --git a/apps/desktop/src/app/chat/composer/status-stack/coding-row.test.tsx b/apps/desktop/src/app/chat/composer/status-stack/coding-row.test.tsx new file mode 100644 index 0000000000..326cdfcfd6 --- /dev/null +++ b/apps/desktop/src/app/chat/composer/status-stack/coding-row.test.tsx @@ -0,0 +1,93 @@ +import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' +import { atom } from 'nanostores' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { $notifications, clearNotifications } from '@/store/notifications' + +vi.mock('@/store/coding-status', () => ({ + registerRepoStatusCwd: () => undefined, + repoStatusForCwd: () => + atom({ + added: 12, + ahead: 0, + behind: 0, + branch: 'bb/hitbox', + defaultBranch: 'main', + detached: false, + removed: 3, + untracked: 0 + }), + repoWorktreesForCwd: () => atom([]) +})) + +const { CodingStatusRow } = await import('./coding-row') + +describe('CodingStatusRow', () => { + afterEach(() => { + cleanup() + }) + + it('opens the review pane from the branch and the diff counts, never the bar itself', () => { + const onOpen = vi.fn() + + const { container } = render() + + const bar = container.querySelector('.coding-status-bar') + + expect(bar).not.toBeNull() + + fireEvent.click(bar!) + expect(onOpen).not.toHaveBeenCalled() + + fireEvent.click(screen.getByText('bb/hitbox')) + expect(onOpen).toHaveBeenCalledTimes(1) + + fireEvent.click(screen.getByText('12')) + expect(onOpen).toHaveBeenCalledTimes(2) + }) + + it('wraps the click targets without adding a layout box', () => { + const { container } = render( undefined} repoPath="/repo" />) + + // `display: contents` is what keeps the branch label and the counts direct + // flex children of the row — the hit areas cost nothing visually. + expect(screen.getByText('bb/hitbox').parentElement?.classList.contains('contents')).toBe(true) + expect(screen.getByText('12').closest('button')?.classList.contains('contents')).toBe(true) + // The glyph button fills the row's existing 3.5 leading slot exactly. + expect(container.querySelector('button[class~="size-3.5"]')).not.toBeNull() + }) + + it('parks the copy glyph against the end of the path, not the end of the row', () => { + render( undefined} repoPath="/Users/someone/www/repo" />) + + const path = screen.getByText('~/www/repo') + + // The path sizes to its content and the glyph is its immediate sibling, so + // the pair reads as one unit. `flex-1` belongs to the wrapper (which holds + // the row's slack open) — on the label it stretched the text and pushed the + // glyph out to the kebab. + expect(path.classList.contains('flex-1')).toBe(false) + expect(path.parentElement?.classList.contains('flex-1')).toBe(true) + expect(path.nextElementSibling?.tagName).toBe('BUTTON') + }) + + it('copies the absolute cwd inline — checkmark feedback, no toast', async () => { + const writeText = vi.fn().mockResolvedValue(undefined) + Object.defineProperty(navigator, 'clipboard', { configurable: true, value: { writeText } }) + clearNotifications() + + render( undefined} repoPath="/Users/someone/www/repo" />) + + // Painted tildified, copied raw. + expect(screen.getByText('~/www/repo')).toBeTruthy() + + const copy = screen.getByRole('button', { name: 'Copy Path' }) + + fireEvent.click(copy) + + await waitFor(() => expect(writeText).toHaveBeenCalledWith('/Users/someone/www/repo')) + // Confirmation is the button turning into a checkmark, not a notification. + await waitFor(() => expect(screen.getByRole('button', { name: 'Copied' })).toBeTruthy()) + expect($notifications.get()).toHaveLength(0) + }) +}) diff --git a/apps/desktop/src/app/chat/composer/status-stack/coding-row.tsx b/apps/desktop/src/app/chat/composer/status-stack/coding-row.tsx index c4b1a56956..842b1ac37e 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/coding-row.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/coding-row.tsx @@ -12,9 +12,11 @@ import { } from '@/components/ui/actions-menu' import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' +import { CopyButton } from '@/components/ui/copy-button' import { DiffCount } from '@/components/ui/diff-count' import type { HermesGitBranch } from '@/global' import { useI18n } from '@/i18n' +import { displayPath } from '@/lib/display-path' import { registerRepoStatusCwd, repoStatusForCwd, repoWorktreesForCwd } from '@/store/coding-status' import { notifyError } from '@/store/notifications' import { $newWorktreeRequest } from '@/store/projects' @@ -62,6 +64,7 @@ export const CodingStatusRow = memo(function CodingStatusRow({ const { t } = useI18n() const s = t.statusStack.coding const p = t.sidebar.projects + const fileMenu = t.fileMenu const resolvedRepoPath = repoPath?.trim() || undefined // This surface's OWN worktree, always — never the primary's. The row used to // fall back to the global `$repoStatus` for a blank repoPath, which painted @@ -219,16 +222,52 @@ export const CodingStatusRow = memo(function CodingStatusRow({ // once `status` exists, so a spinner here only ever fired on *refreshes* // of an already-loaded repo (window focus, turn settle), reading as an // annoying icon "blip" with no first-load value. Refreshes are silent. - leading={} - onActivate={onOpen} + // It's a button (not the whole row) so the glyph opens the review pane + // while the strip around it stays inert; size-3.5 fills the slot exactly. + leading={ + + } >
- - {branchLabel} - + {/* Branch name — the other half of the review-pane target. `contents` + so the button lays out nothing of its own: the label stays the + same flex child it always was, and the hit area is the text. */} + + + {/* Worktree path + copy — plain muted text, not a chip. Always in the + flex so hover doesn't reflow the row; opacity alone reveals the + pair. The path sizes to its content (the `flex-1` lives on the + wrapper) so the glyph sits against the end of the text instead of + drifting to the far edge of the row. `displayPath` collapses + home → ~; the copy still takes the real absolute path, and it's + the shared `CopyButton` so it confirms with the same inline + checkmark as every other copy in the app. */} + {resolvedRepoPath && ( +
+ + {displayPath(resolvedRepoPath)} + + +
+ )} {/* Branch actions kebab — same pattern as the session/worktree rows. ALWAYS laid out; only its opacity flips on hover/focus/open, so @@ -246,14 +285,6 @@ export const CodingStatusRow = memo(function CodingStatusRow({
- {(status.ahead > 0 || status.behind > 0) && ( - - {status.ahead > 0 && ( - - ↑ - {status.ahead} + {/* The counts describe what's in the review pane, so clicking them + opens it. `contents` again: the two spans stay direct flex children + of the row, keeping their gap and `ml-auto` behaviour untouched. */} + {(status.ahead > 0 || status.behind > 0 || hasLineDelta || untrackedOnly) && ( + + )} diff --git a/apps/desktop/src/app/chat/composer/status-stack/goal-indicator.test.tsx b/apps/desktop/src/app/chat/composer/status-stack/goal-indicator.test.tsx index 09a7454b9c..5b51223282 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/goal-indicator.test.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/goal-indicator.test.tsx @@ -1,5 +1,5 @@ import { cleanup, render, screen } from '@testing-library/react' -import { MemoryRouter } from 'react-router-dom' +import { MemoryRouter } from 'react-router' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { I18nProvider } from '@/i18n' diff --git a/apps/desktop/src/app/chat/composer/status-stack/index.tsx b/apps/desktop/src/app/chat/composer/status-stack/index.tsx index 6f48847c41..de86ba14d0 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/index.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/index.tsx @@ -1,6 +1,6 @@ import { useStore } from '@nanostores/react' import { type ReactNode, useEffect, useMemo } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { blurComposerInput } from '@/app/chat/composer/focus' import { AGENTS_ROUTE } from '@/app/routes' diff --git a/apps/desktop/src/app/chat/index.tsx b/apps/desktop/src/app/chat/index.tsx index 0cfe8e0205..3d72bf3410 100644 --- a/apps/desktop/src/app/chat/index.tsx +++ b/apps/desktop/src/app/chat/index.tsx @@ -4,7 +4,7 @@ import { useQuery } from '@tanstack/react-query' import type { ReadableAtom } from 'nanostores' import type * as React from 'react' import { Suspense, useCallback, useEffect, useMemo, useState } from 'react' -import { useLocation } from 'react-router-dom' +import { useLocation } from 'react-router' import type { SubmitTextOptions } from '@/app/session/hooks/use-prompt-actions/utils' import { Thread } from '@/components/assistant-ui/thread' diff --git a/apps/desktop/src/app/chat/sidebar/index.tsx b/apps/desktop/src/app/chat/sidebar/index.tsx index a2bb3455ba..10ec6e4720 100644 --- a/apps/desktop/src/app/chat/sidebar/index.tsx +++ b/apps/desktop/src/app/chat/sidebar/index.tsx @@ -3,7 +3,7 @@ import { sortableKeyboardCoordinates } from '@dnd-kit/sortable' import { useStore } from '@nanostores/react' import type * as React from 'react' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import { useLocation } from 'react-router-dom' +import { useLocation } from 'react-router' import { PlatformAvatar } from '@/app/messaging/platform-icon' import { Button } from '@/components/ui/button' diff --git a/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx b/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx index fc9b6326be..1bd5ad3c1d 100644 --- a/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx +++ b/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx @@ -20,7 +20,7 @@ import { import { CSS } from '@dnd-kit/utilities' import { useStore } from '@nanostores/react' import { useEffect, useRef, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { CodeEditor } from '@/components/chat/code-editor' import { Button } from '@/components/ui/button' diff --git a/apps/desktop/src/app/chat/sidebar/projects/entered-content.tsx b/apps/desktop/src/app/chat/sidebar/projects/entered-content.tsx index 9d32259316..6b09788157 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/entered-content.tsx +++ b/apps/desktop/src/app/chat/sidebar/projects/entered-content.tsx @@ -15,6 +15,7 @@ import { import type { HermesGitWorktree } from '@/global' import type { SessionInfo } from '@/hermes' import { useI18n } from '@/i18n' +import { displayPath } from '@/lib/display-path' import { $dismissedWorktreeIds, dismissWorktree, setWorkspaceNodeOpen } from '@/store/layout' import { notifyError } from '@/store/notifications' import { removeWorktreePath } from '@/store/projects' @@ -277,7 +278,7 @@ function RepoFlatSection({ label={repo.label} onToggle={toggleOpen} open={open} - title={repo.path ?? undefined} + title={repo.path ? displayPath(repo.path) : undefined} /> {open && {body}} {removeDialog} diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-group.tsx b/apps/desktop/src/app/chat/sidebar/projects/workspace-group.tsx index dd38e84a2b..5c05869b5a 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-group.tsx +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-group.tsx @@ -4,6 +4,7 @@ import { useState } from 'react' import { Codicon } from '@/components/ui/codicon' import type { SessionInfo } from '@/hermes' import { useI18n } from '@/i18n' +import { displayPath } from '@/lib/display-path' import { setWorkspaceNodeOpen } from '@/store/layout' import { notifyError } from '@/store/notifications' import { newSessionInProfile } from '@/store/profile' @@ -132,7 +133,7 @@ export function SidebarWorkspaceGroup({ group, renderRows, onNewSession, onRemov label={group.label} onToggle={toggleOpen} open={open} - title={group.path ?? undefined} + title={group.path ? displayPath(group.path) : undefined} /> {open && ( diff --git a/apps/desktop/src/app/command-palette/index.tsx b/apps/desktop/src/app/command-palette/index.tsx index 27ace69694..f9a640000a 100644 --- a/apps/desktop/src/app/command-palette/index.tsx +++ b/apps/desktop/src/app/command-palette/index.tsx @@ -2,7 +2,7 @@ import { useStore } from '@nanostores/react' import { useQuery } from '@tanstack/react-query' import { Dialog as DialogPrimitive } from 'radix-ui' import { memo, useCallback, useDeferredValue, useEffect, useMemo, useRef, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { HUD_HEADING, @@ -13,7 +13,6 @@ import { HUD_SURFACE, HUD_TEXT } from '@/app/floating-hud' -import { setTerminalTakeover } from '@/app/right-sidebar/store' import { codiconIcon } from '@/components/ui/codicon' import { Command, CommandGroup, CommandInput, CommandItem, CommandList } from '@/components/ui/command' import { HighlightMatches } from '@/components/ui/highlight-matches' @@ -52,7 +51,6 @@ import { SlidersHorizontal, Starmap, Sun, - Terminal, Users, Wrench, Zap @@ -67,7 +65,7 @@ import { closeCommandPalette, setCommandPaletteOpen } from '@/store/command-palette' -import { $bindings } from '@/store/keybinds' +import { $bindings, bindingsFor } from '@/store/keybinds' import { $dismissedAutoProjectIds, filterVisibleProjects } from '@/store/layout' import { openPetGenerate } from '@/store/pet-generate' import { $projectTree, goToProject, openFolderAsProject, requestStartWorkSession } from '@/store/projects' @@ -328,7 +326,9 @@ const PaletteRow = memo(function PaletteRow({ const Icon = item.icon // The row's live keybind, else a static modifier-variant hint (⌘↵). One slot, // so every downstream `ml-auto` fallback below keeps working unchanged. - const combo = (item.action ? bindings[item.action]?.[0] : undefined) ?? item.comboHint + // `bindingsFor`, not a raw lookup: a plugin's action is contributed after + // $bindings was seeded, so its combo only resolves through the fallback chain. + const combo = (item.action ? bindingsFor(item.action, bindings)[0] : undefined) ?? item.comboHint // While ⌘/⌃ is held, a row with a modifier variant previews it: the label // swaps to the variant's copy so Enter reads as what it will actually do. const modPreview = modHeld && Boolean(item.modLabel) @@ -389,6 +389,7 @@ type NonConfigSettingsLabel = | 'keysSettings' | 'keysTools' | 'mcp' + | 'plugins' | 'providerAccounts' | 'providerApiKeys' @@ -423,6 +424,12 @@ const NON_CONFIG_SETTINGS: ReadonlyArray<{ labelKey: 'keysSettings', tab: 'keys&kview=settings' }, + { + icon: Package, + keywords: ['plugins', 'extensions', 'desktop plugins', 'addon', 'add-on'], + labelKey: 'plugins', + tab: 'plugins' + }, { icon: Archive, keywords: ['history', 'archived'], labelKey: 'archivedChats', tab: 'sessions' }, { icon: Info, keywords: ['version', 'about'], labelKey: 'about', tab: 'about' } ] @@ -762,14 +769,6 @@ function CommandPaletteBody({ onExited }: { onExited: () => void }) { } ] : []), - { - action: 'view.showTerminal', - icon: Terminal, - id: 'nav-terminal', - keywords: ['terminal', 'shell', 'console'], - label: t.keybinds.actions['view.showTerminal'], - run: () => setTerminalTakeover(true) - }, { action: 'nav.settings', icon: Settings, diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index 78304432c8..da4ff828cb 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -1,5 +1,5 @@ import { useStore } from '@nanostores/react' -import { computed } from 'nanostores' +import { atom, computed } from 'nanostores' import type { CSSProperties, ReactElement, PointerEvent as ReactPointerEvent } from 'react' import { PREVIEW_RAIL_MAX_WIDTH, PREVIEW_RAIL_MIN_WIDTH } from '@/app/chat/right-rail' @@ -13,20 +13,23 @@ import { LayoutTreeRoot } from '@/components/pane-shell/tree/renderer' import type { DoubleTapContext } from '@/components/pane-shell/tree/renderer/drag-session' import { $layoutTree, + bindPaneVisibility, + bindToolPaneCollapse, bindTreeSideVisibility, declareDefaultTree, dismissTreePane, dockPaneBeside, + isPaneVisible, markCollapsePane, mirrorLayoutTree, paneRootSide, registerLayoutResetHandler, registerPaneCloser, registerPaneOpener, + removeTreePane, resetLayoutTree, revealTreePane, - setPaneCollapsed, - setTreePaneHidden, + togglePaneVisible, watchContributedPanes } from '@/components/pane-shell/tree/store' import { SidebarProvider } from '@/components/ui/sidebar' @@ -36,9 +39,8 @@ import { useContributions } from '@/contrib/react/use-contributions' import { registry } from '@/contrib/registry' import { discoverRuntimePlugins } from '@/contrib/runtime-loader' import { sessionTitle as storedSessionTitle } from '@/lib/chat-runtime' -import { FileText, LayoutDashboard, PanelBottom, Zap } from '@/lib/icons' +import { FileText, LayoutDashboard, PanelBottom, Terminal, Zap } from '@/lib/icons' import { type KeybindContribution, KEYBINDS_AREA } from '@/lib/keybinds/actions' -import { Codecs, persistentAtom } from '@/lib/persisted' import { setYoloEnabled } from '@/lib/yolo-session' import { pruneComposerPopoutZones } from '@/store/composer-popout' import { @@ -54,7 +56,7 @@ import { SIDEBAR_MAX_WIDTH } from '@/store/layout' import { $previewOpenRequest, $previewTabs, closeRightRail } from '@/store/preview' -import { $reviewOpen, closeReview, REVIEW_PANE_ID } from '@/store/review' +import { $reviewOpen, closeReview, openReview, REVIEW_PANE_ID } from '@/store/review' import { $currentCwd, $selectedStoredSessionId, $sessions, $yoloActive, sessionMatchesStoredId } from '@/store/session' import { watchSessionPins } from '@/store/session-pin-sync' import { $statusbarVisible } from '@/store/statusbar-prefs' @@ -175,7 +177,11 @@ registry.registerMany([ // staying collapsed behind the ⌃` toggle. height sizes the fixed track (a // single-pane zone declaring a height is a fixed track — the preset weight // is moot): a short deck, not a third of the window. - data: { placement: 'bottom', height: '20vh', minHeight: '7.5rem', maxHeight: '80vh', revealOnPreset: true }, + // + // NO minHeight: a tool panel drags all the way down to its collapsed + // header (the sash floors it at COLLAPSED_ZONE_PX and folds the zone to + // its rail there). A real floor left a sliver of unusable terminal. + data: { placement: 'bottom', height: '20vh', maxHeight: '80vh', revealOnPreset: true }, render: () => }, { @@ -227,17 +233,6 @@ registry.registerMany([ maxWidth: FILE_BROWSER_MAX_WIDTH }, render: () => idle() - }, - { - // Optional chrome — in NO default layout. Adoption stacks it with the - // terminal; $logsOpen (default off, ⌘K "Toggle logs") reveals it. - id: 'logs', - area: 'panes', - title: 'logs', - // revealOnPreset: the Quad layout places logs, so applying it turns the - // logs pane on (like a ⌘K "Toggle logs") instead of leaving it collapsed. - data: { placement: 'bottom', height: '20vh', minHeight: '7.5rem', maxHeight: '80vh', revealOnPreset: true }, - render: () => idle() } ]) @@ -381,7 +376,7 @@ const QUAD_TREE = split( 'column', [ split('row', [group(['sessions', 'files']), group(['workspace'])], [1, 3]), - split('row', [group(['terminal']), group(['preview', 'review', 'logs'])], [1.4, 1]) + split('row', [group(['terminal']), group(['preview', 'review'])], [1.4, 1]) ], [3, 1] ) @@ -466,44 +461,13 @@ registerLayoutResetHandler(stackSessionTilesIntoMain) // toggle mirrors the root row. // --------------------------------------------------------------------------- -function bindPaneVisibility( - paneId: string, - $open: { get(): boolean; listen(fn: (open: boolean) => void): void }, - close?: () => void, - open?: () => void -) { - setTreePaneHidden(paneId, !$open.get()) - $open.listen(isOpen => setTreePaneHidden(paneId, !isOpen)) +// HIDE-STYLE PANES (files, review, preview): the binding lives in the tree +// store — bindPaneVisibility — alongside bindToolPaneCollapse, so both are +// testable against the real function instead of a copy. - // The tab menu's Close routes through the owning store (never dismissal), - // so the pane's toggle buttons stay truthful. - if (close) { - registerPaneCloser(paneId, close) - } - - // The opener is the mirror: preset application (revealOnPreset) shows the - // pane through the same store, so the toggle stays truthful. - if (open) { - registerPaneOpener(paneId, open) - } -} - -// TOOL PANELS (terminal, logs): like bindPaneVisibility but the toggle COLLAPSES -// the zone to a persistent rail (tab stays) instead of hiding it — the -// IntelliJ/VS-Code tool-window model. Restore routes back through `open` (rail -// click / chevron) so ⌃`/the button stay truthful; the tab's ✕ removes it. -function bindPaneCollapse( - paneId: string, - $open: { get(): boolean; listen(fn: (open: boolean) => void): void }, - close: () => void, - open: () => void -) { - markCollapsePane(paneId) - setPaneCollapsed(paneId, !$open.get()) - $open.listen(isOpen => setPaneCollapsed(paneId, !isOpen)) - registerPaneCloser(paneId, close) - registerPaneOpener(paneId, open) -} +// TOOL PANELS (terminal, logs): the binding lives in the tree store — +// bindToolPaneCollapse — so the boot rule it encodes is testable against the +// real function instead of a copy. See its docblock for the semantics. // SIDES have one source of truth: the TREE. The legacy $panesFlipped flag is // DERIVED from where the sessions zone actually sits (TitlebarControls maps @@ -556,24 +520,50 @@ const $hasWorkspace = computed($currentCwd, cwd => Boolean(cwd.trim())) // The tree pane's own presence tracks ⌘J directly, not just the column's // collapse — otherwise revealing a preview (which opens that shared column) // would drag the tree along with it. See revealPreview. +// +// Both get a CLOSER and an OPENER. The closer keeps ⌘J/⌘G truthful when the +// pane is closed from the tab menu; the opener is its mirror, so bringing the +// pane back through the tree (the toggle's reveal path, the rail, a preset) +// writes the store too. Without the opener the boolean went stale the moment +// anything but the toggle showed the pane — the divergence this whole change +// is about. bindPaneVisibility( 'files', - computed([$hasWorkspace, $fileBrowserOpen], (workspace, open) => workspace && open) + computed([$hasWorkspace, $fileBrowserOpen], (workspace, open) => workspace && open), + () => setFileBrowserOpen(false), + () => setFileBrowserOpen(true) ) // ⌘G — the review sidebar appears/disappears (and comes to the front). bindPaneVisibility( 'review', computed([$reviewOpen, $hasWorkspace], (open, workspace) => open && workspace), - closeReview + closeReview, + openReview ) // ⌃` / statusbar toggle — the terminal COLLAPSES to a rail (tab stays), not // hides; PTYs stay alive while collapsed (see PersistentTerminal). -bindPaneCollapse( +bindToolPaneCollapse( 'terminal', $terminalTakeover, () => setTerminalTakeover(false), () => setTerminalTakeover(true) ) +// ⌘K door onto the same pane the keybind and statusbar pill flip — was a +// one-way "open" row under Go to, so it never showed on/off and couldn't hide. +// Reads the TREE like every other pane toggle: `$terminalTakeover` stays true +// behind a stacked sibling tab or a minimized zone, which would light the row +// "on" for a terminal that isn't on screen. +registry.register( + paletteToggle({ + id: 'view.showTerminal', + label: 'Toggle terminal', + action: 'view.showTerminal', + icon: Terminal, + keywords: ['terminal', 'shell', 'console', 'pty'], + get: () => isPaneVisible('terminal'), + set: () => togglePaneVisible('terminal') + }) +) // Preview EXISTS only while something is previewed (old-shell semantics: // closing the last preview tab closes the pane; a new target opens + fronts @@ -583,23 +573,74 @@ const $previewVisible = computed($previewTabs, tabs => tabs.length > 0) bindPaneVisibility('preview', $previewVisible, closeRightRail) -// Logs are optional chrome: off by default, toggled from ⌘K, persisted. -const $logsOpen = persistentAtom('hermes.desktop.logsOpen', false, Codecs.bool) +// Logs are ⌘K-ONLY chrome: the pane contribution EXISTS only while $logsOpen +// is on. Off (the default) keeps logs out of the registry and the tree +// entirely — no secondary tab riding the terminal strip, no preset or +// adoption path that resurrects it. Session-only on purpose (not persisted): +// a fresh boot never re-opens logs automatically. The palette toggle is the +// single door in; tab ✕ / ⌘W / the toggle itself remove it again. +const $logsOpen = atom(false) + +let unregisterLogsPane: (() => void) | null = null + +const syncLogsPane = (open: boolean) => { + if (open) { + unregisterLogsPane ??= registry.register({ + id: 'logs', + area: 'panes', + title: 'logs', + // Same tool-panel sizing rule as the terminal above — no minHeight, so + // the sash floors it at COLLAPSED_ZONE_PX and folds the zone to its rail + // rather than leaving a sliver. dock: its OWN zone beside the terminal — + // never a tab in the terminal's strip. + data: { + placement: 'bottom', + dock: { pane: 'terminal', pos: 'right' }, + height: '20vh', + maxHeight: '80vh' + }, + render: () => idle() + }) + // Summoning logs is explicit intent — front it (un-dismisses if a ✕ close + // left a dismissal record behind). + revealTreePane('logs') + } else { + unregisterLogsPane?.() + unregisterLogsPane = null + + // No dismissal record — the next toggle-on must re-adopt cleanly. Also + // sweeps 'logs' out of persisted trees from before it was summon-only. + // Guarded: removePane rebuilds the tree even for an absent pane, and a + // no-op boot sweep would commit (and persist) a fresh identical tree. + const tree = $layoutTree.get() + + if (tree && allPaneIds(tree).includes('logs')) { + removeTreePane('logs') + } + } +} + +// Tool-panel tab semantics (✕ / ⌘W route through the store) so the palette +// toggle stays truthful either way. +markCollapsePane('logs') +registerPaneCloser('logs', () => $logsOpen.set(false)) +registerPaneOpener('logs', () => $logsOpen.set(true)) +syncLogsPane($logsOpen.get()) +$logsOpen.listen(syncLogsPane) -bindPaneCollapse( - 'logs', - $logsOpen, - () => $logsOpen.set(false), - () => $logsOpen.set(true) -) registry.register( paletteToggle({ id: 'logs.toggle', label: 'Toggle logs', icon: FileText, keywords: ['logs', 'agent log', 'tail', 'debug'], - get: () => $logsOpen.get(), - set: enabled => $logsOpen.set(enabled) + // On-screen, not the store's boolean. Summon-only keeps the two in step + // while logs sits in its own zone, but the user can still drag it into the + // terminal's strip or minimize its zone — and then `$logsOpen` reads true + // with nothing visible, so the row would show "on" and its press would + // spend itself re-asserting a value it already held. + get: () => isPaneVisible('logs'), + set: () => togglePaneVisible('logs') }) ) diff --git a/apps/desktop/src/app/contrib/panes.tsx b/apps/desktop/src/app/contrib/panes.tsx index 0f4d013151..36782638ab 100644 --- a/apps/desktop/src/app/contrib/panes.tsx +++ b/apps/desktop/src/app/contrib/panes.tsx @@ -31,8 +31,9 @@ import { $previewTarget, openPreview } from '@/store/preview' import { $currentCwd } from '@/store/session' // --------------------------------------------------------------------------- -// Logs — live agent-log tail. OPTIONAL chrome: not in any default layout, -// hidden until the ⌘K "Toggle logs" command opens it ($logsOpen). +// Logs — live agent-log tail. ⌘K-only chrome: the pane contribution exists +// only while the "Toggle logs" palette command has it summoned ($logsOpen in +// the controller) — never in a default layout, never a standing tab. // --------------------------------------------------------------------------- export function LogsPane() { diff --git a/apps/desktop/src/app/contrib/surfaces.tsx b/apps/desktop/src/app/contrib/surfaces.tsx index 852c5311a2..750248fbfe 100644 --- a/apps/desktop/src/app/contrib/surfaces.tsx +++ b/apps/desktop/src/app/contrib/surfaces.tsx @@ -9,7 +9,7 @@ import { useStore } from '@nanostores/react' import { type ComponentProps, lazy, memo, type ReactNode, Suspense, useMemo } from 'react' -import { Navigate, Route, Routes, useParams } from 'react-router-dom' +import { Navigate, Route, Routes, useParams } from 'react-router' import { ContribBoundary } from '@/contrib/react/boundary' import { useContributions } from '@/contrib/react/use-contributions' @@ -56,7 +56,7 @@ export const SidebarSurface = memo(function SidebarSurface({ export const TerminalSurface = memo(function TerminalSurface() { return ( -
+
) diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index 1306ffb641..a39a16dd86 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -11,7 +11,7 @@ import { useStore } from '@nanostores/react' import { useQueryClient } from '@tanstack/react-query' import { type CSSProperties, lazy, type ReactNode, Suspense, useCallback, useEffect, useMemo, useRef } from 'react' -import { useLocation, useNavigate } from 'react-router-dom' +import { useLocation, useNavigate } from 'react-router' import { formatRefValue } from '@/components/assistant-ui/directive-text' import { BootFailureOverlay } from '@/components/boot-failure-overlay' @@ -29,6 +29,7 @@ import { type ChatMessage, chatMessageText, preserveLocalAssistantErrors, toChat import { sessionMessagesSignature } from '@/lib/session-signatures' import { isMessagingSource } from '@/lib/session-source' import { latestSessionTodos } from '@/lib/todos' +import { activateWakeIndicator } from '@/lib/wake-indicator' import { playWakeSound } from '@/lib/wake-sound' import { $billingSettingsRequest } from '@/store/billing-block' import { requestVoiceConversationStart } from '@/store/composer' @@ -688,6 +689,7 @@ export function ContribWiring({ children }: { children: ReactNode }) { // Audible confirmation that the wake registered, before voice capture // starts. Gated by the shared sound-mute toggle. playWakeSound() + activateWakeIndicator() // Multi-profile routing: a wake phrase enrolled by another profile // re-homes the gateway to that profile first (live swap — same path diff --git a/apps/desktop/src/app/hooks/use-keybinds.ts b/apps/desktop/src/app/hooks/use-keybinds.ts index 1b4d535fa4..32d9caaba2 100644 --- a/apps/desktop/src/app/hooks/use-keybinds.ts +++ b/apps/desktop/src/app/hooks/use-keybinds.ts @@ -1,10 +1,16 @@ import { useEffect, useRef } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { closeActiveTab } from '@/app/chat/close-tab' -import { $terminalTakeover, setTerminalTakeover } from '@/app/right-sidebar/store' +import { setTerminalTakeover } from '@/app/right-sidebar/store' import { closeActiveTerminal, createTerminal, cycleTerminal } from '@/app/right-sidebar/terminal/terminals' -import { activateTreeTabSlot, cycleTreeTabInFocusedZone, layoutHasRootSide } from '@/components/pane-shell/tree/store' +import { + activateTreeTabSlot, + cycleTreeTabInFocusedZone, + isPaneVisible, + layoutHasRootSide, + togglePaneVisible +} from '@/components/pane-shell/tree/store' import { onReleaseTypingFocus } from '@/components/ui/keyboard-first' import { findBarClaimsCombo } from '@/lib/find-in-page' import { contributedKeybindHandler, PROFILE_SLOT_COUNT, SESSION_SLOT_COUNT } from '@/lib/keybinds/actions' @@ -184,22 +190,23 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { // terminal-on-bottom) would leave it a dead key, so it falls back to the // terminal there. The single "secondary panel" toggle. 'view.toggleRightSidebar': () => - layoutHasRootSide('right') ? toggleFileBrowserOpen() : setTerminalTakeover(!$terminalTakeover.get()), + layoutHasRootSide('right') ? toggleFileBrowserOpen() : togglePaneVisible('terminal'), 'view.toggleReview': toggleReview, 'view.toggleStatusbar': toggleStatusbarVisible, 'view.showFiles': showFiles, - 'view.showTerminal': () => setTerminalTakeover(!$terminalTakeover.get()), + 'view.showTerminal': () => togglePaneVisible('terminal'), // Create first so the pane's open-effect ensure sees a non-empty set and // doesn't also spawn one — net effect is exactly one fresh terminal. 'view.newTerminal': () => { createTerminal() setTerminalTakeover(true) }, - // Switch / close only act while the pane is open (no focus-scoping here, so - // this stands in for "terminal is showing"). - 'view.nextTerminal': () => $terminalTakeover.get() && cycleTerminal(1), - 'view.prevTerminal': () => $terminalTakeover.get() && cycleTerminal(-1), - 'view.closeTerminal': () => $terminalTakeover.get() && closeActiveTerminal(), + // Switch / close only act while the terminal is actually ON SCREEN — ask + // the tree, not the toggle store (which stays true behind a stacked + // sibling tab or a minimized zone). + 'view.nextTerminal': () => isPaneVisible('terminal') && cycleTerminal(1), + 'view.prevTerminal': () => isPaneVisible('terminal') && cycleTerminal(-1), + 'view.closeTerminal': () => isPaneVisible('terminal') && closeActiveTerminal(), 'view.flipPanes': togglePanesFlipped, // ⌘W: close the focused tab (terminal / preview target / zone tree tab). // On the main tab with session tabs stacked, it shifts the next one in — diff --git a/apps/desktop/src/app/hooks/use-route-enum-param.ts b/apps/desktop/src/app/hooks/use-route-enum-param.ts index 24de1dfe0f..ecef99cd60 100644 --- a/apps/desktop/src/app/hooks/use-route-enum-param.ts +++ b/apps/desktop/src/app/hooks/use-route-enum-param.ts @@ -1,5 +1,5 @@ import { useCallback, useMemo } from 'react' -import { useLocation, useNavigate } from 'react-router-dom' +import { useLocation, useNavigate } from 'react-router' // Read/write an enum-shaped URL search param (e.g. ?tab=foo). Used to make // tabbed views survive a refresh. Always navigates with replace so tab clicks diff --git a/apps/desktop/src/app/hooks/use-route-overlay-active.ts b/apps/desktop/src/app/hooks/use-route-overlay-active.ts index 261f9de15d..95db2f5528 100644 --- a/apps/desktop/src/app/hooks/use-route-overlay-active.ts +++ b/apps/desktop/src/app/hooks/use-route-overlay-active.ts @@ -1,4 +1,4 @@ -import { useLocation } from 'react-router-dom' +import { useLocation } from 'react-router' import { appViewForPath, isOverlayView } from '@/app/routes' diff --git a/apps/desktop/src/app/messaging/index.test.tsx b/apps/desktop/src/app/messaging/index.test.tsx index 025f5a58e8..6005131ccd 100644 --- a/apps/desktop/src/app/messaging/index.test.tsx +++ b/apps/desktop/src/app/messaging/index.test.tsx @@ -1,6 +1,6 @@ // @vitest-environment jsdom import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' -import { MemoryRouter } from 'react-router-dom' +import { MemoryRouter } from 'react-router' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { MessagingPlatformInfo } from '@/types/hermes' diff --git a/apps/desktop/src/app/pet-generate/pet-generate-content.tsx b/apps/desktop/src/app/pet-generate/pet-generate-content.tsx index 0f183678b1..aead2dd147 100644 --- a/apps/desktop/src/app/pet-generate/pet-generate-content.tsx +++ b/apps/desktop/src/app/pet-generate/pet-generate-content.tsx @@ -1,6 +1,6 @@ import { useStore } from '@nanostores/react' import { useEffect, useRef, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { useGatewayRequest } from '@/app/gateway/hooks/use-gateway-request' import { SETTINGS_ROUTE } from '@/app/routes' diff --git a/apps/desktop/src/app/profiles/index.tsx b/apps/desktop/src/app/profiles/index.tsx index 1db67b8778..0ec9993441 100644 --- a/apps/desktop/src/app/profiles/index.tsx +++ b/apps/desktop/src/app/profiles/index.tsx @@ -8,6 +8,7 @@ import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' import { getProfileSoul, type ProfileInfo, updateProfileSoul } from '@/hermes' import { useI18n } from '@/i18n' +import { displayPath } from '@/lib/display-path' import { AlertTriangle, Save } from '@/lib/icons' import { profileColorSoft, resolveProfileColor } from '@/lib/profile-color' import { normalize } from '@/lib/text' @@ -258,8 +259,11 @@ function ProfileDetail({ profile }: { profile: ProfileInfo }) { {profile.is_default && {p.defaultBadge}} {profile.has_env && .env}
-

- {profile.path} +

+ {displayPath(profile.path)}

diff --git a/apps/desktop/src/app/right-sidebar/files/remote-picker.tsx b/apps/desktop/src/app/right-sidebar/files/remote-picker.tsx index b502c014c7..3374348e20 100644 --- a/apps/desktop/src/app/right-sidebar/files/remote-picker.tsx +++ b/apps/desktop/src/app/right-sidebar/files/remote-picker.tsx @@ -5,6 +5,7 @@ import { Codicon } from '@/components/ui/codicon' import { Dialog, DialogContent, DialogDescription, DialogTitle } from '@/components/ui/dialog' import { useI18n } from '@/i18n' import { readDesktopDir, setDesktopFsRemotePicker } from '@/lib/desktop-fs' +import { displayPath, pathLeaf } from '@/lib/display-path' import { cn } from '@/lib/utils' function clean(path: string) { @@ -24,7 +25,7 @@ function parentDir(path: string) { } function pathName(path: string) { - return path.split('/').filter(Boolean).pop() || path + return pathLeaf(path) || path } interface PendingSelection { @@ -167,7 +168,7 @@ export function RemoteFolderPicker() {
-
{currentPath}
+
{displayPath(currentPath)}
+
+ + {TERMINAL_FONT_SUGGESTIONS.map(font => ( + +
+ + {copy.terminalFontPreview} + + ~/project git:main ❯ +
+
+ } + description={copy.terminalFontDesc} + title={copy.terminalFontTitle} + wide + /> + ) +} diff --git a/apps/desktop/src/app/settings/toolset-config-panel.test.tsx b/apps/desktop/src/app/settings/toolset-config-panel.test.tsx index fb057d6387..df509c198e 100644 --- a/apps/desktop/src/app/settings/toolset-config-panel.test.tsx +++ b/apps/desktop/src/app/settings/toolset-config-panel.test.tsx @@ -1,8 +1,8 @@ import { QueryClient, QueryClientProvider } from '@tanstack/react-query' import { cleanup, fireEvent, render as rtlRender, screen, waitFor } from '@testing-library/react' import type { ReactElement } from 'react' -import { MemoryRouter } from 'react-router-dom' -import type * as ReactRouterDom from 'react-router-dom' +import { MemoryRouter } from 'react-router' +import type * as ReactRouterDom from 'react-router' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { ToolsetConfig } from '@/types/hermes' @@ -11,7 +11,7 @@ import type { ToolsetConfig } from '@/types/hermes' // needs a router context. The navigate spy asserts the deep-link target. const navigateSpy = vi.fn() -vi.mock('react-router-dom', async importOriginal => ({ +vi.mock('react-router', async importOriginal => ({ ...(await importOriginal()), useNavigate: () => navigateSpy })) diff --git a/apps/desktop/src/app/settings/toolset-config-panel.tsx b/apps/desktop/src/app/settings/toolset-config-panel.tsx index 382e767095..bc9c150aad 100644 --- a/apps/desktop/src/app/settings/toolset-config-panel.tsx +++ b/apps/desktop/src/app/settings/toolset-config-panel.tsx @@ -1,5 +1,5 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { SETTINGS_ROUTE } from '@/app/routes' import { Button } from '@/components/ui/button' diff --git a/apps/desktop/src/app/settings/use-deep-link-highlight.ts b/apps/desktop/src/app/settings/use-deep-link-highlight.ts index a7f9e2fb78..215202ff8e 100644 --- a/apps/desktop/src/app/settings/use-deep-link-highlight.ts +++ b/apps/desktop/src/app/settings/use-deep-link-highlight.ts @@ -1,5 +1,5 @@ import { useEffect } from 'react' -import { useSearchParams } from 'react-router-dom' +import { useSearchParams } from 'react-router' interface DeepLinkHighlightOptions { param: string diff --git a/apps/desktop/src/app/shell/approval-mode-menu.test.tsx b/apps/desktop/src/app/shell/approval-mode-menu.test.tsx index 05a122f8bf..040471a19e 100644 --- a/apps/desktop/src/app/shell/approval-mode-menu.test.tsx +++ b/apps/desktop/src/app/shell/approval-mode-menu.test.tsx @@ -1,5 +1,5 @@ import { cleanup, fireEvent, render, screen, waitFor, within } from '@testing-library/react' -import { MemoryRouter } from 'react-router-dom' +import { MemoryRouter } from 'react-router' import { afterEach, beforeAll, describe, expect, it, vi } from 'vitest' import { StatusbarControls } from '@/app/shell/statusbar-controls' diff --git a/apps/desktop/src/app/shell/hooks/use-overlay-routing.ts b/apps/desktop/src/app/shell/hooks/use-overlay-routing.ts index bb92cfd10a..594e1eae9f 100644 --- a/apps/desktop/src/app/shell/hooks/use-overlay-routing.ts +++ b/apps/desktop/src/app/shell/hooks/use-overlay-routing.ts @@ -1,5 +1,5 @@ import { useCallback, useEffect, useMemo, useRef } from 'react' -import { useLocation, useNavigate } from 'react-router-dom' +import { useLocation, useNavigate } from 'react-router' import { type CommandCenterSection } from '@/app/command-center' import { diff --git a/apps/desktop/src/app/shell/hooks/use-statusbar-items.tsx b/apps/desktop/src/app/shell/hooks/use-statusbar-items.tsx index 0aa3c69fe3..a4081e9ac7 100644 --- a/apps/desktop/src/app/shell/hooks/use-statusbar-items.tsx +++ b/apps/desktop/src/app/shell/hooks/use-statusbar-items.tsx @@ -2,13 +2,14 @@ import { useStore } from '@nanostores/react' import { useCallback, useMemo } from 'react' import type { CommandCenterSection } from '@/app/command-center' -import { $terminalTakeover, setTerminalTakeover } from '@/app/right-sidebar/store' import { useApprovalModeStatusbarItem } from '@/app/shell/approval-mode-menu' import { ContextUsagePanel } from '@/app/shell/context-usage-panel' import { GatewayMenuPanel } from '@/app/shell/gateway-menu-panel' +import { $paneVisible, togglePaneVisible } from '@/components/pane-shell/tree/store' import { Codicon } from '@/components/ui/codicon' import { GlyphSpinner } from '@/components/ui/glyph-spinner' import { useI18n } from '@/i18n' +import { displayPath, pathLeaf } from '@/lib/display-path' import { Activity, AlertCircle, Clock, Command, FolderOpen, Globe, Hash, Loader2, Terminal } from '@/lib/icons' import type { RuntimeReadinessResult } from '@/lib/runtime-readiness' import { contextBarLabel, LiveDuration, usageContextLabel } from '@/lib/statusbar' @@ -29,6 +30,7 @@ import { $sessions, $sessionStartedAt, $turnStartedAt, + idsShareLineage, sessionMatchesStoredId, setCurrentUsage } from '@/store/session' @@ -50,13 +52,6 @@ import type { StatusbarItem } from '../statusbar-controls' const EMPTY_USAGE = { calls: 0, input: 0, output: 0, total: 0 } as const -function workspaceLabel(cwd: string): string { - const normalized = cwd.replace(/[\\/]+$/, '') - const leaf = normalized.split(/[\\/]/).filter(Boolean).pop() - - return leaf || cwd -} - interface StatusbarItemsOptions { agentsOpen: boolean chatOpen: boolean @@ -92,15 +87,15 @@ export function useStatusbarItems({ const fileMenu = t.fileMenu const primaryActiveSessionId = useStore($activeSessionId) const activeGatewayProfile = useStore($activeGatewayProfile) - const terminalTakeover = useStore($terminalTakeover) + // What the button paints and flips is whether the terminal is ON SCREEN — + // the takeover store alone stays true behind a stacked sibling tab or a + // minimized zone, which lit the button for a pane the user couldn't see. + const terminalShowing = useStore($paneVisible('terminal')) const primaryBusy = useStore($busy) - const currentCwd = useStore($currentCwd) - // Derive the workspace's project name from the already-cached project tree - // (backend truth via projects.*), so the status item labels by project without - // a second per-session copy of the same fact. Re-derives whenever the cwd or - // the tree changes; null (no named project) falls back to the cwd leaf below. - const projectTree = useStore($projectTree) - const projectName = useMemo(() => projectNameForCwd(currentCwd), [currentCwd, projectTree]) + // Draft / primary composer atom — used only while the focused surface is the + // primary (or a draft with no runtime slice yet). A focused TILE keeps its + // own cwd in `$sessionStates` and must not paint the primary's workspace. + const primaryCwd = useStore($currentCwd) const primaryUsage = useStore($currentUsage) const gatewayRestarting = useStore($gatewayRestarting) const primarySessionStartedAt = useStore($sessionStartedAt) @@ -128,15 +123,14 @@ export function useStatusbarItems({ // The FOCUSED session (interacted tile, else the primary — the same // derivation the titlebar title follows): every session-scoped readout - // below (context count, timers, busy pulse) tracks it, so clicking into a - // tile makes the statusbar describe THAT session. + // below (workspace cwd, context count, timers, busy pulse) tracks it, so + // clicking into a tile makes the statusbar describe THAT session. const focusedStoredSessionId = useStore($focusedStoredSessionId) const focusedRuntimeId = useStore($focusedRuntimeId) // `$focusedSessionState` is a projection of `$sessionStates`, which is // republished on EVERY message delta — tens of times a second during a turn. - // Only three fields are read off it here, so subscribing to the whole object - // re-ran this hook (and re-created all ~9 statusbar items) per token. Select - // each field individually so an unchanged readout bails out instead. + // Only the fields read here are selected, so an unchanged readout bails out + // instead of rebuilding all ~9 statusbar items per token. const focusedBusy = useStoreSelector($focusedSessionState, state => Boolean(state?.busy)) const focusedTurnStartedAt = useStoreSelector($focusedSessionState, state => state?.turnStartedAt ?? null) // `usage` is an object, so it can't be compared as a scalar. It IS however @@ -144,6 +138,14 @@ export function useStatusbarItems({ // reports new usage — far rarer than a delta — so its reference is a valid // bail-out key on its own. const focusedUsage = useStoreSelector($focusedSessionState, state => state?.usage ?? null) + const focusedStateCwd = useStoreSelector($focusedSessionState, state => state?.cwd?.trim() || '') + + // Runtime slices carry the stored id they were bound for. During a primary + // tab switch the runtime id can lag a frame behind the new selection — the + // slice still describes the PREVIOUS chat. Gate live cwd on ownership so we + // never paint session A's workspace while the tab already shows session B. + const focusedStateStoredId = useStoreSelector($focusedSessionState, state => state?.storedSessionId?.trim() || null) + const selectedStoredSessionId = useStore($selectedStoredSessionId) const primaryFocused = !focusedStoredSessionId || focusedStoredSessionId === selectedStoredSessionId @@ -156,16 +158,65 @@ export function useStatusbarItems({ const turnStartedAt = primaryFocused ? primaryTurnStartedAt : focusedTurnStartedAt - // A tile's session-start comes from its stored row (the cache only knows - // runtime state); seconds → ms. Only this ONE scalar is read off - // `$sessions`, so select it — a whole-list `useStore` re-ran the hook on - // every session-list write (title updates, poll refreshes, archives). + // A tile's session-start + cold cwd come from its stored row (the cache only + // knows runtime state). Only these scalars are read off `$sessions`, so + // select them — a whole-list `useStore` re-ran the hook on every session-list + // write (title updates, poll refreshes, archives). const focusedRowStartedAt = useStoreSelector($sessions, sessions => focusedStoredSessionId ? (sessions.find(s => sessionMatchesStoredId(s, focusedStoredSessionId))?.started_at ?? null) : null ) + const focusedRowCwd = useStoreSelector($sessions, sessions => { + if (!focusedStoredSessionId) { + return '' + } + + const row = sessions.find(s => sessionMatchesStoredId(s, focusedStoredSessionId)) + + return row?.cwd?.trim() || '' + }) + + // Live runtime cwd is authoritative once it belongs to the focused chat + // (agent can relocate mid-turn). Until then — cold tabs, mid-switch lag — + // the stored session row is the selection's project. Primary drafts fall + // through to `$currentCwd`. A focused TILE must never inherit the primary's + // workspace — an empty tile cwd stays empty rather than lying about another + // project's path. + // + // Lineage match is a pure derivation of ($sessions + the two ids). Select it + // so a session-list write only re-renders when the answer actually flips — + // not on every title/archive refresh of an unrelated row. + const liveCwdSharesFocusLineage = useStoreSelector($sessions, sessions => { + if (!focusedStoredSessionId || !focusedStateStoredId) { + return false + } + + return idsShareLineage(focusedStoredSessionId, focusedStateStoredId, sessions) + }) + + const liveCwdBelongsToFocus = + Boolean(focusedStateCwd) && + (!focusedStoredSessionId || + !focusedStateStoredId || + focusedStateStoredId === focusedStoredSessionId || + liveCwdSharesFocusLineage) + + const currentCwd = ( + (liveCwdBelongsToFocus ? focusedStateCwd : '') || + focusedRowCwd || + (primaryFocused ? primaryCwd : '') || + '' + ).trim() + + // Derive the workspace's project name from the already-cached project tree + // (backend truth via projects.*), so the status item labels by project without + // a second per-session copy of the same fact. Re-derives whenever the cwd or + // the tree changes; null (no named project) falls back to the cwd leaf below. + const projectTree = useStore($projectTree) + const projectName = useMemo(() => projectNameForCwd(currentCwd), [currentCwd, projectTree]) + const sessionStartedAt = primaryFocused ? primarySessionStartedAt : focusedRowStartedAt @@ -366,33 +417,33 @@ export function useStatusbarItems({ hidden: !currentCwd, icon: , id: 'workspace-cwd', - // Prefer the named project; fall back to the cwd leaf. The full cwd is - // always in the tooltip (`title` below), so hovering reveals where the - // session actually sits — the worktree/subfolder, not just the project. - label: projectName || (currentCwd ? workspaceLabel(currentCwd) : undefined), + // Prefer the named project; fall back to the cwd leaf. Hover tip uses + // the shared display formatter (home → ~) so statusbar and branch bar + // agree on how a path looks. + label: projectName || (currentCwd ? pathLeaf(currentCwd) : undefined), menuItems: currentCwd ? [ { id: 'copy-workspace-path', label: fileMenu.copyPath, onSelect: () => void copyFilePath(currentCwd), - title: currentCwd + title: displayPath(currentCwd) }, { id: 'reveal-workspace-finder', label: fileMenu.revealFileManager, onSelect: () => void revealFile(currentCwd), - title: currentCwd + title: displayPath(currentCwd) }, { id: 'reveal-workspace-sidebar', label: fileMenu.revealInSidebar, onSelect: () => revealFileInTree(currentCwd), - title: currentCwd + title: displayPath(currentCwd) } ] : undefined, - title: currentCwd || undefined, + title: currentCwd ? displayPath(currentCwd) : undefined, toggleLabel: copy.toggleWorkspace, variant: 'menu' }, @@ -506,12 +557,12 @@ export function useStatusbarItems({ }, { actionId: 'view.showTerminal', - className: `w-7 justify-center px-0${terminalTakeover ? ' bg-accent/55 text-foreground' : ''}`, + className: `w-7 justify-center px-0${terminalShowing ? ' bg-accent/55 text-foreground' : ''}`, hidden: !chatOpen, icon: , id: 'terminal', - onSelect: () => setTerminalTakeover(!$terminalTakeover.get()), - title: terminalTakeover ? copy.hideTerminal : copy.showTerminal, + onSelect: () => togglePaneVisible('terminal'), + title: terminalShowing ? copy.hideTerminal : copy.showTerminal, toggleLabel: copy.toggleTerminal, variant: 'action' }, @@ -533,7 +584,7 @@ export function useStatusbarItems({ requestGateway, sessionStartedAt, gatewayState, - terminalTakeover, + terminalShowing, turnStartedAt ] ) diff --git a/apps/desktop/src/app/shell/model-catalog-menu.test.tsx b/apps/desktop/src/app/shell/model-catalog-menu.test.tsx new file mode 100644 index 0000000000..27c6c18807 --- /dev/null +++ b/apps/desktop/src/app/shell/model-catalog-menu.test.tsx @@ -0,0 +1,108 @@ +import { QueryClient, QueryClientProvider } from '@tanstack/react-query' +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +import { DropdownMenu, DropdownMenuContent } from '@/components/ui/dropdown-menu' +import { + $modelVisibilityOpen, + $visibleModels, + modelVisibilityKey, + setModelVisibilityOpen, + setVisibleModels +} from '@/store/model-visibility' + +import { ModelCatalogMenu, type ModelMenuController } from './model-catalog-menu' + +// Radix calls these on open; jsdom doesn't implement them. +beforeAll(() => { + Element.prototype.scrollIntoView = vi.fn() + Element.prototype.hasPointerCapture = vi.fn(() => false) + Element.prototype.releasePointerCapture = vi.fn() +}) + +const getGlobalModelOptions = vi.fn() + +vi.mock('@/hermes', () => ({ + getGlobalModelOptions: (...args: unknown[]) => getGlobalModelOptions(...args), + setApiRequestProfile: vi.fn() +})) + +beforeEach(() => { + $visibleModels.set(null) + setModelVisibilityOpen(false) + getGlobalModelOptions.mockResolvedValue({ + providers: [{ models: ['gemini-3.1-pro', 'gemini-2.5-flash'], name: 'Google', slug: 'google' }] + }) +}) + +afterEach(() => { + cleanup() + vi.clearAllMocks() +}) + +// A minimal controller — these tests are about the CATALOG's own behaviour +// (what it lists, what it offers), not about what any host does with a pick. +function renderMenu() { + const select = vi.fn() + + const controller: ModelMenuController = { + applyPreset: vi.fn(), + current: { effort: '', fast: false, model: '', provider: '' }, + presetFor: () => ({}), + select, + setOptions: vi.fn() + } + + const client = new QueryClient({ defaultOptions: { queries: { retry: false } } }) + + render( + + + + + + + + ) + + return select +} + +// Curation is ONE global preference, so it belongs to the catalog rather than +// to whichever surface mounted it. If a host had to opt in, the composer and +// the kanban board would end up disagreeing about what "my models" means — +// which is exactly the drift extracting this component was meant to prevent. +describe('the catalog owns model curation', () => { + it('honours the stored Edit Models shortlist', async () => { + setVisibleModels(new Set([modelVisibilityKey('google', 'gemini-2.5-flash')])) + + renderMenu() + + await screen.findByText(/Gemini 2\.5 Flash/i) + expect(screen.queryByText(/Gemini 3\.1 Pro/i)).toBeNull() + }) + + it('still finds a hidden model by search — curation narrows the default view, not the catalog', async () => { + setVisibleModels(new Set([modelVisibilityKey('google', 'gemini-2.5-flash')])) + + renderMenu() + await screen.findByText(/Gemini 2\.5 Flash/i) + + const input = screen.getByRole('textbox', { name: 'Search models' }) + + fireEvent.change(input, { target: { value: 'gemini-3.1' } }) + + await vi.waitFor(() => { + expect(screen.queryByText(/Gemini 3\.1 Pro/i)).not.toBeNull() + }) + }) + + it('offers Edit Models without the host wiring it up', async () => { + renderMenu() + await screen.findByText(/Gemini 3\.1 Pro/i) + + fireEvent.click(screen.getByText('Edit Models…')) + + expect($modelVisibilityOpen.get()).toBe(true) + }) +}) diff --git a/apps/desktop/src/app/shell/model-catalog-menu.tsx b/apps/desktop/src/app/shell/model-catalog-menu.tsx new file mode 100644 index 0000000000..782bf579d8 --- /dev/null +++ b/apps/desktop/src/app/shell/model-catalog-menu.tsx @@ -0,0 +1,589 @@ +import { useStore } from '@nanostores/react' +import { useQuery } from '@tanstack/react-query' +import { createContext, type ReactNode, useContext, useEffect, useMemo, useRef, useState } from 'react' + +import { Codicon } from '@/components/ui/codicon' +import { DisclosureCaret } from '@/components/ui/disclosure-caret' +import { + DropdownMenuGroup, + DropdownMenuItem, + DropdownMenuLabel, + dropdownMenuRow, + DropdownMenuSearch, + dropdownMenuSectionLabel, + DropdownMenuSeparator, + DropdownMenuSub, + DropdownMenuSubTrigger +} from '@/components/ui/dropdown-menu' +import { HighlightMatches } from '@/components/ui/highlight-matches' +import { usePointerQuiet } from '@/components/ui/keyboard-first' +import { Skeleton } from '@/components/ui/skeleton' +import type { HermesGateway } from '@/hermes' +import { useI18n } from '@/i18n' +import { modelOptionsQueryKey, requestModelOptions } from '@/lib/model-options' +import { displayModelName, modelDisplayParts } from '@/lib/model-status-label' +import { DEFAULT_REASONING_EFFORT, reasoningEffortLabel } from '@/lib/reasoning-effort' +import { normalize } from '@/lib/text' +import { cn } from '@/lib/utils' +import { + $visibleModels, + collapseModelFamilies, + DEFAULT_VISIBLE_PER_PROVIDER, + effectiveVisibleKeys, + type ModelFamily, + modelVisibilityKey, + setModelVisibilityOpen +} from '@/store/model-visibility' +import { $collapsedProviders, toggleCollapsedProvider } from '@/store/provider-collapse' +import { $defaultReasoningEffort } from '@/store/session' +import type { ModelOptionProvider, ModelOptionsResponse } from '@/types/hermes' + +import { type FastControl, ModelEditSubmenu, resolveFastControl } from './model-edit-submenu' + +// Lets the host dropdown (model-pill, a kanban field trigger, …) hand the panel +// a way to dismiss itself so clicking a model row commits + closes, while the +// hover-revealed edit submenu (reasoning/fast) stays open to play with (its +// items preventDefault on select). +export const ModelMenuCloseContext = createContext<() => void>(() => {}) + +/** One model choice, everything a caller needs to act on a selection. + * `effort` is '' for "inherit the default" and 'none' for thinking off. */ +export interface ModelChoice { + effort: string + fast: boolean + model: string + provider: string +} + +/** + * What a surface DOES with the catalog. The menu renders and navigates; the + * controller owns meaning — the composer writes through to a live session, + * the kanban override just holds a value in dialog state. + * + * `presetFor` supplies the remembered settings shown on a non-active row. + * Returning `{}` is fine — the row then shows Hermes' defaults. + */ +export interface ModelMenuController { + /** Restore a model's remembered settings after it is selected. Separate from + * `setOptions` because it is one atomic "apply this model's preset" write, + * not a user editing one control — surfaces that write through to a session + * need to batch it. Values are already capability-gated by the menu. */ + applyPreset: (preset: { effort?: string; fast?: boolean }, row: { model: string; provider: string }) => void + current: ModelChoice + presetFor: (provider: string, model: string) => { effort?: string; fast?: boolean } + /** Commit a model row. Return false to abort (a failed session switch). */ + select: (model: string, provider: string) => Promise | void + /** Edit ONE option on a row. `isActive` says whether it's the current model. */ + setOptions: ( + patch: { effort?: string; fast?: boolean }, + row: { isActive: boolean; model: string; provider: string } + ) => void +} + +interface ModelCatalogMenuProps { + controller: ModelMenuController + /** Rows appended under the catalog (Refresh Models, Edit Models, …). */ + footer?: ReactNode + gateway?: HermesGateway + /** Render the virtual `moa` provider's presets as a selectable section. + * Off for override surfaces, where a MoA preset isn't a worker model. */ + includeMoa?: boolean + profile?: string + /** Session whose catalog to fetch. A live session's catalog can differ from + * the profile-global one, and the app invalidates the SESSION-scoped query + * key on model changes — a surface bound to a session must pass it or its + * menu goes stale. Detached surfaces (per-task overrides) omit it. */ + sessionId?: null | string +} + +interface ProviderGroup { + families: ModelFamily[] + provider: ModelOptionProvider +} + +/** + * THE model catalog menu: searchable, provider-grouped, `-fast` families + * collapsed to one row, per-row hover submenu for thinking/effort/fast, full + * keyboard selection. Shared verbatim by the composer's model pill and by + * plugin surfaces that pick a model without a session behind it — so the two + * can never drift apart. + */ +export function ModelCatalogMenu({ + controller, + footer, + gateway, + includeMoa = false, + profile = 'default', + sessionId = null +}: ModelCatalogMenuProps) { + const { t } = useI18n() + const copy = t.shell.modelMenu + const closeMenu = useContext(ModelMenuCloseContext) + const [search, setSearch] = useState('') + const collapsedProviders = useStoreCollapsed() + const defaultEffort = useDefaultEffort() + // Which models the user curated in Edit Models. Read HERE rather than taken + // as a prop: it's one global preference, so every surface that shows a + // catalog must show the same shortlist. A per-caller opt-in is how the board + // and the composer would end up disagreeing about what "my models" means. + const visibleModels = useStore($visibleModels) + + const modelOptions = useQuery({ + queryKey: modelOptionsQueryKey(profile, sessionId), + // Gateway-first even with no session: a connected (possibly remote) + // gateway owns the model catalog, including virtual providers the local + // REST fallback can't know about (#53817). + queryFn: (): Promise => requestModelOptions({ gateway, sessionId }) + }) + + const loading = modelOptions.isPending && !modelOptions.data + + const error = modelOptions.error + ? modelOptions.error instanceof Error + ? modelOptions.error.message + : String(modelOptions.error) + : null + + const providers = modelOptions.data?.providers + + // The catalog carries MoA presets as a virtual `moa` provider row. Keep it + // out of the main groups so presets never show up twice. + const moaPresets = useMemo( + () => (includeMoa ? (providers?.find(p => p.slug.toLowerCase() === 'moa')?.models ?? []) : []), + [providers, includeMoa] + ) + + const pickerProviders = useMemo( + () => providers?.filter(provider => provider.slug.toLowerCase() !== 'moa') ?? [], + [providers] + ) + + const current = controller.current + + // Resolve visibility HERE, against the catalog we actually fetched: an empty + // provider list would otherwise resolve to an empty key set that reads as + // "user hid everything" and blanks the menu on first open. + const shownKeys = useMemo( + () => effectiveVisibleKeys(visibleModels, pickerProviders), + [visibleModels, pickerProviders] + ) + + const groups = useMemo( + () => groupModels(pickerProviders, search, { model: current.model, provider: current.provider }, shownKeys), + [pickerProviders, search, current.model, current.provider, shownKeys] + ) + + const q = normalize(search) + + // Presets are searchable rows like everything else — an unfiltered preset + // sitting under zero model matches would otherwise become the "first match" + // Enter commits. + const shownMoaPresets = useMemo( + () => (q ? moaPresets.filter(preset => `moa ${preset}`.toLowerCase().includes(q)) : moaPresets), + [moaPresets, q] + ) + + const selectFamily = async (family: ModelFamily, provider: ModelOptionProvider) => { + const caps = provider.capabilities?.[family.id] + const preset = controller.presetFor(provider.slug, family.id) + + // Variant-fast models (no speed param) express "fast" as a separate `-fast` + // id, so honor the remembered preset by selecting that sibling. Param-fast + // is applied through setOptions below instead. + const variantFast = !(caps?.fast ?? false) && !!family.fastId + const targetId = variantFast && preset.fast === true ? family.fastId! : family.id + + if ((await controller.select(targetId, provider.slug)) === false) { + return + } + + controller.applyPreset( + { + effort: (caps?.reasoning ?? true) ? (preset.effort ?? defaultEffort) : undefined, + fast: (caps?.fast ?? false) ? (preset.fast ?? false) : undefined + }, + { model: family.id, provider: provider.slug } + ) + } + + const selectMoaPreset = async (preset: string) => { + if ((await controller.select(preset, 'moa')) === false) { + return + } + + closeMenu() + } + + // ── Keyboard selection (cmdk semantics on a Radix menu) ─────────────────── + // One flat list mirroring EXACTLY what's rendered (collapse, filter, presets), + // so the selection can never sit on a hidden row. + type KbRow = + | { family: ModelFamily; key: string; kind: 'family'; provider: ModelOptionProvider } + | { key: string; kind: 'moa'; preset: string } + + const kbRows = useMemo( + () => [ + ...groups.flatMap(group => + collapsedProviders.includes(group.provider.slug) && !search + ? [] + : group.families.map((family): KbRow => ({ + family, + key: `${group.provider.slug}:${family.id}`, + kind: 'family', + provider: group.provider + })) + ), + ...shownMoaPresets.map((preset): KbRow => ({ key: `moa:${preset}`, kind: 'moa', preset })) + ], + [groups, collapsedProviders, search, shownMoaPresets] + ) + + const [kbOverride, setKbOverride] = useState(null) + // A parked cursor is not a cursor in use: until the mouse actually moves, + // hover can't take rows out from under the keyboard. + const pointerQuiet = usePointerQuiet() + + const currentKey = current.provider === 'moa' ? `moa:${current.model}` : `${current.provider}:${current.model}` + + const autoIndex = q + ? kbRows.length > 0 + ? 0 + : -1 + : kbRows.findIndex(row => row.key === currentKey || (row.kind === 'family' && row.family.fastId === current.model)) + + const kbIndex = kbOverride !== null && kbOverride < kbRows.length ? kbOverride : autoIndex + const kbActiveKey = kbIndex >= 0 ? kbRows[kbIndex].key : null + + const stepKb = (delta: -1 | 1) => { + if (kbRows.length === 0) { + return + } + + const from = kbIndex >= 0 ? kbIndex : delta === 1 ? -1 : 0 + + setKbOverride((from + delta + kbRows.length) % kbRows.length) + } + + const commitKbRow = () => { + const row = kbIndex >= 0 ? kbRows[kbIndex] : undefined + + if (!row) { + return + } + + if (row.kind === 'moa') { + void selectMoaPreset(row.preset) + + return + } + + if (row.key !== currentKey && row.family.fastId !== current.model) { + void selectFamily(row.family, row.provider) + } + + closeMenu() + } + + // Keep the selected row in view while arrowing through the scrollable list. + const listRef = useRef(null) + + useEffect(() => { + listRef.current?.querySelector('[data-kb-active]')?.scrollIntoView({ block: 'nearest' }) + }, [kbActiveKey]) + + const kbRowProps = (key: string) => { + const active = kbActiveKey === key + + return { + className: cn(dropdownMenuRow, active && 'bg-(--ui-control-active-background) text-foreground'), + ...(active ? { 'data-kb-active': '' } : {}) + } + } + + // Rows are hover-selectable, so they go inert with the pointer. + const quietRows = pointerQuiet && 'pointer-events-none' + + return ( + <> + { + // Claim arrows and Enter from Radix so DOM focus stays in the input + // and Enter commits the highlighted row without a DownArrow first. + if (event.key === 'ArrowDown' || event.key === 'ArrowUp') { + event.preventDefault() + event.stopPropagation() + stepKb(event.key === 'ArrowDown' ? 1 : -1) + } else if (event.key === 'Enter') { + event.preventDefault() + event.stopPropagation() + commitKbRow() + } + }} + onValueChange={value => { + setSearch(value) + setKbOverride(null) + }} + placeholder={copy.search} + value={search} + /> + + + + {loading ? ( + + {Array.from({ length: 4 }, (_, index) => ( + event.preventDefault()} + > + + + ))} + + ) : error ? ( + + {error} + + ) : groups.length === 0 && moaPresets.length === 0 ? ( + + {copy.noModels} + + ) : ( +
+ {groups.map(group => { + const slug = group.provider.slug + + // Collapsed when the user stored it (and not while searching, which + // spans every model regardless of collapse state). + const collapsed = collapsedProviders.includes(slug) && !search + + return ( + + { + event.preventDefault() + toggleCollapsedProvider(slug) + }} + textValue="" + > + + + + + + {!collapsed && + group.families.map(family => { + // The active id may be the base or its -fast sibling; either + // way this one family row represents both. + const activeId = + group.provider.slug === current.provider && + (current.model === family.id || current.model === family.fastId) + ? current.model + : null + + const isCurrent = activeId !== null + const name = modelDisplayParts(family.id).name + const caps = group.provider.capabilities?.[family.id] + + // Effective settings for this row: the live choice when it's + // the active model, otherwise its remembered preset. Row + // label AND submenu read from these so they never disagree. + const preset = controller.presetFor(group.provider.slug, family.id) + const effEffort = isCurrent ? current.effort : (preset.effort ?? '') + const effFast = isCurrent ? current.fast : (preset.fast ?? false) + + const fastControl: FastControl = resolveFastControl( + activeId ?? family.id, + group.provider.models ?? [], + caps?.fast ?? false, + effFast + ) + + const meta = [ + fastControl.kind !== 'none' && fastControl.on ? copy.fast : null, + (caps?.reasoning ?? true) ? reasoningEffortLabel(effEffort || defaultEffort) : null + ] + .filter(Boolean) + .join(' ') + + // Clicking the row commits the model and closes; the edit + // submenu (reasoning/fast) is reached by HOVER, so you can + // tweak those without the click dismissing everything. + const activate = () => { + if (!isCurrent) { + void selectFamily(family, group.provider) + } + + closeMenu() + } + + return ( + + { + if (event.key === 'Enter' || event.key === ' ') { + activate() + } + }} + {...kbRowProps(`${group.provider.slug}:${family.id}`)} + > + + + {meta ? {meta} : null} + + {isCurrent ? ( + + ) : null} + + controller.select(nextModel, group.provider.slug)} + onSetOptions={patch => + controller.setOptions(patch, { + isActive: isCurrent, + model: family.id, + provider: group.provider.slug + }) + } + provider={group.provider.slug} + reasoning={caps?.reasoning ?? true} + /> + + ) + })} + + ) + })} +
+ )} + + {shownMoaPresets.length > 0 ? ( +
+ + MoA presets + {shownMoaPresets.map(preset => { + const isCurrentMoa = current.provider === 'moa' && current.model === preset + + return ( + { + event.preventDefault() + void selectMoaPreset(preset) + }} + {...kbRowProps(`moa:${preset}`)} + > + + MoA: + + {isCurrentMoa ? : null} + + ) + })} +
+ ) : null} + + {/* Curation belongs to the catalog, not to one host: wherever you can + pick a model you can say which models you want, and the shortlist is + the same everywhere because it's one stored preference. It shares the + host footer's group rather than opening a second one, so a host that + contributes rows (the composer's Refresh Models) keeps the single + trailing block it has always rendered. */} + + {footer} + setModelVisibilityOpen(true)} + > + + {copy.editModels} + + + ) +} + +/** Re-exported so callers building a footer row match the catalog's rows. */ +export { dropdownMenuRow } + +// Collapsed we show the user's chosen models (or the curated default); typing +// spans every available model so anything is reachable past the cut. A search +// is itself a narrowing action, so we do NOT cap per-provider matches. +function groupModels( + providers: ModelOptionProvider[], + search: string, + current: { model: string; provider: string }, + visible: Set | null +): ProviderGroup[] { + const q = normalize(search) + const groups: ProviderGroup[] = [] + + for (const provider of providers) { + const allFamilies = collapseModelFamilies(provider.models ?? []) + + if (allFamilies.length === 0) { + continue + } + + const matches = (family: ModelFamily) => + `${family.id} ${family.fastId ?? ''} ${provider.name} ${provider.slug} ${displayModelName(family.id)}` + .toLowerCase() + .includes(q) + + let shown: Set + + if (q) { + // Search spans every family, regardless of visibility. + shown = new Set(allFamilies.filter(matches).map(family => family.id)) + } else if (visible) { + // User has customized which models show — honor their selection exactly. + shown = new Set( + allFamilies.filter(family => visible.has(modelVisibilityKey(provider.slug, family.id))).map(family => family.id) + ) + } else { + shown = new Set(allFamilies.slice(0, DEFAULT_VISIBLE_PER_PROVIDER).map(family => family.id)) + } + + // Always include the active model — but keep every row in the provider's + // stable curated order, so selecting a model can't shuffle the list. While + // SEARCHING the pin is skipped: a query means "show me matches". + const activeId = + !q && provider.slug === current.provider && current.model + ? allFamilies.find(family => family.id === current.model || family.fastId === current.model)?.id + : undefined + + const families = allFamilies.filter(family => shown.has(family.id) || family.id === activeId) + + if (families.length > 0) { + groups.push({ families, provider }) + } + } + + // Stable, logical group order: alphabetical by provider name. (The backend + // floats the current provider first, which would reshuffle on every switch.) + groups.sort((a, b) => a.provider.name.localeCompare(b.provider.name)) + + return groups +} + +// Small hooks kept at the bottom so the component reads top-down. +function useStoreCollapsed(): string[] { + return useStore($collapsedProviders) +} + +function useDefaultEffort(): string { + return useStore($defaultReasoningEffort) || DEFAULT_REASONING_EFFORT +} diff --git a/apps/desktop/src/app/shell/model-edit-submenu.test.tsx b/apps/desktop/src/app/shell/model-edit-submenu.test.tsx index 4e552303b5..3685f8e44a 100644 --- a/apps/desktop/src/app/shell/model-edit-submenu.test.tsx +++ b/apps/desktop/src/app/shell/model-edit-submenu.test.tsx @@ -1,5 +1,5 @@ import { cleanup, fireEvent, render, screen } from '@testing-library/react' -import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeAll, describe, expect, it, vi } from 'vitest' import { DropdownMenu, @@ -7,26 +7,9 @@ import { DropdownMenuSub, DropdownMenuSubTrigger } from '@/components/ui/dropdown-menu' -import type * as HermesApi from '@/hermes' -import { $modelPresets, getModelPreset } from '@/store/model-presets' -import { - $activeSessionId, - $currentFastMode, - $currentReasoningEffort, - getCurrentModelSource, - setCurrentFastMode, - setCurrentModelSource, - setCurrentReasoningEffort -} from '@/store/session' import { type FastControl, ModelEditSubmenu } from './model-edit-submenu' -vi.mock('@/hermes', async importOriginal => { - const actual = await importOriginal() - - return { ...actual, setApiRequestProfile: vi.fn() } -}) - // Radix calls these on open; jsdom doesn't implement them. beforeAll(() => { Element.prototype.scrollIntoView = vi.fn() @@ -34,35 +17,36 @@ beforeAll(() => { Element.prototype.releasePointerCapture = vi.fn() }) -beforeEach(() => { - $modelPresets.set({}) - $activeSessionId.set(null) - setCurrentFastMode(false) - setCurrentModelSource('') - setCurrentReasoningEffort('') -}) - afterEach(() => { cleanup() vi.clearAllMocks() }) // Render the submenu inside an open menu/sub so its content (switches) mounts. -function renderSubmenu(opts: { fastControl: FastControl; reasoning: boolean; requestGateway: () => Promise }) { +function renderSubmenu(opts: { + defaultEffort?: string + effort?: string + fastControl: FastControl + isActive?: boolean + onSelectModel?: (model: string) => void + onSetOptions: (patch: { effort?: string; fast?: boolean }) => void + reasoning: boolean +}) { return render( edit @@ -70,43 +54,78 @@ function renderSubmenu(opts: { fastControl: FastControl; reasoning: boolean; req ) } -// Regression: editing the active row before a live session exists must stay -// preset-only — the gateway's config.set falls back to global config when no -// session matches, so it must not be called. (Caught in the second review.) -describe('ModelEditSubmenu no-session guard', () => { - it('param fast: records explicit off in the draft but skips the gateway without a session', () => { - const requestGateway = vi.fn().mockResolvedValue({}) - setCurrentFastMode(true) - renderSubmenu({ fastControl: { kind: 'param', on: true }, reasoning: false, requestGateway }) +// The submenu is PURE: it reports edits and never writes to a session, a +// preset store, or the gateway. That's the invariant that lets the same +// component drive a live chat session AND a detached per-task override — if it +// ever writes directly again, picking an effort for a kanban card would reach +// over and change the user's live chat. +describe('ModelEditSubmenu reports edits without performing them', () => { + it('param fast: reports the toggle', () => { + const onSetOptions = vi.fn() + renderSubmenu({ fastControl: { kind: 'param', on: true }, onSetOptions, reasoning: false }) fireEvent.click(screen.getByRole('switch')) - expect(getModelPreset('p1', 'm1').fast).toBe(false) - expect($currentFastMode.get()).toBe(false) - expect(getCurrentModelSource()).toBe('manual') - expect(requestGateway).not.toHaveBeenCalled() + expect(onSetOptions).toHaveBeenCalledWith({ fast: false }) }) - it('reasoning: records the preset but skips the gateway without a session', () => { - const requestGateway = vi.fn().mockResolvedValue({}) - renderSubmenu({ fastControl: { kind: 'none' }, reasoning: true, requestGateway }) + it('thinking: toggling off reports the none level', () => { + const onSetOptions = vi.fn() + renderSubmenu({ fastControl: { kind: 'none' }, onSetOptions, reasoning: true }) - // Thinking starts on (medium); toggling it off routes through patchReasoning. + // Thinking starts on (medium); toggling it off reports 'none'. fireEvent.click(screen.getByRole('switch')) - expect(getModelPreset('p1', 'm1').effort).toBe('none') - expect($currentReasoningEffort.get()).toBe('none') - expect(getCurrentModelSource()).toBe('manual') - expect(requestGateway).not.toHaveBeenCalled() + expect(onSetOptions).toHaveBeenCalledWith({ effort: 'none' }) }) - it('param fast: pushes to the gateway once a session is active', async () => { - const requestGateway = vi.fn().mockResolvedValue({}) - $activeSessionId.set('sess1') - renderSubmenu({ fastControl: { kind: 'param', on: false }, reasoning: false, requestGateway }) + it('thinking: toggling back on restores the row level, not the hardcoded default', () => { + const onSetOptions = vi.fn() + renderSubmenu({ + defaultEffort: 'high', + effort: 'none', + fastControl: { kind: 'none' }, + onSetOptions, + reasoning: true + }) fireEvent.click(screen.getByRole('switch')) - expect(requestGateway).toHaveBeenCalledWith('config.set', { key: 'fast', session_id: 'sess1', value: 'fast' }) + expect(onSetOptions).toHaveBeenCalledWith({ effort: 'high' }) + }) + + it('variant fast: swaps the model only when the row is active', () => { + const onSelectModel = vi.fn() + const onSetOptions = vi.fn() + + renderSubmenu({ + fastControl: { baseId: 'm1', fastId: 'm1-fast', kind: 'variant', on: false }, + isActive: false, + onSelectModel, + onSetOptions, + reasoning: false + }) + + fireEvent.click(screen.getByRole('switch')) + + // Inactive rows stay preference-only — no model switch. + expect(onSetOptions).toHaveBeenCalledWith({ fast: true }) + expect(onSelectModel).not.toHaveBeenCalled() + }) + + it('variant fast: active row swaps to the -fast sibling', () => { + const onSelectModel = vi.fn() + const onSetOptions = vi.fn() + + renderSubmenu({ + fastControl: { baseId: 'm1', fastId: 'm1-fast', kind: 'variant', on: false }, + onSelectModel, + onSetOptions, + reasoning: false + }) + + fireEvent.click(screen.getByRole('switch')) + + expect(onSelectModel).toHaveBeenCalledWith('m1-fast') }) }) diff --git a/apps/desktop/src/app/shell/model-edit-submenu.tsx b/apps/desktop/src/app/shell/model-edit-submenu.tsx index 6fca3997a0..1d5b5dc598 100644 --- a/apps/desktop/src/app/shell/model-edit-submenu.tsx +++ b/apps/desktop/src/app/shell/model-edit-submenu.tsx @@ -1,6 +1,3 @@ -import { useStore } from '@nanostores/react' - -import { useSessionView } from '@/app/chat/session-view' import { DropdownMenuItem, DropdownMenuLabel, @@ -13,21 +10,7 @@ import { } from '@/components/ui/dropdown-menu' import { Switch } from '@/components/ui/switch' import { useI18n } from '@/i18n' -import { - DEFAULT_REASONING_EFFORT, - isThinkingEnabled, - REASONING_EFFORTS, - resolveReasoningEffort -} from '@/lib/reasoning-effort' -import { setModelPreset } from '@/store/model-presets' -import { notifyError } from '@/store/notifications' -import { - $defaultReasoningEffort, - markComposerSelectionManual, - setCurrentFastMode, - setCurrentReasoningEffort -} from '@/store/session' -import { sessionTileDelegate } from '@/store/session-states' +import { isThinkingEnabled, REASONING_EFFORTS, resolveReasoningEffort } from '@/lib/reasoning-effort' // Hermes' real reasoning levels live in lib/reasoning-effort; `none` is owned // by the Thinking toggle, not the radio. @@ -37,9 +20,7 @@ import { sessionTileDelegate } from '@/store/session-states' * - `variant`: a separate `…-fast` sibling model selected via the model field. */ export type FastControl = - | { kind: 'none' } - | { kind: 'param'; on: boolean } - | { kind: 'variant'; baseId: string; fastId: string; on: boolean } + { kind: 'none' } | { kind: 'param'; on: boolean } | { kind: 'variant'; baseId: string; fastId: string; on: boolean } /** Resolve the fast mechanism for a model: prefer the speed=fast parameter * when the backend supports it, else fall back to a `…-fast` sibling model. */ @@ -78,6 +59,9 @@ export function resolveFastControl( } interface ModelEditSubmenuProps { + /** The profile's configured default effort — what an unset row inherits. + * Passed in (not read from a store) so this submenu stays pure. */ + defaultEffort: string /** This row's effective reasoning effort (live for the active model, else its * preset) — the submenu shows and edits from this, never the raw session. */ effort: string @@ -85,15 +69,19 @@ interface ModelEditSubmenuProps { fastControl: FastControl /** Whether this row's model is the active one. */ isActive: boolean - /** This row's model id — edits persist as its global preset. */ + /** This row's model id. */ model: string /** Switch to a specific model id (used to swap base ⇄ -fast variant). */ - onSelectModel: (model: string) => Promise | void - /** This row's provider slug — edits persist as its global preset. */ + onSelectModel: (model: string) => Promise | void + /** Report an option change. This submenu is PURE: it never writes to a + * session, a preset store, or the gateway itself — the owning surface's + * controller decides what an edit means. That's what lets the same submenu + * drive a live chat session and a detached per-task override. */ + onSetOptions: (patch: { effort?: string; fast?: boolean }) => void + /** This row's provider slug. */ provider: string /** Whether this model supports reasoning effort. */ reasoning: boolean - requestGateway: (method: string, params?: Record) => Promise } export function ModelEditSubmenu(props: ModelEditSubmenuProps) { @@ -110,72 +98,26 @@ export function ModelEditSubmenu(props: ModelEditSubmenuProps) { } function ModelEditSubmenuBody({ + defaultEffort, effort, fastControl, isActive, - model, onSelectModel, - provider, - reasoning, - requestGateway + onSetOptions, + reasoning }: ModelEditSubmenuProps) { const { t } = useI18n() const copy = t.shell.modelOptions - const view = useSessionView() - const activeSessionId = useStore(view.$runtimeId) - const touchesPrimary = view.kind === 'primary' - const defaultEffort = useStore($defaultReasoningEffort) || DEFAULT_REASONING_EFFORT const effortValue = resolveReasoningEffort(effort, defaultEffort) const thinkingOn = isThinkingEnabled(effort, defaultEffort) - // Editing always records the model's global preset (keyed by provider::model, - // not per-surface — a tile edit re-applies to that model everywhere); the - // active model also gets it pushed onto its OWN session (primary → globals, - // tile → its slice). Non-active edits stay preset-only — no model switch. - const patchReasoning = async (next: string) => { - setModelPreset(provider, model, { effort: next }) - - if (!isActive) { - return - } - - if (touchesPrimary) { - markComposerSelectionManual() - setCurrentReasoningEffort(next) - } else if (activeSessionId) { - sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, reasoningEffort: next })) - } - - // Preset-only without a session: `isActive` holds for the global/default - // row pre-session, and the gateway's `config.set` falls back to global - // config when none matches — so don't reach it (preset + optimistic store - // are the whole effect). Same guard in applyModelPreset / setFast. - if (!activeSessionId) { - return - } - - try { - await requestGateway('config.set', { key: 'reasoning', session_id: activeSessionId, value: next }) - } catch (err) { - if (touchesPrimary) { - setCurrentReasoningEffort(effort) - } else if (activeSessionId) { - sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, reasoningEffort: effort })) - } - - setModelPreset(provider, model, { effort }) - notifyError(err, copy.updateFailed) - } - } - const setFast = (enabled: boolean) => { if (fastControl.kind === 'variant') { - // Fast is a separate model id. Record the choice on the base model's - // preset (selectFamily picks the `-fast` sibling later when set), and - // only swap models now if this is the active row — inactive edits must - // stay preset-only, same as the param path below. - setModelPreset(provider, fastControl.baseId, { fast: enabled }) + // Fast is a separate model id. Report the choice so the controller can + // record it against the base model, and only swap models now if this is + // the active row — inactive edits stay preference-only. + onSetOptions({ fast: enabled }) if (isActive) { void onSelectModel(enabled ? fastControl.fastId : fastControl.baseId) @@ -185,41 +127,7 @@ function ModelEditSubmenuBody({ } if (fastControl.kind === 'param') { - setModelPreset(provider, model, { fast: enabled }) - - if (!isActive) { - return - } - - if (touchesPrimary) { - markComposerSelectionManual() - setCurrentFastMode(enabled) - } else if (activeSessionId) { - sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, fast: enabled })) - } - - // Preset-only without a session (see patchReasoning). - if (!activeSessionId) { - return - } - void (async () => { - try { - await requestGateway('config.set', { - key: 'fast', - session_id: activeSessionId, - value: enabled ? 'fast' : 'normal' - }) - } catch (err) { - if (touchesPrimary) { - setCurrentFastMode(!enabled) - } else if (activeSessionId) { - sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, fast: !enabled })) - } - - setModelPreset(provider, model, { fast: !enabled }) - notifyError(err, copy.fastFailed) - } - })() + onSetOptions({ fast: enabled }) } } @@ -237,7 +145,7 @@ function ModelEditSubmenuBody({ void patchReasoning(checked ? effortValue || defaultEffort : 'none')} + onCheckedChange={checked => onSetOptions({ effort: checked ? effortValue || defaultEffort : 'none' })} size="xs" /> @@ -252,7 +160,7 @@ function ModelEditSubmenuBody({ <> {copy.effort} - void patchReasoning(value)} value={effortValue}> + onSetOptions({ effort: value })} value={effortValue}> {REASONING_EFFORTS.map(value => ( void>(() => {}) +export { ModelMenuCloseContext } from './model-catalog-menu' export interface ModelSelection { model: string @@ -62,16 +42,15 @@ interface ModelMenuPanelProps { requestGateway: (method: string, params?: Record) => Promise } -interface ProviderGroup { - families: ModelFamily[] - provider: ModelOptionProvider -} - +/** + * The composer's model menu: `ModelCatalogMenu` (the shared renderer) plus the + * controller that gives a selection its meaning HERE — write through to this + * surface's session, remember the pick as a global preset, keep the optimistic + * stores honest, and roll back on a failed gateway write. + */ export function ModelMenuPanel({ gateway, onSelectModel, profile = 'default', requestGateway }: ModelMenuPanelProps) { const { t } = useI18n() const copy = t.shell.modelMenu - const closeMenu = useContext(ModelMenuCloseContext) - const [search, setSearch] = useState('') const [refreshing, setRefreshing] = useState(false) const queryClient = useQueryClient() // Bind to THIS surface's SessionView (primary or tile) so each pane's menu @@ -85,13 +64,15 @@ export function ModelMenuPanel({ gateway, onSelectModel, profile = 'default', re const modelPresets = useStore($modelPresets) const defaultEffort = useStore($defaultReasoningEffort) || DEFAULT_REASONING_EFFORT const visibleModels = useStore($visibleModels) - const collapsedProviders = useStore($collapsedProviders) + const touchesPrimary = view.kind === 'primary' + // Subscribe to the SAME query the menu runs (identical key ⇒ React Query + // dedupes, no second fetch). It must be a live subscription, not a cache + // peek: with no model in the session store yet, currentPickerSelection falls + // back to the catalog's reported current, and a non-reactive read would + // never repaint that fallback once the catalog resolved. const modelOptions = useQuery({ queryKey: modelOptionsQueryKey(profile, activeSessionId), - // Gateway-first even with no session yet: a connected (possibly remote) - // gateway owns the model catalog, including virtual providers like `moa` - // that the local REST fallback can't know about (#53817). queryFn: (): Promise => requestModelOptions({ gateway, sessionId: activeSessionId }) }) @@ -100,42 +81,6 @@ export function ModelMenuPanel({ gateway, onSelectModel, profile = 'default', re modelOptions.data ) - const loading = modelOptions.isPending && !modelOptions.data - - const error = modelOptions.error - ? modelOptions.error instanceof Error - ? modelOptions.error.message - : String(modelOptions.error) - : null - - const providers = modelOptions.data?.providers - - // The catalog carries MoA presets as a virtual `moa` provider row. Render - // them in their dedicated section below and keep the row out of the main - // provider groups so presets don't show up twice. - const moaPresets = useMemo( - () => providers?.find(provider => provider.slug.toLowerCase() === 'moa')?.models ?? [], - [providers] - ) - - const pickerProviders = useMemo( - () => providers?.filter(provider => provider.slug.toLowerCase() !== 'moa') ?? [], - [providers] - ) - - const effectiveVisibleModels = useMemo( - () => effectiveVisibleKeys(visibleModels, pickerProviders), - [visibleModels, pickerProviders] - ) - - // The composer picker never persists the profile default. With a session it - // scopes the switch to that session; with none it's UI state shipped on the - // next session.create (see selectModel). The default lives in Settings → Model. - // Always stamp sessionId from this surface so a tile switch never hits the - // primary (busy) session by accident. - const switchTo = (model: string, provider: string) => - onSelectModel({ model, provider, sessionId: activeSessionId || null }) - // Explicit "Refresh Models": re-fetch the catalog with refresh:true so the // backend busts its 1h provider-model disk cache and re-pulls each provider's // live list. Fixes live-only models (e.g. OpenCode Zen free tier) vanishing @@ -162,450 +107,138 @@ export function ModelMenuPanel({ gateway, onSelectModel, profile = 'default', re } } - // Selecting a model row restores that model's remembered preset onto the - // session (effort/fast), gated by capability. Unset → Hermes defaults. - const selectFamily = async (family: ModelFamily, provider: ModelOptionProvider) => { - const caps = provider.capabilities?.[family.id] - const preset = modelPresets[modelPresetKey(provider.slug, family.id)] ?? {} + // Push a reasoning change onto the session that owns it, with rollback. + const patchReasoning = async (next: string, previous: string, provider: string, model: string) => { + if (touchesPrimary) { + markComposerSelectionManual() + setCurrentReasoningEffort(next) + } else if (activeSessionId) { + sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, reasoningEffort: next })) + } - // Variant-fast models (no speed param) express "fast" as a separate `-fast` - // id, so honor the saved preset by selecting that sibling. Param-fast is - // applied via applyModelPreset below instead. - const variantFast = !(caps?.fast ?? false) && !!family.fastId - const targetId = variantFast && preset.fast === true ? family.fastId! : family.id - - if ((await switchTo(targetId, provider.slug)) === false) { + // Preset-only without a session: the gateway's `config.set` falls back to + // global config when none matches — so don't reach it (preset + optimistic + // store are the whole effect). + if (!activeSessionId) { return } - await applyModelPreset( - { - effort: (caps?.reasoning ?? true) ? (preset.effort ?? defaultEffort) : undefined, - fast: (caps?.fast ?? false) ? (preset.fast ?? false) : undefined - }, - { + try { + await requestGateway('config.set', { key: 'reasoning', session_id: activeSessionId, value: next }) + } catch (err) { + if (touchesPrimary) { + setCurrentReasoningEffort(previous) + } else { + sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, reasoningEffort: previous })) + } + + setModelPreset(provider, model, { effort: previous }) + notifyError(err, t.shell.modelOptions.updateFailed) + } + } + + const patchFast = async (enabled: boolean, provider: string, model: string) => { + if (touchesPrimary) { + markComposerSelectionManual() + setCurrentFastMode(enabled) + } else if (activeSessionId) { + sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, fast: enabled })) + } + + if (!activeSessionId) { + return + } + + try { + await requestGateway('config.set', { + key: 'fast', + session_id: activeSessionId, + value: enabled ? 'fast' : 'normal' + }) + } catch (err) { + if (touchesPrimary) { + setCurrentFastMode(!enabled) + } else { + sessionTileDelegate()?.updateSession(activeSessionId, state => ({ ...state, fast: !enabled })) + } + + setModelPreset(provider, model, { fast: !enabled }) + notifyError(err, t.shell.modelOptions.fastFailed) + } + } + + const controller: ModelMenuController = { + // Selecting a model row restores that model's remembered preset onto the + // session (effort/fast). applyModelPreset owns the batched gateway write. + applyPreset: (preset, row) => { + setModelPreset(row.provider, row.model, preset) + + void applyModelPreset(preset, { failMessage: t.shell.modelOptions.updateFailed, - primary: view.kind === 'primary', + primary: touchesPrimary, request: requestGateway, sessionId: activeSessionId + }) + }, + + current: { + effort: currentReasoningEffort, + fast: currentFastMode, + model: optionsModel, + provider: optionsProvider + }, + + presetFor: (provider, model) => modelPresets[modelPresetKey(provider, model)] ?? {}, + + // The composer picker never persists the profile default. With a session it + // scopes the switch to that session; with none it's UI state shipped on the + // next session.create. Always stamp sessionId from this surface so a tile + // switch never hits the primary (busy) session by accident. + select: (model, provider) => onSelectModel({ model, provider, sessionId: activeSessionId || null }), + + setOptions: (patch, row) => { + // Editing always records the model's global preset (keyed by + // provider::model, not per-surface — a tile edit re-applies to that model + // everywhere); the active model also gets it pushed onto its OWN session. + // Non-active edits stay preset-only — no model switch, no session write. + if (patch.effort !== undefined || patch.fast !== undefined) { + setModelPreset(row.provider, row.model, patch) } - ) - } - // Selecting a MoA preset switches the session to it PERSISTENTLY, using the - // same path real provider selections use (onSelectModel → config.set with - // --session for live sessions → the gateway's persistent switch_model). - // Previously this dispatched the one-shot `/moa` command, which ran a single - // turn through MoA and then silently reverted to the prior model (#54670) — - // the dropdown presented presets like persistent selections but they weren't. - // No session gate: like regular model rows, a pre-session pick is UI state - // shipped on the next session.create. - const selectMoaPreset = async (preset: string) => { - if ((await switchTo(preset, 'moa')) === false) { - return - } + if (!row.isActive) { + return + } - closeMenu() - } + if (patch.effort !== undefined) { + void patchReasoning(patch.effort, currentReasoningEffort, row.provider, row.model) + } - const groups = useMemo( - () => - groupModels(pickerProviders, search, { model: optionsModel, provider: optionsProvider }, effectiveVisibleModels), - [pickerProviders, search, optionsModel, optionsProvider, effectiveVisibleModels] - ) - - const q = normalize(search) - - // Presets are searchable rows like everything else — an unfiltered preset - // sitting under zero model matches would otherwise become the "first match" - // Enter commits. - const shownMoaPresets = useMemo( - () => (q ? moaPresets.filter(preset => `moa ${preset}`.toLowerCase().includes(q)) : moaPresets), - [moaPresets, q] - ) - - // ── Keyboard selection (cmdk semantics on a Radix menu) ─────────────────── - // One flat list mirroring EXACTLY what's rendered (collapse, filter, presets), - // so the selection can never sit on a hidden row. The selected index is - // derived — current model with no query (Enter = close), first match while - // typing — with an arrow-key override that resets on every keystroke. Focus - // stays in the search input throughout: ⌘⇧M → type → ↑/↓ → Enter. - type KbRow = - | { family: ModelFamily; key: string; kind: 'family'; provider: ModelOptionProvider } - | { key: string; kind: 'moa'; preset: string } - - const kbRows = useMemo( - () => [ - ...groups.flatMap(group => - collapsedProviders.includes(group.provider.slug) && !search - ? [] - : group.families.map( - (family): KbRow => ({ - family, - key: `${group.provider.slug}:${family.id}`, - kind: 'family', - provider: group.provider - }) - ) - ), - ...shownMoaPresets.map((preset): KbRow => ({ key: `moa:${preset}`, kind: 'moa', preset })) - ], - [groups, collapsedProviders, search, shownMoaPresets] - ) - - const [kbOverride, setKbOverride] = useState(null) - // A parked cursor is not a cursor in use: until the mouse actually moves, - // hover can't take rows out from under the keyboard (rows re-flow beneath it - // as the filter narrows). One real movement hands hover back. - const pointerQuiet = usePointerQuiet() - - const currentKey = optionsProvider === 'moa' ? `moa:${optionsModel}` : `${optionsProvider}:${optionsModel}` - - const autoIndex = q - ? kbRows.length > 0 - ? 0 - : -1 - : kbRows.findIndex(row => row.key === currentKey || (row.kind === 'family' && row.family.fastId === optionsModel)) - - const kbIndex = kbOverride !== null && kbOverride < kbRows.length ? kbOverride : autoIndex - const kbActiveKey = kbIndex >= 0 ? kbRows[kbIndex].key : null - - const stepKb = (delta: -1 | 1) => { - if (kbRows.length === 0) { - return - } - - const from = kbIndex >= 0 ? kbIndex : delta === 1 ? -1 : 0 - - setKbOverride((from + delta + kbRows.length) % kbRows.length) - } - - const commitKbRow = () => { - const row = kbIndex >= 0 ? kbRows[kbIndex] : undefined - - if (!row) { - return - } - - if (row.kind === 'moa') { - void selectMoaPreset(row.preset) - - return - } - - if (row.key !== currentKey && row.family.fastId !== optionsModel) { - void selectFamily(row.family, row.provider) - } - - closeMenu() - } - - // Keep the selected row in view while arrowing through the scrollable list. - const listRef = useRef(null) - - useEffect(() => { - listRef.current?.querySelector('[data-kb-active]')?.scrollIntoView({ block: 'nearest' }) - }, [kbActiveKey]) - - // The keyboard-selected row, styled + tagged for scrollIntoView. Pointer - // suppression is NOT here — it belongs on the containers (below), so one - // class covers every row inside them. - const kbRowProps = (key: string) => { - const active = kbActiveKey === key - - return { - className: cn(dropdownMenuRow, active && 'bg-(--ui-control-active-background) text-foreground'), - ...(active ? { 'data-kb-active': '' } : {}) + if (patch.fast !== undefined) { + void patchFast(patch.fast, row.provider, row.model) + } } } - // Rows are hover-selectable, so they go inert with the pointer (usePointerQuiet). - const quietRows = pointerQuiet && 'pointer-events-none' - return ( - <> - { - // Claim arrows and Enter from Radix so DOM focus stays in the input - // and Enter commits the highlighted row without a DownArrow first - // (VS Code's checked-or-first pattern). - if (event.key === 'ArrowDown' || event.key === 'ArrowUp') { + { event.preventDefault() - event.stopPropagation() - stepKb(event.key === 'ArrowDown' ? 1 : -1) - } else if (event.key === 'Enter') { - event.preventDefault() - event.stopPropagation() - commitKbRow() - } - }} - onValueChange={value => { - setSearch(value) - setKbOverride(null) - }} - placeholder={copy.search} - value={search} - /> - - - - {loading ? ( - - {Array.from({ length: 4 }, (_, index) => ( - event.preventDefault()} - > - - - ))} - - ) : error ? ( - - {error} + void refreshModels() + }} + > + + {copy.refreshModels} - ) : groups.length === 0 && moaPresets.length === 0 ? ( - - {copy.noModels} - - ) : ( -
- {groups.map(group => { - const slug = group.provider.slug - - // Collapsed when the user stored it (and not while searching, which - // spans every model regardless of collapse state). - const collapsed = collapsedProviders.includes(slug) && !search - - return ( - - { - event.preventDefault() - toggleCollapsedProvider(slug) - }} - textValue="" - > - - - - - - {!collapsed && - group.families.map(family => { - // The active id may be the base or its -fast sibling; either - // way this one family row represents both. - const activeId = - group.provider.slug === optionsProvider && - (optionsModel === family.id || optionsModel === family.fastId) - ? optionsModel - : null - - const isCurrent = activeId !== null - const name = modelDisplayParts(family.id).name - // Capabilities are looked up against the active/base id; the - // -fast variant carries the same param support as its base. - const caps = group.provider.capabilities?.[family.id] - - // Effective settings for this row: live session state when it's - // the active model, otherwise its remembered preset (Hermes - // defaults when unset). Row label AND submenu read from these so - // they never disagree. - const preset = modelPresets[modelPresetKey(group.provider.slug, family.id)] ?? {} - const effEffort = isCurrent ? currentReasoningEffort : (preset.effort ?? '') - const effFast = isCurrent ? currentFastMode : (preset.fast ?? false) - - const fastControl = resolveFastControl( - activeId ?? family.id, - group.provider.models ?? [], - caps?.fast ?? false, - effFast - ) - - const meta = [ - fastControl.kind !== 'none' && fastControl.on ? copy.fast : null, - (caps?.reasoning ?? true) ? reasoningEffortLabel(effEffort || defaultEffort) : null - ] - .filter(Boolean) - .join(' ') - - // Every row is a hover-Edit submenu trigger. Activating it - // (pointer or keyboard) switches to the family's base model and - // restores its preset; the Fast toggle inside swaps to the -fast - // sibling (or flips the speed param). The sub-trigger has no - // `onSelect`, so wire both click and Enter/Space for keyboard parity. - // Clicking the row commits the model and closes the picker; the - // edit submenu (reasoning/fast) is reached by HOVER, so you can - // still tweak those without the click dismissing everything. - const activate = () => { - if (!isCurrent) { - void selectFamily(family, group.provider) - } - - closeMenu() - } - - return ( - - { - if (event.key === 'Enter' || event.key === ' ') { - activate() - } - }} - {...kbRowProps(`${group.provider.slug}:${family.id}`)} - > - - - {meta ? {meta} : null} - - {isCurrent ? ( - - ) : null} - - switchTo(nextModel, group.provider.slug)} - provider={group.provider.slug} - reasoning={caps?.reasoning ?? true} - requestGateway={requestGateway} - /> - - ) - })} - - ) - })} -
- )} - - - - {shownMoaPresets.length > 0 ? ( -
- MoA presets - {shownMoaPresets.map(preset => { - const isCurrentMoa = optionsProvider === 'moa' && optionsModel === preset - - return ( - { - event.preventDefault() - void selectMoaPreset(preset) - }} - {...kbRowProps(`moa:${preset}`)} - > - - MoA: - - {isCurrentMoa ? : null} - - ) - })} - -
- ) : null} - - { - event.preventDefault() - void refreshModels() - }} - > - - {copy.refreshModels} - - - setModelVisibilityOpen(true)} - > - - {copy.editModels} - - + } + gateway={gateway} + includeMoa + profile={profile} + sessionId={activeSessionId} + /> ) } - -// Collapsed we show the user's chosen models (or the curated default); typing -// spans every available model so anything is reachable past the cut. A search -// is itself a narrowing action, so we do NOT cap per-provider matches — a -// provider serving 19 models (e.g. opencode-go) must show all 19 when the user -// searches for it, not a truncated subset. (#47077 follow-up) - -function groupModels( - providers: ModelOptionProvider[], - search: string, - current: { model: string; provider: string }, - visible: Set | null -): ProviderGroup[] { - const q = normalize(search) - const groups: ProviderGroup[] = [] - - for (const provider of providers) { - const allFamilies = collapseModelFamilies(provider.models ?? []) - - if (allFamilies.length === 0) { - continue - } - - const matches = (family: ModelFamily) => - `${family.id} ${family.fastId ?? ''} ${provider.name} ${provider.slug} ${displayModelName(family.id)}` - .toLowerCase() - .includes(q) - - // Which model ids to show (the active one is always added on top of this). - let shown: Set - - if (q) { - // Search spans every family, regardless of visibility. - shown = new Set(allFamilies.filter(matches).map(family => family.id)) - } else if (visible) { - // User has customized which models show — honor their selection exactly. - shown = new Set( - allFamilies.filter(family => visible.has(modelVisibilityKey(provider.slug, family.id))).map(family => family.id) - ) - } else { - // Default: curated top-N families per provider. - shown = new Set(allFamilies.slice(0, DEFAULT_VISIBLE_PER_PROVIDER).map(family => family.id)) - } - - // Always include the active model — but keep every row in the provider's - // stable curated order (filter `allFamilies`, never reorder), so selecting - // a model can't shuffle the list. While SEARCHING, the pin is skipped: a - // query means "show me matches", and a pinned non-match sitting above them - // reads like the top result (type "grok", see the current Fable first). - const activeId = - !q && provider.slug === current.provider && current.model - ? allFamilies.find(family => family.id === current.model || family.fastId === current.model)?.id - : undefined - - const families = allFamilies.filter(family => shown.has(family.id) || family.id === activeId) - - if (families.length > 0) { - groups.push({ families, provider }) - } - } - - // Stable, logical group order: alphabetical by provider name. (The backend - // floats the current provider first, which would reshuffle on every switch.) - groups.sort((a, b) => a.provider.name.localeCompare(b.provider.name)) - - return groups -} diff --git a/apps/desktop/src/app/shell/statusbar-controls.tsx b/apps/desktop/src/app/shell/statusbar-controls.tsx index 7dbc9335f7..855d016bff 100644 --- a/apps/desktop/src/app/shell/statusbar-controls.tsx +++ b/apps/desktop/src/app/shell/statusbar-controls.tsx @@ -1,6 +1,6 @@ import { useStore } from '@nanostores/react' import { type ComponentProps, memo, type ReactNode, useMemo, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { ContextMenu, diff --git a/apps/desktop/src/app/shell/statusbar-visibility.test.tsx b/apps/desktop/src/app/shell/statusbar-visibility.test.tsx index 9d334a0330..b4bce684bb 100644 --- a/apps/desktop/src/app/shell/statusbar-visibility.test.tsx +++ b/apps/desktop/src/app/shell/statusbar-visibility.test.tsx @@ -1,5 +1,5 @@ import { cleanup, fireEvent, render, screen, within } from '@testing-library/react' -import { MemoryRouter } from 'react-router-dom' +import { MemoryRouter } from 'react-router' import { afterEach, beforeAll, describe, expect, it, vi } from 'vitest' import { StatusbarControls, type StatusbarItem } from '@/app/shell/statusbar-controls' diff --git a/apps/desktop/src/app/shell/titlebar-controls.tsx b/apps/desktop/src/app/shell/titlebar-controls.tsx index debbaf3248..6c3c3e5fc5 100644 --- a/apps/desktop/src/app/shell/titlebar-controls.tsx +++ b/apps/desktop/src/app/shell/titlebar-controls.tsx @@ -1,6 +1,6 @@ import { useStore } from '@nanostores/react' import { type ComponentProps, type MouseEvent, type ReactNode, useEffect, useState } from 'react' -import { useLocation, useNavigate } from 'react-router-dom' +import { useLocation, useNavigate } from 'react-router' import { toggleLayoutEditMode } from '@/components/pane-shell/edit-mode' import { resetLayoutTree } from '@/components/pane-shell/tree/store' diff --git a/apps/desktop/src/app/skills/index.test.tsx b/apps/desktop/src/app/skills/index.test.tsx index 62f1f606c7..a22d9aa521 100644 --- a/apps/desktop/src/app/skills/index.test.tsx +++ b/apps/desktop/src/app/skills/index.test.tsx @@ -1,8 +1,8 @@ // @vitest-environment jsdom import { QueryClientProvider } from '@tanstack/react-query' import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' -import { MemoryRouter } from 'react-router-dom' -import type * as ReactRouterDom from 'react-router-dom' +import { MemoryRouter } from 'react-router' +import type * as ReactRouterDom from 'react-router' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as HermesApi from '@/hermes' @@ -40,7 +40,7 @@ vi.mock('@/store/notifications', () => ({ // so the deep-link target is assertable. const navigateSpy = vi.fn() -vi.mock('react-router-dom', async importOriginal => ({ +vi.mock('react-router', async importOriginal => ({ ...(await importOriginal()), useNavigate: () => navigateSpy })) diff --git a/apps/desktop/src/app/skills/index.tsx b/apps/desktop/src/app/skills/index.tsx index c66cd1cc4b..9335d293dc 100644 --- a/apps/desktop/src/app/skills/index.tsx +++ b/apps/desktop/src/app/skills/index.tsx @@ -2,7 +2,7 @@ import { useStore } from '@nanostores/react' import { useQuery } from '@tanstack/react-query' import type * as React from 'react' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import { useNavigate } from 'react-router-dom' +import { useNavigate } from 'react-router' import { ArchiveSkillConfirmDialog } from '@/app/learning/archive-skill-confirm-dialog' import { CodeEditor } from '@/components/chat/code-editor' diff --git a/apps/desktop/src/app/wake-indicator/wake-indicator-app.tsx b/apps/desktop/src/app/wake-indicator/wake-indicator-app.tsx new file mode 100644 index 0000000000..d90b9a8958 --- /dev/null +++ b/apps/desktop/src/app/wake-indicator/wake-indicator-app.tsx @@ -0,0 +1,42 @@ +import './wake-indicator.css' + +import { useEffect, useState } from 'react' + +import type { WakeIndicatorState } from '@/lib/wake-indicator' + +export function WakeIndicatorApp() { + const [state, setState] = useState('hidden') + + useEffect(() => { + const api = window.hermesDesktop?.wakeIndicator + let mounted = true + let receivedLiveState = false + + const unsubscribe = api?.onState(next => { + if (mounted) { + receivedLiveState = true + setState(next) + } + }) + + void api + ?.getState() + .then(next => { + if (mounted && !receivedLiveState) { + setState(next) + } + }) + .catch(() => undefined) + + return () => { + mounted = false + unsubscribe?.() + } + }, []) + + return ( +
+
+
+ ) +} diff --git a/apps/desktop/src/app/wake-indicator/wake-indicator-root.tsx b/apps/desktop/src/app/wake-indicator/wake-indicator-root.tsx new file mode 100644 index 0000000000..fc4a99288c --- /dev/null +++ b/apps/desktop/src/app/wake-indicator/wake-indicator-root.tsx @@ -0,0 +1,26 @@ +import { StrictMode } from 'react' +import { createRoot } from 'react-dom/client' + +import { ErrorBoundary } from '@/components/error-boundary' + +import { WakeIndicatorApp } from './wake-indicator-app' + +export function mountWakeIndicator(): void { + const style = document.createElement('style') + style.textContent = 'html,body,#root{background:transparent !important;overflow:hidden;}' + document.head.appendChild(style) + + const root = document.getElementById('root') + + if (!root) { + return + } + + createRoot(root).render( + + + + + + ) +} diff --git a/apps/desktop/src/app/wake-indicator/wake-indicator.css b/apps/desktop/src/app/wake-indicator/wake-indicator.css new file mode 100644 index 0000000000..c9d92d509d --- /dev/null +++ b/apps/desktop/src/app/wake-indicator/wake-indicator.css @@ -0,0 +1,61 @@ +.wake-indicator-surface { + align-items: flex-start; + background: transparent; + display: flex; + height: 100vh; + justify-content: center; + padding-top: 5px; + pointer-events: none; + width: 100vw; +} + +.wake-indicator-light { + backdrop-filter: blur(20px) saturate(1.2); + background: rgba(124, 58, 237, 0.62); + border: 1px solid rgba(216, 180, 254, 0.82); + border-radius: 999px; + box-shadow: + 0 0 10px rgba(124, 58, 237, 0.7), + 0 0 24px rgba(168, 85, 247, 0.48); + height: 28px; + opacity: 0; + transform: scale(0.96); + transition: + opacity 500ms ease-out, + transform 500ms ease-out; + width: 120px; +} + +.wake-indicator-surface[data-state='detected'] .wake-indicator-light { + animation: wake-indicator-breathe 2.5s ease-in-out infinite; +} + +.wake-indicator-surface[data-state='capturing'] .wake-indicator-light { + opacity: 1; + transform: scale(1); +} + +@keyframes wake-indicator-breathe { + 0%, + 100% { + opacity: 0.34; + transform: scale(0.96); + } + + 50% { + opacity: 0.94; + transform: scale(1); + } +} + +@media (prefers-reduced-motion: reduce) { + .wake-indicator-light { + transition: opacity 200ms ease-out; + } + + .wake-indicator-surface[data-state='detected'] .wake-indicator-light { + animation: none; + opacity: 0.72; + transform: scale(1); + } +} diff --git a/apps/desktop/src/components/assistant-ui/directive-text.tsx b/apps/desktop/src/components/assistant-ui/directive-text.tsx index 6d149e2dd2..538f7770d7 100644 --- a/apps/desktop/src/components/assistant-ui/directive-text.tsx +++ b/apps/desktop/src/components/assistant-ui/directive-text.tsx @@ -6,7 +6,9 @@ import type { FC } from 'react' import { Fragment, useEffect, useMemo, useState } from 'react' import { ZoomableImage } from '@/components/chat/zoomable-image' +import type { I18nContextValue } from '@/i18n' import { extractEmbeddedImages } from '@/lib/embedded-images' +import { openExternalLink } from '@/lib/external-link' import { triggerHaptic } from '@/lib/haptics' import { gatewayMediaDataUrl, isRemoteGateway } from '@/lib/media' import { useSessionLinkTitle } from '@/lib/session-link-title' @@ -442,7 +444,7 @@ const DirectiveImage: FC<{ id: string; label: string }> = ({ id, label }) => { * it's already a tile/main, otherwise open a stacked tab (never steals main * from under the chat you're reading). Lazy-imports so the composer's rich * editor can pull this module in without booting the profile/REST stack. */ -function openSessionRef(value: string) { +export function openSessionRef(value: string) { const { sessionId } = parseSessionRefValue(value) if (!sessionId) { @@ -454,6 +456,33 @@ function openSessionRef(value: string) { void import('@/app/open-session').then(({ openSession }) => openSession(sessionId, () => undefined, 'tab')) } +/** What activating a directive of a given kind does. The single source of truth + * for "you can act on this reference," shared by every surface that renders a + * chip: the composer's hover pill (`ComposerDirectiveActions`) and the sent + * message's clickable chip below. A kind with no entry is inert everywhere. + * + * Add a kind here and both surfaces light up — that's the whole point of one + * table. `icon`/`label` are for the pill; the transcript chip carries its own + * glyph and only reads `run`. */ +export interface DirectiveAction { + icon: string + label: (t: I18nContextValue['t']) => string + run: (value: string) => void +} + +export const DIRECTIVE_ACTIONS: Record = { + session: { + icon: 'link-external', + label: t => t.composer.openDirective, + run: openSessionRef + }, + url: { + icon: 'link-external', + label: t => t.composer.openDirective, + run: openExternalLink + } +} + /** A `@session:/` reference in the user transcript (directive * segments), rendered as a chip like the other composer refs. Clicking it * opens the session as a tab. */ @@ -501,14 +530,18 @@ const SlashChip: FC<{ kind: SlashChipKind; label: string; value: string }> = ({
) -/** Inert by default; `onClick` promotes the chip to a real button (session - * refs, which open the session they name). */ +/** A directive reference in a sent message. A kind with a `DIRECTIVE_ACTIONS` + * entry (a url, …) renders as a real button that runs it on click; everything + * else is inert text. `onClick` overrides for chips that resolve their target + * themselves (session, which needs the async navigator). */ const DirectiveChip: FC<{ type: string label: string id: string onClick?: () => void }> = ({ type, label, id, onClick }) => { + const activate = onClick ?? (DIRECTIVE_ACTIONS[type] ? () => DIRECTIVE_ACTIONS[type]!.run(id) : undefined) + const body = ( <> @@ -517,14 +550,14 @@ const DirectiveChip: FC<{ ) const props = { - ...refAttrs(type, cn('wrap-anywhere', onClick && 'cursor-pointer')), + ...refAttrs(type, cn('wrap-anywhere', activate && 'cursor-pointer')), 'data-directive-id': id, 'data-slot': 'aui_directive-chip', title: id } - return onClick ? ( - ) : ( diff --git a/apps/desktop/src/components/assistant-ui/embeds/providers/types.ts b/apps/desktop/src/components/assistant-ui/embeds/providers/types.ts index e9f4ce381d..b2735027a6 100644 --- a/apps/desktop/src/components/assistant-ui/embeds/providers/types.ts +++ b/apps/desktop/src/components/assistant-ui/embeds/providers/types.ts @@ -3,15 +3,7 @@ // the lazy renderers (see ../registry.tsx) keyed off `renderer`. export type EmbedProvider = - | 'googlemaps' - | 'instagram' - | 'openstreetmap' - | 'pinterest' - | 'spotify' - | 'tiktok' - | 'twitter' - | 'vimeo' - | 'youtube' + 'googlemaps' | 'instagram' | 'openstreetmap' | 'pinterest' | 'spotify' | 'tiktok' | 'twitter' | 'vimeo' | 'youtube' /** Which lazy renderer materialises the descriptor. */ export type EmbedRenderer = 'frame' | 'tweet' diff --git a/apps/desktop/src/components/assistant-ui/session-ref-open.test.tsx b/apps/desktop/src/components/assistant-ui/session-ref-open.test.tsx index d8fe7136aa..f10d729d88 100644 --- a/apps/desktop/src/components/assistant-ui/session-ref-open.test.tsx +++ b/apps/desktop/src/components/assistant-ui/session-ref-open.test.tsx @@ -12,9 +12,12 @@ vi.mock('@/app/open-session', () => ({ openSession: (...args: unknown[]) => openSession(...args) })) +const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] } + afterEach(() => { cleanup() openSession.mockClear() + delete desktopWindow.hermesDesktop __resetSessionLinkTitleCache() }) @@ -41,3 +44,23 @@ describe('session refs open the session', () => { await vi.waitFor(() => expect(openSession).toHaveBeenCalledWith('20260101_abc123', expect.any(Function), 'tab')) }) }) + +// A url the user sent renders as a chip too, and it opens in the browser — the +// same door the composer's hover pill uses, so a link behaves the same before +// and after send. +describe('url refs open externally', () => { + it('opens a url chip in the user transcript', () => { + const openExternal = vi.fn().mockResolvedValue(undefined) + + desktopWindow.hermesDesktop = { openExternal } as unknown as Window['hermesDesktop'] + + render() + + const chip = screen.getByTitle('https://example.com/docs') + + expect(chip.tagName).toBe('BUTTON') + fireEvent.click(chip) + + expect(openExternal).toHaveBeenCalledWith('https://example.com/docs') + }) +}) diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index 3af28d39b5..043a60401c 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -71,6 +71,13 @@ export const AssistantMessage: FC<{ // ChatMessage.interim). const isInterim = useAuiState(s => s.message.metadata?.custom?.interim === true) + // The thinking/stall indicator belongs to the TAIL of the thread, period. A + // stale pending bubble mid-transcript (a turn that ended without its settle + // event, a steer race) must never show one — a spinner above a later user + // message reads as the agent answering out of order. Booleans are stable + // across token flushes, so this selector adds no streaming re-renders. + const isLastMessage = useAuiState(s => s.thread.messages[s.thread.messages.length - 1]?.id === s.message.id) + // Preview targets only materialize once the turn completes — while running // the selector returns '' (stable), so per-token flushes skip the regex // scan and the re-render it would cause. @@ -124,7 +131,7 @@ export const AssistantMessage: FC<{ > {/* Todos render in the composer status stack now, not inline. */} - {isPlaceholder ? : isRunning && } + {isLastMessage && (isPlaceholder ? : isRunning && )} {previewTargets.length > 0 && (
{previewTargets.map(target => ( diff --git a/apps/desktop/src/components/assistant-ui/thread/changed-files-card.tsx b/apps/desktop/src/components/assistant-ui/thread/changed-files-card.tsx index 8a29c55de9..021036440d 100644 --- a/apps/desktop/src/components/assistant-ui/thread/changed-files-card.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/changed-files-card.tsx @@ -5,11 +5,17 @@ import { useSessionView } from '@/app/chat/session-view' import { deriveChangedFiles } from '@/components/assistant-ui/thread/changed-files' import { WIDGET_SHELL_CLASS } from '@/components/chat/widget-shell' import { DiffCount } from '@/components/ui/diff-count' +import { FadeScroll } from '@/components/ui/fade-scroll' import { FileTypeIcon } from '@/components/ui/file-type-icon' import { useI18n } from '@/i18n' +import { displayPath } from '@/lib/display-path' import { cn } from '@/lib/utils' import { openReviewForPath, revealReview } from '@/store/review' +// ~5 rows. A turn that rewrites twenty files should still read as one card in +// the transcript, not a wall the user has to scroll past to reach the composer. +const MAX_ROWS_HEIGHT = '9.375rem' + /** * Cursor-style "N files changed" summary closing out the newest assistant turn: * one row per file it edited with that file's +/-, and a Review action opening @@ -47,13 +53,13 @@ export const ChangedFilesCard: FC<{ parts: readonly unknown[] }> = ({ parts }) = {copy.reviewChanges}
-
+ {files.map(file => ( ))} -
+ ) } diff --git a/apps/desktop/src/components/assistant-ui/thread/list.test.ts b/apps/desktop/src/components/assistant-ui/thread/list.test.ts index a09d57deba..413aab8865 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.test.ts +++ b/apps/desktop/src/components/assistant-ui/thread/list.test.ts @@ -6,7 +6,9 @@ import { LIVE_TAIL_MIN_GROUPS, LIVE_TAIL_PARTS, liveTailStart, - type MessageGroup + type MessageGroup, + messageRenderWeight, + RENDER_WEIGHT_CHARS } from './list' // Signature rows are `${index}:${id}:${role}:${weight}` (see the useAuiState @@ -88,6 +90,50 @@ describe('firstVisibleGroupIndex', () => { }) }) +describe('messageRenderWeight', () => { + it('charges large text and tool results by character cost, not only part count', () => { + const text = [{ type: 'text', text: 'x'.repeat(RENDER_WEIGHT_CHARS * 3) }] + + const tool = [ + { + type: 'tool-call', + toolName: 'skill_view', + args: { name: 'hermes-agent' }, + result: { content: 'x'.repeat(RENDER_WEIGHT_CHARS * 100) } + } + ] + + expect(messageRenderWeight(text)).toBe(4) + expect(messageRenderWeight(tool)).toBeGreaterThanOrEqual(101) + }) + + it('makes repeated 51KB tool outputs exceed the normal transcript page', () => { + const toolOutput = () => [ + { + type: 'tool-call', + toolName: 'skill_view', + result: { content: 'x'.repeat(51_236) } + } + ] + + const groups = Array.from({ length: 5 }, (_, index) => ({ + id: `tool-${index}`, + index, + kind: 'standalone' as const, + weight: messageRenderWeight(toolOutput()) + })) + + expect(firstVisibleGroupIndex(groups, 300)).toBeGreaterThan(0) + }) + + it('handles circular tool payloads without recursing forever', () => { + const result: { content: string; self?: unknown } = { content: 'ok' } + result.self = result + + expect(messageRenderWeight([{ type: 'tool-call', result }])).toBe(2) + }) +}) + describe('liveTailStart', () => { const group = (id: string, weight: number): MessageGroup => ({ id, index: 0, kind: 'standalone', weight }) diff --git a/apps/desktop/src/components/assistant-ui/thread/list.tsx b/apps/desktop/src/components/assistant-ui/thread/list.tsx index a9aff632e8..faa0724883 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/list.tsx @@ -31,31 +31,101 @@ import { MessageRenderBoundary } from '../message-render-boundary' type ThreadMessageComponents = ComponentProps['components'] export type MessageGroup = { id: string; weight: number } & ( - | { index: number; kind: 'standalone' } - | { indices: number[]; kind: 'turn' } + { index: number; kind: 'standalone' } | { indices: number[]; kind: 'turn' } ) -// DOM is bounded by a rendered-PART budget, not a message/turn count: a single -// assistant message folds every tool call into a part, so heavy sessions are -// ~40 turns / ~100 messages but ~1000 parts — and parts are what drive node -// count. "Show earlier" prepends another page; whole turns stay intact so the -// sticky human bubble never loses its turn. This is the long-session perf lever -// WITHOUT a virtualizer — pure rendering, never touches scrollTop, so it can't -// fight use-stick-to-bottom (the single scroll owner). +// DOM is bounded by a render-cost budget, not a message/turn count. Every part +// costs one unit, and large strings add another unit per 512 characters. Parts +// approximate component/node count; characters approximate markdown parsing, +// text-node allocation, and tool-result formatting. Counting only parts badly +// underpriced a 51KB tool result as "1", so a handful of huge results let a +// 600KB transcript through the old 300-part cap and could drive Chromium's renderer +// into a GC crash. +// +// "Show earlier" prepends another page; whole turns stay intact so the sticky +// human bubble never loses its turn. This is the long-session perf lever WITHOUT +// a virtualizer — pure rendering, never touches scrollTop, so it can't fight +// use-stick-to-bottom (the single scroll owner). const RENDER_BUDGET = 300 +export const RENDER_WEIGHT_CHARS = 512 +const MAX_MEASURED_MESSAGE_CHARS = RENDER_BUDGET * RENDER_WEIGHT_CHARS // On session switch, paint a small budget first (enough for the bottom turn(s) // the user actually sees after scroll-to-bottom), then bump to the full budget // in a requestAnimationFrame — defers the heavy markdown+syntax-highlight render // past the initial commit, so the switch feels instant. // // 20, down from 60: the first-paint commit is synchronous and uninterruptible, -// and at 60 parts it measured 627ms on a real session (LoAF: block=575ms, no +// and at 60 cost units it measured 627ms on a real session (LoAF: block=575ms, no // attributed script — pure commit). A viewport after scroll-to-bottom shows -// 1-2 turns ≈ 10-20 parts; the transition backfill below fills the rest +// 1-2 normal turns ≈ 10-20 units; the transition backfill below fills the rest // interruptibly, so the only thing a smaller budget changes is how much work // blocks the click-to-paint path. const FIRST_PAINT_BUDGET = 20 +const contentWeightCache = new WeakMap() +const NON_RENDERED_CONTENT_FIELDS = new Set(['id', 'role', 'toolCallId', 'toolName', 'type']) + +/** + * Estimate the synchronous renderer cost of one assistant-ui message. + * + * The traversal is capped once a single message has enough text to consume a + * complete render page. Going further cannot affect which whole turn crosses + * the budget, and avoiding an unbounded walk matters for deeply nested tool + * payloads. A WeakMap keeps settled history O(message count) on later store + * updates; assistant-ui publishes a new content array when a streaming message + * changes, so the live tail still receives a fresh weight. + */ +export function messageRenderWeight(content: unknown): number { + if (!Array.isArray(content)) { + return 1 + } + + const cached = contentWeightCache.get(content) + + if (cached !== undefined) { + return cached + } + + const seen = new WeakSet() + const pending: unknown[] = [...content] + let characters = 0 + + while (pending.length > 0 && characters < MAX_MEASURED_MESSAGE_CHARS) { + const value = pending.pop() + + if (typeof value === 'string') { + characters += Math.min(value.length, MAX_MEASURED_MESSAGE_CHARS - characters) + + continue + } + + if (!value || typeof value !== 'object' || seen.has(value)) { + continue + } + + seen.add(value) + + if (Array.isArray(value)) { + for (const nested of value) { + pending.push(nested) + } + + continue + } + + for (const [key, nested] of Object.entries(value)) { + if (!NON_RENDERED_CONTENT_FIELDS.has(key)) { + pending.push(nested) + } + } + } + + const weight = Math.max(1, content.length) + Math.ceil(characters / RENDER_WEIGHT_CHARS) + contentWeightCache.set(content, weight) + + return weight +} + interface ThreadMessageListProps { clampToComposer: boolean components: ThreadMessageComponents @@ -103,7 +173,7 @@ export function buildGroups(signature: string): MessageGroup[] { return groups } -// Walk turns newest-first, summing their part weights until the budget is met; +// Walk turns newest-first, summing their render weights until the budget is met; // everything before the first kept turn is hidden. Returns the index of that // first visible group. export function firstVisibleGroupIndex(groups: readonly MessageGroup[], budget: number): number { @@ -135,15 +205,16 @@ export function firstVisibleGroupIndex(groups: readonly MessageGroup[], budget: // it changes no height). Off-screen OLDER turns still skip, so the dialog/popover // recalc win on long transcripts is preserved. // -// The tail is budgeted in PARTS, not turns, because that is what the cost -// actually scales with — the same currency as RENDER_BUDGET / FIRST_PAINT_BUDGET. +// The tail is budgeted in render-cost units, not turns, because that is what the +// cost actually scales with — the same currency as RENDER_BUDGET / +// FIRST_PAINT_BUDGET. // A turn-count tail silently defeats itself on agent transcripts: one tool-heavy -// turn is 50-200 parts, so a 6-TURN tail exempted the entire visible transcript +// turn is 50-200 units, so a 6-TURN tail exempted the entire visible transcript // and nothing virtualized at all. Measured on a 5-tile window (7/3/5/3/2 groups // per tile): zero content-visibility containers were active, and every Radix // overlay open paid the full ~610ms whole-document recalc that #66470 fixed. // -// 40 parts ≈ the 1-2 turns a viewport shows after scroll-to-bottom (the same +// 40 units ≈ the 1-2 turns a viewport shows after scroll-to-bottom (the same // reasoning as FIRST_PAINT_BUDGET=20, doubled so a turn that grows mid-stream // doesn't fall out of the tail as it settles). export const LIVE_TAIL_PARTS = 40 @@ -151,31 +222,31 @@ export const LIVE_TAIL_PARTS = 40 // transcript of very heavy turns still keeps the streaming one unvirtualized. export const LIVE_TAIL_MIN_GROUPS = 2 // Ceiling: never exempt more than this many turns, however light they are. On a -// long transcript of tiny turns a parts-only budget would walk back further +// long transcript of tiny turns a weight-only budget would walk back further // than the old turn-count tail did and virtualize LESS — this keeps the new // policy a strict improvement on every shape. export const LIVE_TAIL_MAX_GROUPS = 6 /** * Index of the newest group that still virtualizes — everything at or after it - * is the live tail and stays rendered. Walks newest-first accumulating parts, + * is the live tail and stays rendered. Walks newest-first accumulating weight, * so the tail covers a viewport's worth of content rather than a fixed number * of turns, clamped to [MIN, MAX] turns. Computed once per render, not per row. */ export function liveTailStart( groups: readonly MessageGroup[], - tailParts = LIVE_TAIL_PARTS, + tailWeight = LIVE_TAIL_PARTS, minGroups = LIVE_TAIL_MIN_GROUPS, maxGroups = LIVE_TAIL_MAX_GROUPS ): number { - let parts = 0 + let weight = 0 let start = groups.length for (let i = groups.length - 1; i >= 0; i--) { - parts += groups[i]?.weight ?? 1 + weight += groups[i]?.weight ?? 1 start = i - if (parts > tailParts) { + if (weight > tailWeight) { break } } @@ -198,8 +269,8 @@ const ThreadMessageListInner: FC = ({ }) => { // TWO signatures, deliberately split. The STRUCTURAL one (ids/roles/count) // changes only when messages are added/removed/swapped — it keys the error - // boundaries and the row identity. The WEIGHT one (per-message part counts) - // ticks while a streaming turn appends parts — it feeds only the render + // boundaries and the row identity. The WEIGHT one (parts + character cost) + // ticks while a streaming turn appends content — it feeds only the render // budget. Folding weights into the structural key handed every boundary a // new resetKey per appended part, which reconciled every turn's subtree on // every tick (measured: 540 wasted Block renders per explain() sample with @@ -208,7 +279,9 @@ const ThreadMessageListInner: FC = ({ s.thread.messages.map((message, index) => `${index}:${message.id}:${message.role}`).join('\n') ) - const weightSignature = useAuiState(s => s.thread.messages.map(message => message.content?.length ?? 1).join(',')) + const weightSignature = useAuiState(s => + s.thread.messages.map(message => messageRenderWeight(message.content)).join(',') + ) const { t } = useI18n() // Row structure is memoized on the STRUCTURAL signature only, so streaming @@ -231,7 +304,7 @@ const ThreadMessageListInner: FC = ({ // Cut the budget during RENDER, not in the post-commit layout effect. An // effect-time cut is too late: React would first build the whole tree with - // the full budget (up to 300 parts of markdown + syntax highlighting), + // the full budget (up to 300 cost units of markdown + syntax highlighting), // commit it, and only then re-render at the small budget. The render-phase // state adjustment restarts this component immediately — before any child // renders — so the heavy commit never happens. @@ -308,9 +381,9 @@ const ThreadMessageListInner: FC = ({ return () => cancelAnimationFrame(rafId) }, [anchorBeforePrepend, renderBudget]) - // Weights (per-message part counts) fold into the BUDGET only. Group - // identity stays structural, so a streaming append re-runs this cheap sum — - // not the row JSX. Weighted the same way the old combined signature was. + // Weights (part count + visible character cost) fold into the BUDGET only. + // Group identity stays structural, so a streaming append re-runs this cheap + // sum — not the row JSX. Settled content hits messageRenderWeight's WeakMap. const weightedGroups = useMemo(() => { const weights = weightSignature.split(',').map(w => Number(w) || 1) @@ -327,7 +400,7 @@ const ThreadMessageListInner: FC = ({ const visibleGroups = hiddenCount > 0 ? groups.slice(hiddenCount) : groups // Where the always-rendered live tail begins. Derived from the WEIGHTED - // groups (parts, not turns) so the tail is a viewport's worth of content — + // groups (render cost, not turns) so the tail is a viewport's worth of content — // see liveTailStart. Computed once here rather than per row. const tailStart = useMemo( () => liveTailStart(hiddenCount > 0 ? weightedGroups.slice(hiddenCount) : weightedGroups), diff --git a/apps/desktop/src/components/assistant-ui/thread/status-tail-only.test.tsx b/apps/desktop/src/components/assistant-ui/thread/status-tail-only.test.tsx new file mode 100644 index 0000000000..d48ff5a095 --- /dev/null +++ b/apps/desktop/src/components/assistant-ui/thread/status-tail-only.test.tsx @@ -0,0 +1,107 @@ +// The thinking indicator (dither block) may only ever render at the TAIL of +// the thread. A message stuck status:running mid-transcript — however it got +// there (missed settle event, steer race, upstream state bug) — must render +// its content with no spinner: a live indicator above a later user message +// reads as the agent answering out of order. +import { AssistantRuntimeProvider, type ThreadMessage, useExternalStoreRuntime } from '@assistant-ui/react' +import { cleanup, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { Thread } from '.' + +const createdAt = new Date('2026-05-01T00:00:00.000Z') + +class TestResizeObserver { + observe() {} + unobserve() {} + disconnect() {} +} +vi.stubGlobal('ResizeObserver', TestResizeObserver) +vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => + window.setTimeout(() => callback(performance.now()), 0) +) +vi.stubGlobal('cancelAnimationFrame', (id: number) => window.clearTimeout(id)) +vi.stubGlobal('CSS', { escape: (str: string) => str }) + +Element.prototype.scrollTo = function scrollTo() {} + +// Enter animation fires for running messages; jsdom has no WAAPI. +Element.prototype.animate = function animate() { + return { cancel() {}, finished: Promise.resolve() } as unknown as Animation +} + +afterEach(() => { + cleanup() +}) + +const assistantMetadata = { + unstable_state: null, + unstable_annotations: [], + unstable_data: [], + steps: [], + custom: {} +} + +function user(id: string, text: string): ThreadMessage { + return { + id, + role: 'user', + content: [{ type: 'text', text }], + attachments: [], + createdAt, + metadata: { custom: {} } + } as ThreadMessage +} + +function assistant(id: string, text: string, running: boolean): ThreadMessage { + return { + id, + role: 'assistant', + content: text ? [{ type: 'text', text }] : [], + status: running ? { type: 'running' } : { type: 'complete', reason: 'stop' }, + createdAt, + metadata: assistantMetadata + } as ThreadMessage +} + +function Harness({ messages }: { messages: ThreadMessage[] }) { + const runtime = useExternalStoreRuntime({ + messages, + isRunning: messages.at(-1)?.status?.type === 'running', + onNew: async () => {} + }) + + return ( + + + + ) +} + +describe('thinking indicator is tail-only', () => { + it('shows the loading indicator on a running placeholder at the tail', async () => { + const { container } = render() + + expect(await screen.findByRole('status', { name: 'Hermes is loading a response' })).toBeTruthy() + expect(container.querySelector('[data-slot="aui_response-loading"]')).toBeTruthy() + }) + + it('never shows an indicator on a stale running message mid-transcript', async () => { + // A stranded pending bubble from an earlier turn, then a newer exchange. + const { container } = render( + + ) + + await screen.findByText('answered') + + expect(container.querySelector('[data-slot="aui_response-loading"]')).toBeNull() + expect(container.querySelector('[data-slot="aui_stream-stall"]')).toBeNull() + }) +}) diff --git a/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx b/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx index 1e6ec792d0..dbb90459cc 100644 --- a/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx @@ -13,6 +13,7 @@ import { useState } from 'react' +import { ComposerDirectiveActions } from '@/app/chat/composer/directive-actions' import { COMPOSER_DROP_ACTIVE_CLASS, COMPOSER_DROP_FADE_CLASS } from '@/app/chat/composer/drop-affordance' import { type ComposerInsertMode, @@ -35,7 +36,6 @@ import { } from '@/app/chat/composer/inline-refs' import { chipTypedPathOnSpace, pathifyRefs } from '@/app/chat/composer/path-refs' import { - COMPOSER_PLACEHOLDER_CLASS, composerPlainText, insertComposerContentsAtCaret, placeCaretEnd, @@ -773,7 +773,6 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess autoCorrect="off" className={cn( 'ui-prompt-input-editor__input max-h-48 w-full resize-none bg-transparent p-0 pr-7 text-[length:var(--conversation-text-font-size)] text-foreground/95 outline-none', - COMPOSER_PLACEHOLDER_CLASS, '**:data-ref-text:cursor-default', expanded ? 'min-h-16' : 'min-h-[1.25rem]' )} @@ -795,6 +794,7 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess spellCheck={false} suppressContentEditableWarning /> + { + window.getSelection()?.removeAllRanges() + document.body.replaceChildren() +}) + +describe('hasTextSelection', () => { + it('is false with nothing highlighted', () => { + expect(hasTextSelection()).toBe(false) + }) + + it('is true once the user has a live range', () => { + const node = document.createElement('span') + node.textContent = 'copy me' + document.body.appendChild(node) + + const range = document.createRange() + range.selectNodeContents(node) + const selection = window.getSelection()! + selection.removeAllRanges() + selection.addRange(range) + + expect(hasTextSelection()).toBe(true) + }) +}) diff --git a/apps/desktop/src/components/assistant-ui/thread/user-message.tsx b/apps/desktop/src/components/assistant-ui/thread/user-message.tsx index 16c6963e24..2ea43309df 100644 --- a/apps/desktop/src/components/assistant-ui/thread/user-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/user-message.tsx @@ -16,6 +16,13 @@ import { cn } from '@/lib/utils' import { notifyThreadEditOpen } from '@/store/thread-scroll' import { isWatchWindow } from '@/store/windows' +/** True when the user has a live text highlight (drag-select / triple-click). */ +export function hasTextSelection(): boolean { + const selection = window.getSelection() + + return Boolean(selection && !selection.isCollapsed && selection.toString().length > 0) +} + export function StickyHumanMessageContainer({ attachments, children, @@ -279,10 +286,16 @@ export const UserMessage: FC<{
{ + if (hasTextSelection()) { + return + } + event.preventDefault() setPickerOpen(true) } @@ -295,7 +308,9 @@ export const UserMessage: FC<{ aria-expanded={bodyClamped ? expanded : undefined} className={cn(bubbleClassName, !bodyClamped && 'cursor-default')} onClick={() => { - if (!bodyClamped) { + // Drag-select ends on mouseup→click; don't collapse the + // clamp just because the highlight finished. + if (hasTextSelection() || !bodyClamped) { return } @@ -310,12 +325,29 @@ export const UserMessage: FC<{ ) : ( // Always editable — clicking opens the edit composer even while a // turn streams; sending the edit reverts (interrupt + rewind). + // A live text highlight wins: finishing a drag-select must not + // open the editor and throw the selection away. + + ) +} + +const plugin: HermesPlugin = { + id: 'example', + name: 'Example Plugin', + defaultEnabled: false, + register(ctx) { + // Persisted count: hydrate once, write through on every change. + $clicks.set(ctx.storage.get('clicks', 0)) + $clicks.listen(clicks => ctx.storage.set('clicks', clicks)) + + // Hear the live gateway stream (deltas, lifecycle, tools — everything). + host.onEvent('*', () => $events.set($events.get() + 1)) + + const reset = () => { + $clicks.set(0) + host.notify({ kind: 'info', message: 'Example plugin: counter reset' }) + } + + // Provenance (source: 'plugin:example') and the namespaced registry ids + // (example:counter, …) are stamped by ctx — authors write plain + // contributions. The shared `example.reset` ACTION id links the palette + // row's hotkey hint to the keybind panel's live binding. + ctx.registerMany([ + { + id: 'counter', + area: STATUSBAR_AREAS.right, + order: 100, + render: () => + }, + { + id: 'reset', + area: PALETTE_AREA, + data: { + id: 'example.reset', + action: 'example.reset', + label: 'Example: Reset click counter', + keywords: ['example', 'plugin', 'counter'], + run: reset + } satisfies PaletteContribution + }, + { + id: 'reset', + area: KEYBINDS_AREA, + data: { + id: 'example.reset', + label: 'Example: Reset click counter', + defaults: [], + run: reset + } satisfies KeybindContribution + } + ]) + } +} + +export default plugin diff --git a/apps/desktop/src/plugins/gateway-pill/plugin.tsx b/apps/desktop/src/plugins/gateway-pill/plugin.tsx new file mode 100644 index 0000000000..60064f1c53 --- /dev/null +++ b/apps/desktop/src/plugins/gateway-pill/plugin.tsx @@ -0,0 +1,375 @@ +/** + * Gateway pill — the core statusbar gateway-health item implemented 1:1 as a + * plugin: same trigger chrome (declarative `variant: 'menu'` StatusbarItem → + * the app's own portal/popover plumbing, so nothing clips), same menu panel + * (connection/inference rows, restart, reason, RECENT ACTIVITY tail, + * messaging platforms), same copy (`useI18n`), same readiness logic + * (`evaluateRuntimeReadiness` over `host.request`). The point: a plugin can + * rebuild a REAL core feature through the SDK alone — only + * `@hermes/plugin-sdk` + react (lint-fenced). + * + * Pattern notes: + * - a module-level `atom` shares the readiness poll between the live label + * elements and the menu panel (the same primitive `host.state` uses); + * - label/detail/icon of a DATA item are ReactNodes, so they can be tiny + * components that subscribe — a static item shape with live innards. + */ + +import { + atom, + Button, + cn, + evaluateRuntimeReadiness, + type HermesPlugin, + host, + icons, + LogView, + type RuntimeReadinessResult, + type StatusbarItem, + StatusDot, + type StatusResponse, + type StatusTone, + Tip, + useI18n, + useValue +} from '@hermes/plugin-sdk' +import { type ReactNode, useEffect, useRef, useState } from 'react' + +const READINESS_POLL_MS = 15_000 +const LOG_TAIL = 120 +const LOG_VISIBLE = 40 +const LOG_POLL_MS = 3_000 + +// Per-connection WebSocket churn (accept/close/heartbeat) drowns out anything +// useful — strip it so the tail reads as real gateway activity at a glance. +const LOG_NOISE_RE = /\bws (?:accepted|closed|response sent|ping|pong)\b/i + +// Strip leading "YYYY-MM-DD HH:MM:SS,mmm " and "[runtime_id] " prefixes. +const TIMESTAMP_RE = /^\d{4}-\d{2}-\d{2}[ T]\d{2}:\d{2}:\d{2}[,.\d]*\s+/ +const RUNTIME_BRACKET_RE = /^\[[^\]]+]\s+/ +const trimLogLine = (raw: string) => raw.trim().replace(TIMESTAMP_RE, '').replace(RUNTIME_BRACKET_RE, '') + +const PLATFORM_TONE: Record = { + connected: 'good', + connecting: 'warn', + retrying: 'warn', + pending_restart: 'warn', + startup_failed: 'bad', + fatal: 'bad' +} + +const prettyState = (state: string) => state.replace(/_/g, ' ').replace(/^./, c => c.toUpperCase()) + +const SYSTEM_PANEL_ROUTE = '/command-center?section=system' + +// --------------------------------------------------------------------------- +// Readiness poll — one loop at plugin scope, shared by label + panel. +// --------------------------------------------------------------------------- + +const $readiness = atom(null) + +function startReadinessPoll() { + let timer: null | number = null + + const stop = () => { + if (timer !== null) { + window.clearInterval(timer) + timer = null + } + + $readiness.set(null) + } + + const refresh = () => + evaluateRuntimeReadiness(host.request) + .then(next => $readiness.set(next)) + .catch(() => undefined) + + const sync = (gateway: string) => { + if (gateway !== 'open') { + stop() + + return + } + + if (timer === null) { + void refresh() + timer = window.setInterval(() => void refresh(), READINESS_POLL_MS) + } + } + + sync(host.state.gateway.get()) + host.state.gateway.listen(sync) +} + +// --------------------------------------------------------------------------- +// Live trigger innards (the item is static DATA; these subscribe). +// --------------------------------------------------------------------------- + +function useHealth() { + const gateway = useValue(host.state.gateway) + const readiness = useValue($readiness) + + return { + connecting: gateway === 'connecting', + open: gateway === 'open', + readiness, + ready: gateway === 'open' && readiness?.ready === true + } +} + +function PillIcon() { + const { connecting, open, ready } = useHealth() + + return ( + + {ready ? : } + + ) +} + +function PillDetail() { + const { t } = useI18n() + const copy = t.shell.statusbar + const { connecting, open, readiness, ready } = useHealth() + + const detail = open + ? ready + ? copy.gatewayReady + : readiness + ? copy.gatewayNeedsSetup + : copy.gatewayChecking + : connecting + ? copy.gatewayConnecting + : copy.gatewayOffline + + return <>{detail} +} + +// --------------------------------------------------------------------------- +// The menu panel — the real GatewayMenuPanel, rebuilt on SDK doors. +// --------------------------------------------------------------------------- + +/** Live gui-log tail while the popover is mounted (i.e. open). */ +function useGatewayLogTail(): string[] { + const [lines, setLines] = useState([]) + + useEffect(() => { + let cancelled = false + + const load = () => + host + .logs({ file: 'gui', lines: LOG_TAIL }) + .then(res => { + if (!cancelled) { + setLines( + res.lines + .map(line => line.trim()) + .filter(line => line && !LOG_NOISE_RE.test(line)) + .slice(-LOG_VISIBLE) + ) + } + }) + .catch(() => undefined) + + void load() + const timer = window.setInterval(load, LOG_POLL_MS) + + return () => { + cancelled = true + window.clearInterval(timer) + } + }, []) + + return lines +} + +function Section({ children, className }: { children: ReactNode; className?: string }) { + return
{children}
+} + +function SectionLabel({ children }: { children: string }) { + return ( +
{children}
+ ) +} + +function GatewayMenuPanel({ onClose }: { onClose: () => void }) { + const { t } = useI18n() + const copy = t.shell.gatewayMenu + const gateway = useValue(host.state.gateway) + const { readiness, ready } = useHealth() + const [snapshot, setSnapshot] = useState(null) + const recentLogs = useGatewayLogTail() + + useEffect(() => { + void host + .status() + .then(setSnapshot) + .catch(() => undefined) + }, []) + + const openSystem = () => { + onClose() + host.navigate(SYSTEM_PANEL_ROUTE) + } + + const restart = () => { + onClose() + void host.restartGateway().catch(() => undefined) + } + + const gatewayOpen = gateway === 'open' + const gatewayConnecting = gateway === 'connecting' + + const connectionLabel = gatewayOpen + ? copy.connected + : gatewayConnecting + ? copy.connecting + : prettyState(gateway || copy.offline) + + const inferenceLabel = gatewayOpen + ? readiness?.ready + ? copy.inferenceReady + : readiness + ? copy.inferenceNotReady + : copy.checkingInference + : copy.disconnected + + const platforms = Object.entries(snapshot?.gateway_platforms || {}).sort(([l], [r]) => l.localeCompare(r)) + + // Keep the tail pinned to the latest line as it streams. + const logScrollRef = useRef(null) + + useEffect(() => { + const el = logScrollRef.current + + if (el) { + el.scrollTop = el.scrollHeight + } + }, [recentLogs]) + + return ( +
+
+
+ + + {connectionLabel} + + + + {inferenceLabel} + +
+
+ + + + + + +
+
+ + {readiness?.reason && ( +
+
{readiness.reason}
+
+ )} + + {recentLogs.length > 0 && ( +
+
+ {copy.recentActivity} + +
+ + {recentLogs.map(trimLogLine).join('\n')} + +
+ )} + + {platforms.length > 0 && ( +
+ {copy.messagingPlatforms} +
    + {platforms.map(([name, platform]) => ( +
  • + {name} + + + {prettyState(platform.state)} + +
  • + ))} +
+
+ )} +
+ ) +} + +function PillLabel() { + const { t } = useI18n() + + return <>{t.shell.statusbar.gateway} +} + +// --------------------------------------------------------------------------- + +const plugin: HermesPlugin = { + id: 'gateway-pill', + name: 'Gateway Pill', + register(ctx) { + startReadinessPoll() + + // Declarative menu item — the app's own trigger/popover chrome renders it + // (portal, w-72, side=top), the plugin supplies live innards + the panel. + ctx.register({ + id: 'pill', + area: 'statusBar.right', + order: 90, + data: { + icon: , + id: 'gateway-pill', + label: , + detail: , + menuClassName: 'w-72', + menuContent: (close: () => void) => , + variant: 'menu' + } satisfies StatusbarItem + }) + } +} + +export default plugin diff --git a/apps/desktop/src/plugins/hello-runtime/plugin.runtime.js b/apps/desktop/src/plugins/hello-runtime/plugin.runtime.js new file mode 100644 index 0000000000..6647138e1f --- /dev/null +++ b/apps/desktop/src/plugins/hello-runtime/plugin.runtime.js @@ -0,0 +1,38 @@ +/** + * Runtime-loaded example — this file is NOT bundled as a module: it ships as + * raw text (`?raw`) and goes through the real runtime pipeline (specifier + * rewrite -> SDK/react shim blobs -> blob import -> register). Plain ESM js + * with `jsx()` calls — exactly what an agent (or a compiler) writes into + * `~/.hermes/desktop-plugins//plugin.js`. + */ + +import { cn, host, Tip, useValue } from '@hermes/plugin-sdk' +import { jsx, jsxs } from 'react/jsx-runtime' + +function RuntimeChip() { + const gateway = useValue(host.state.gateway) + + return jsx(Tip, { + label: `Loaded at RUNTIME through blob import + SDK injection (gateway: ${gateway})`, + children: jsxs('span', { + className: cn( + 'inline-flex h-full items-center gap-1 px-1.5 text-[0.6875rem]', + 'text-(--ui-text-tertiary)' + ), + children: [jsx('span', { 'aria-hidden': true, children: '⚡' }), jsx('span', { children: 'runtime' })] + }) + }) +} + +export default { + id: 'hello-runtime', + name: 'Hello Runtime', + register(ctx) { + ctx.register({ + id: 'chip', + area: 'statusBar.right', + order: 110, + render: () => jsx(RuntimeChip, {}) + }) + } +} diff --git a/apps/desktop/src/plugins/kanban/api.ts b/apps/desktop/src/plugins/kanban/api.ts new file mode 100644 index 0000000000..16403bbc4e --- /dev/null +++ b/apps/desktop/src/plugins/kanban/api.ts @@ -0,0 +1,253 @@ +/** + * Kanban data layer. Everything goes through `ctx.rest` — the plugin's own + * `/api/plugins/kanban/*` FastAPI router (`plugins/kanban/dashboard/plugin_api.py`), + * reused as-is via the desktop's namespace-scoped REST door. No new backend. + * + * Fetching, caching, polling, dedupe, and invalidation are React Query's job + * (the app's standard, via the SDK). This module owns the query keys, the REST + * calls, and the selected-board atom — every call passes `?board=` so the + * desktop's selection never flips the server-wide current-board pointer. + */ + +import { atom, type PluginRestOptions, type PluginStorage, queryClient } from '@hermes/plugin-sdk' + +import type { + BoardMeta, + BoardsResponse, + KanbanBoard, + KanbanProfile, + KanbanProject, + KanbanTask, + KanbanTaskDetail, + OrchestrationSettings, + TaskEstimate, + WorkerLog +} from './types' + +type Rest = (path: string, opts?: PluginRestOptions) => Promise +type Socket = (path: string, onMessage: (data: unknown) => void) => () => void + +let rest: null | Rest = null + +/** Selected board slug ('' = the server's current board). Persisted. */ +export const $boardSlug = atom('') + +/** Whether the "how this board works" intro was dismissed. Persisted. */ +export const $introDismissed = atom(false) + +/** Sub-group the Running lane by assignee (the dashboard's "lanes by + * profile"). Persisted. */ +export const $lanesByProfile = atom(false) + +/** Per-lane collapse OVERRIDES (true=collapsed, false=expanded). Absence means + * auto: empty lanes collapse to a rail, occupied lanes expand. Persisted. */ +export const $collapsedLanes = atom>({}) + +const BOARD_SLUG_KEY = 'boardSlug' +const INTRO_KEY = 'introDismissed' +const LANES_KEY = 'lanesByProfile' +const COLLAPSED_KEY = 'collapsedLanes' + +/** One live `task_events` frame → precise cache invalidation: the board, plus + * each touched task's detail. The polls (8s board / 4s drawer) stay as the + * fallback — the socket just makes the board feel instant. */ +function onEventsFrame(slug: string, data: unknown): void { + const events = (data as { events?: Array<{ task_id?: string }> })?.events + + if (!events?.length) { + return + } + + void queryClient.invalidateQueries({ queryKey: ['kanban', 'board'] }) + // Any event can change a board's card count — keep the switcher badge honest. + void queryClient.invalidateQueries({ queryKey: BOARDS_KEY }) + + for (const taskId of new Set(events.map(event => event.task_id).filter(Boolean))) { + void queryClient.invalidateQueries({ queryKey: taskKey(slug, taskId!) }) + } +} + +// A persisted, subscribable atom (the structural slice we need — avoids +// importing nanostore's type just to describe one). +interface Persisted { + get(): T + set(value: T): void + listen(cb: (value: T) => void): () => void +} + +/** Bind the plugin's doors at register time and return a disposer the host + * runs on unload/disable — so nothing (store sync, socket) survives a toggle + * or duplicates on re-enable. The events socket is pinned to a board at + * handshake, so a board switch closes + reopens it. */ +export function bindApi(r: Rest, storage: PluginStorage, socket: Socket): () => void { + rest = r + const unsubs: Array<() => void> = [] + + // Hydrate an atom from storage and keep storage in sync with it. + const persist = (atom: Persisted, key: string, fallback: T) => { + atom.set(storage.get(key, fallback)) + unsubs.push(atom.listen(value => storage.set(key, value))) + } + + persist($boardSlug, BOARD_SLUG_KEY, '') + persist($introDismissed, INTRO_KEY, false) + persist($lanesByProfile, LANES_KEY, false) + persist($collapsedLanes, COLLAPSED_KEY, {}) + + let close: (() => void) | null = null + + const open = (slug: string) => { + close?.() + close = socket(slug ? `/events?board=${encodeURIComponent(slug)}` : '/events', data => onEventsFrame(slug, data)) + } + + open($boardSlug.get()) + unsubs.push($boardSlug.listen(open)) + + return () => { + unsubs.forEach(unsub => unsub()) + close?.() + rest = null + } +} + +function call(path: string, opts?: PluginRestOptions): Promise { + return rest ? rest(path, opts) : Promise.reject(new Error('kanban api not ready')) +} + +/** Append the selected board (and other params) to a path. */ +function withBoard(path: string, params: Record = {}): string { + const search = new URLSearchParams(params) + const slug = $boardSlug.get() + + if (slug) { + search.set('board', slug) + } + + const qs = search.toString() + + return qs ? `${path}?${qs}` : path +} + +// ── query keys (all board-scoped so switching boards is a clean cache miss) ── + +export const boardKey = (slug: string, archived: boolean) => ['kanban', 'board', slug, archived] as const +export const taskKey = (slug: string, id: string) => ['kanban', 'task', slug, id] as const +export const logKey = (slug: string, id: string) => ['kanban', 'log', slug, id] as const +export const BOARDS_KEY = ['kanban', 'boards'] as const +export const PROFILES_KEY = ['kanban', 'profiles'] as const +export const PROJECTS_KEY = ['kanban', 'projects'] as const +export const ORCHESTRATION_KEY = ['kanban', 'orchestration'] as const + +// ── reads ───────────────────────────────────────────────────────────────────── + +export const fetchBoard = (archived: boolean) => + call(withBoard('/board', archived ? { include_archived: 'true' } : {})) + +export const fetchTask = (id: string) => call(withBoard(`/tasks/${id}`)) + +/** Worker stdout/stderr tail (last 16 KiB — plenty for the drawer). */ +export const fetchLog = (id: string) => call(withBoard(`/tasks/${id}/log`, { tail: '16384' })) + +export const fetchBoards = () => call('/boards') + +export const fetchProfiles = () => call<{ profiles: KanbanProfile[] }>('/profiles') + +/** First-class Hermes projects, for scoping a board's default workspace. */ +export const fetchProjects = () => call<{ projects: KanbanProject[] }>('/projects') + +export const fetchOrchestration = () => call('/orchestration') + +// ── writes ──────────────────────────────────────────────────────────────────── + +// Every board edit nudges the dispatcher (debounced, fire-and-forget) so the +// change takes effect NOW instead of on the next 60s tick — create a ready +// task and the worker spawns immediately, no manual "nudge" ritual. The tick +// is lock-guarded and ~1ms when there's nothing to do, so over-nudging is +// free; failures are non-events (the periodic tick still exists). +let nudgeTimer: null | ReturnType = null + +function autoNudge(): void { + if (nudgeTimer != null) { + clearTimeout(nudgeTimer) + } + + nudgeTimer = setTimeout(() => { + nudgeTimer = null + nudgeDispatcher().catch(() => undefined) + }, 400) +} + +/** Resolve the write, then kick the dispatcher. Rejections pass through. */ +function nudged(write: Promise): Promise { + return write.then(value => { + autoNudge() + + return value + }) +} + +export const patchTask = (id: string, patch: Record) => + nudged(call(withBoard(`/tasks/${id}`), { method: 'PATCH', body: patch })) + +export const createTask = (body: Record) => + nudged(call<{ task: KanbanTask | null; warning?: string }>(withBoard('/tasks'), { method: 'POST', body })) + +// Deleting can unblock dependants (a gone parent no longer gates), so it +// nudges too. +export const deleteTask = (id: string) => nudged(call(withBoard(`/tasks/${id}`), { method: 'DELETE' })) + +/** One patch, many ids — independent per-id application; returns per-id + * outcomes so the UI can toast partial failures. */ +export const bulkTasks = (ids: string[], patch: Record) => + nudged( + call<{ results: Array<{ id: string; ok: boolean; error?: string }> }>(withBoard('/tasks/bulk'), { + method: 'POST', + body: { ids, ...patch } + }) + ) + +export const addComment = (id: string, body: string) => + call(withBoard(`/tasks/${id}/comments`), { method: 'POST', body: { author: 'desktop', body } }) + +export const reassignTask = (id: string, profile: string) => + nudged(call(withBoard(`/tasks/${id}/reassign`), { method: 'POST', body: { profile, reclaim_first: true } })) + +export const reclaimTask = (id: string) => nudged(call(withBoard(`/tasks/${id}/reclaim`), { method: 'POST', body: {} })) + +export const uploadAttachment = (id: string, upload: { filename: string; contentType?: string; bytes: ArrayBuffer }) => + call(withBoard(`/tasks/${id}/attachments`), { method: 'POST', upload }) + +export const createBoard = (slug: string, name: string, projectId?: string) => + call<{ board: { slug: string } }>('/boards', { + method: 'POST', + body: { slug, name, ...(projectId ? { project_id: projectId } : {}) } + }) + +/** Rough auxiliary-model estimate for a task (tokens + complexity). Makes a + * model call — gate behind an explicit user action + disclaimer. */ +export const estimateTask = (id: string) => + call(withBoard(`/tasks/${id}/estimate`), { method: 'POST', body: {} }) + +/** Estimate from typed title/body before a task exists (create dialog). */ +export const estimateNew = (title: string, body: string) => + call('/estimate', { method: 'POST', body: { title, body: body || undefined } }) + +/** Edit a board's display metadata + default project directory. Pass + * `default_workdir: ''` to clear it. Slug is immutable. */ +export const updateBoard = (slug: string, patch: Record) => + call<{ board: BoardMeta }>(`/boards/${encodeURIComponent(slug)}`, { method: 'PATCH', body: patch }) + +export const nudgeDispatcher = () => call<{ spawned?: unknown[] }>(withBoard('/dispatch'), { method: 'POST', body: {} }) + +export const saveOrchestration = (patch: Record) => + call('/orchestration', { method: 'PUT', body: patch }) + +export const saveProfileDescription = (name: string, description: string) => + call(`/profiles/${encodeURIComponent(name)}`, { method: 'PATCH', body: { description } }) + +export const autoDescribeProfile = (name: string) => + call<{ ok: boolean; reason?: null | string; description?: null | string }>( + `/profiles/${encodeURIComponent(name)}/describe-auto`, + { method: 'POST', body: { overwrite: true } } + ) diff --git a/apps/desktop/src/plugins/kanban/board-switcher.tsx b/apps/desktop/src/plugins/kanban/board-switcher.tsx new file mode 100644 index 0000000000..6fe961eee8 --- /dev/null +++ b/apps/desktop/src/plugins/kanban/board-switcher.tsx @@ -0,0 +1,243 @@ +/** + * Titlebar board switcher — the board page projects this into `titleBar.center` + * (where chat shows the session-title dropdown) via ``, so it + * exists exactly while the page is mounted — no route sniffing. Same chrome as + * the session title: quiet label + chevron, menu on click. + */ + +import { + Button, + Codicon, + Dialog, + DialogContent, + DialogFooter, + DialogHeader, + DialogTitle, + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger, + host, + Input, + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue, + useMutation, + useQuery, + useQueryClient, + useValue +} from '@hermes/plugin-sdk' +import { useEffect, useState } from 'react' + +import { $boardSlug, BOARDS_KEY, createBoard, fetchBoards, fetchProjects, PROJECTS_KEY, updateBoard } from './api' +import type { BoardMeta } from './types' +import { errText, FIELD_LABEL, useKanban } from './ui' + +const NO_PROJECT = '__none__' + +/** Board scope = a first-class Hermes project. Its primary repo becomes the + * board's default workspace root; new tasks inherit it as a worktree with a + * deterministic branch. "No project" falls back to scratch sandboxes. */ +function ProjectPicker({ onChange, value }: { onChange: (id: string) => void; value: string }) { + const k = useKanban() + const { data } = useQuery({ queryKey: PROJECTS_KEY, queryFn: fetchProjects, staleTime: 30_000 }) + const projects = data?.projects ?? [] + + return ( + + ) +} + +function NewBoardDialog({ onClose, open }: { onClose: () => void; open: boolean }) { + const k = useKanban() + const qc = useQueryClient() + const [name, setName] = useState('') + const [project, setProject] = useState('') + + const slug = name + .trim() + .toLowerCase() + .replace(/[^a-z0-9]+/g, '-') + .replace(/^-+|-+$/g, '') + + useEffect(() => { + if (open) { + setName('') + setProject('') + } + }, [open]) + + const create = useMutation({ + mutationFn: () => createBoard(slug, name.trim(), project || undefined), + onError: err => host.notify({ kind: 'error', message: errText(err) }), + onSuccess: result => { + $boardSlug.set(result.board.slug) + void qc.invalidateQueries({ queryKey: BOARDS_KEY }) + onClose() + } + }) + + return ( + !o && onClose()} open={open}> + + + {k.newBoard} + +
+ + +
+ + + + +
+
+ ) +} + +function BoardSettingsDialog({ board, onClose }: { board: BoardMeta | null; onClose: () => void }) { + const k = useKanban() + const qc = useQueryClient() + const [name, setName] = useState('') + const [project, setProject] = useState('') + + useEffect(() => { + if (board) { + setName(board.name || '') + setProject(board.project_id || '') + } + }, [board]) + + const save = useMutation({ + // Slug is immutable; send name + project_id ('' clears the scope, which + // also drops the mirrored default_workdir on the backend). + mutationFn: () => updateBoard(board!.slug, { name: name.trim(), project_id: project }), + onError: err => host.notify({ kind: 'error', message: errText(err) }), + onSuccess: () => { + void qc.invalidateQueries({ queryKey: BOARDS_KEY }) + onClose() + } + }) + + return ( + !o && onClose()} open={Boolean(board)}> + + + {board ? k.boardSettingsFor(board.name || board.slug) : k.boardSettings} + +
+ + +
+ + + + +
+
+ ) +} + +export function BoardSwitcher() { + const k = useKanban() + const slug = useValue($boardSlug) + const { data: boards } = useQuery({ queryFn: fetchBoards, queryKey: BOARDS_KEY, staleTime: 30_000 }) + const [adding, setAdding] = useState(false) + const [settingsFor, setSettingsFor] = useState(null) + + if (!boards) { + return null + } + + const currentSlug = slug || boards.current + const current = boards.boards.find(meta => meta.slug === currentSlug) + const label = current?.name || current?.slug || k.board + + return ( + <> + + + + + + {boards.boards.map(meta => ( + $boardSlug.set(meta.slug === boards.current ? '' : meta.slug)} + > + {meta.name || meta.slug} + {typeof meta.total === 'number' && ( + {meta.total} + )} + {meta.slug === currentSlug && } + + ))} + + {current && ( + setSettingsFor(current)}> + + {k.boardSettings} + + )} + setAdding(true)}> + + {k.newBoardDots} + + + + setAdding(false)} open={adding} /> + setSettingsFor(null)} /> + + ) +} diff --git a/apps/desktop/src/plugins/kanban/board.tsx b/apps/desktop/src/plugins/kanban/board.tsx new file mode 100644 index 0000000000..3c4cdd9f97 --- /dev/null +++ b/apps/desktop/src/plugins/kanban/board.tsx @@ -0,0 +1,1430 @@ +/** + * The Kanban board page — mounted at `/kanban` (a ROUTES_AREA contribution) in + * the workspace pane. The desktop port of the dashboard board: one compact + * header row (count, filter kebab, search, settings, new task — the board + * SWITCHER lives in the titlebar, see board-switcher.tsx), columns in + * BOARD_COLUMNS order, drag-to-move (optimistic, workflow-checked), + * ⌘-click multi-select with a floating bulk bar, right-click actions, and + * the detail drawer. Dispatch nudges ride every write (see api.ts). + */ + +import { + Button, + cn, + Codicon, + compactNumber, + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger, + Contribute, + Dialog, + DialogContent, + DialogFooter, + DialogHeader, + DialogTitle, + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger, + ErrorState, + host, + Input, + Loader, + SearchField, + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue, + Switch, + Textarea, + Tip, + TITLEBAR_AREAS, + useGrabScroll, + useMutation, + useQuery, + useQueryClient, + useValue +} from '@hermes/plugin-sdk' +import { + type CSSProperties, + type DragEvent as ReactDragEvent, + type ReactNode, + useEffect, + useMemo, + useRef, + useState +} from 'react' + +import { + $boardSlug, + $collapsedLanes, + $introDismissed, + $lanesByProfile, + boardKey, + BOARDS_KEY, + bulkTasks, + createTask, + deleteTask, + estimateNew, + fetchBoard, + fetchBoards, + fetchProfiles, + patchTask, + PROFILES_KEY +} from './api' +import { BoardSwitcher } from './board-switcher' +import { TaskDrawer } from './drawer' +import { EMPTY_OVERRIDE, ModelOverrideField, overrideCreateFields, type TaskModelOverride } from './model-override' +import { OrchestrationPanel } from './orchestration' +import { columnMeta, type KanbanBoard, type KanbanTask, type TaskEstimate } from './types' +import { + $newTaskLane, + ago, + type ArcState, + arcState, + Avatar, + columnHelp, + columnLabel, + errText, + FIELD_LABEL, + isLockedTarget, + lockedReason, + RunClock, + shortId, + useDefaultAssignee, + useKanban, + useOrchestration +} from './ui' + +// ── optimistic board edits (reconciled by the follow-up refresh) ───────────── + +function moveCard(board: KanbanBoard, id: string, toStatus: string): KanbanBoard { + let moved: KanbanTask | undefined + + const columns = board.columns.map(col => ({ + ...col, + tasks: col.tasks.filter(task => { + if (task.id !== id) { + return true + } + + moved = { ...task, status: toStatus } + + return false + }) + })) + + if (!moved) { + return board + } + + return { + ...board, + columns: columns.map(col => (col.name === toStatus ? { ...col, tasks: [moved!, ...col.tasks] } : col)) + } +} + +function removeCard(board: KanbanBoard, id: string): KanbanBoard { + return { ...board, columns: board.columns.map(col => ({ ...col, tasks: col.tasks.filter(t => t.id !== id) })) } +} + +// ── card ───────────────────────────────────────────────────────────────────── + +function Meta({ children, icon }: { children: ReactNode; icon: string }) { + return ( + + + {children} + + ) +} + +function CardFooter({ arc, task }: { arc: ArcState | null; task: KanbanTask }) { + const k = useKanban() + const created = ago(task.created_at) + const links = task.link_counts ? task.link_counts.parents + task.link_counts.children : 0 + const fallback = useDefaultAssignee() + const orchestrator = useOrchestration()?.resolved_orchestrator_profile ?? '' + // Ready + no assignee: with a configured default assignee the dispatcher + // auto-assigns on its next tick (#27145) — say THAT, not "won't run". Only + // a board with no fallback has the genuine silent failure. + const unassignedReady = task.status === 'ready' && !task.assignee + + // The agent on the hook for a queued card: the explicit assignee, else the + // auto-default (ready), else the specifier that rewrites triage cards. + const attached = task.assignee || (task.status === 'ready' ? fallback : task.status === 'triage' ? orchestrator : '') + + const meta = columnMeta(task.status) + + return ( +
+ {arc === 'queued' && attached ? ( + // WHO is coming for the card. The arc only animates once the agent is + // actually working; while queued, the named chip carries "attached". + + + + + {!task.assignee && '→ '} + {attached} + + + + ) : task.assignee ? ( + + ) : null} + {arc === 'running' && ( + + + + + + )} + {arc === 'stale' && ( + + {k.noHeartbeat} + + )} + {unassignedReady && !fallback && ( + + + + {k.wontRun} + + + )} +
+ {typeof task.priority === 'number' && task.priority > 0 && ( + + + {task.priority} + + )} + {task.progress && task.progress.total > 0 && ( + + {task.progress.done}/{task.progress.total} + + )} + {Boolean(task.comment_count) && {task.comment_count}} + {links > 0 && {links}} + {task.warnings && task.warnings.count > 0 && ( + + + {task.warnings.count} + + )} + {created && !task.assignee && !unassignedReady ? ( + {created} + ) : null} + {shortId(task.id)} +
+
+ ) +} + +function Card({ + columns, + onDelete, + onMove, + onOpen, + onToggleSelect, + selected, + task +}: { + columns: string[] + onDelete: (id: string) => void + onMove: (id: string, status: string) => void + onOpen: (id: string) => void + onToggleSelect: (id: string) => void + selected: boolean + task: KanbanTask +}) { + const k = useKanban() + const [dragging, setDragging] = useState(false) + const meta = columnMeta(task.status) + const summary = task.latest_summary || task.body + const fallback = useDefaultAssignee() + const arc = arcState(task, fallback) + + return ( + + +
(event.metaKey || event.ctrlKey ? onToggleSelect(task.id) : onOpen(task.id))} + onDragEnd={() => setDragging(false)} + onDragStart={event => { + event.dataTransfer.setData('text/plain', task.id) + event.dataTransfer.effectAllowed = 'move' + // Snapshot the drag image before dimming the source, so the ghost + // stays a solid card (dimming first would bake 40% into it). + event.dataTransfer.setDragImage(event.currentTarget, event.nativeEvent.offsetX, event.nativeEvent.offsetY) + setDragging(true) + }} + style={{ '--kanban-tone': meta.tone, borderLeftColor: meta.tone } as CSSProperties} + > + {/* Machine-activity arc: animates ONLY while an agent is actually on + the card (claimed + working; amber when the heartbeat is gone). + Queued attachment is the footer's named-agent chip — a moving + border on an idle card would lie. Hidden during drag/selection + so those states stay legible. */} + {(arc === 'running' || arc === 'stale') && !dragging && !selected && ( + + )} + + {task.title || task.id} + + {summary && ( + {summary} + )} + +
+
+ + onOpen(task.id)}> + + {k.open} + + onToggleSelect(task.id)}> + + {selected ? k.deselect : k.select} + + + {columns + .filter(name => name !== task.status && !isLockedTarget(name)) + .map(name => ( + onMove(task.id, name)}> + + {k.moveTo(columnLabel(k, name))} + + ))} + + onDelete(task.id)} variant="destructive"> + + {k.delete} + + +
+ ) +} + +// ── column ─────────────────────────────────────────────────────────────────── + +function Column({ + collapsed, + column, + columns, + onAdd, + onDelete, + onDropTask, + onMove, + onOpen, + onToggle, + onToggleSelect, + selected +}: { + collapsed: boolean + column: { name: string; tasks: KanbanTask[] } + columns: string[] + onAdd: (status: string) => void + onDelete: (id: string) => void + onDropTask: (id: string, status: string) => void + onMove: (id: string, status: string) => void + onOpen: (id: string) => void + onToggle: () => void + onToggleSelect: (id: string) => void + selected: ReadonlySet +}) { + const k = useKanban() + const [over, setOver] = useState(false) + const meta = columnMeta(column.name) + const label = columnLabel(k, column.name) + const locked = isLockedTarget(column.name) + const byProfile = useValue($lanesByProfile) + + // The dashboard's "lanes by profile": sub-group Running by assignee so a + // fleet's in-flight work reads per-worker. Null = flat (off, or trivial). + const lanes = useMemo(() => { + if (!byProfile || column.name !== 'running' || column.tasks.length === 0) { + return null + } + + const groups = new Map() + + for (const task of column.tasks) { + const key = task.assignee || UNASSIGNED_LANE + groups.set(key, [...(groups.get(key) ?? []), task]) + } + + return [...groups.entries()].sort(([a], [b]) => a.localeCompare(b)) + }, [byProfile, column]) + + const dragHandlers = { + onDragLeave: () => setOver(false), + onDragOver: (event: ReactDragEvent) => { + // Locked lanes don't preventDefault → the OS shows the no-drop cursor + // and the drop event never fires. The lane is honest about itself. + if (locked) { + event.dataTransfer.dropEffect = 'none' + + return + } + + event.preventDefault() + event.dataTransfer.dropEffect = 'move' + setOver(true) + }, + onDrop: (event: ReactDragEvent) => { + event.preventDefault() + setOver(false) + const id = event.dataTransfer.getData('text/plain') + + if (id) { + onDropTask(id, column.name) + } + } + } + + const wash = over && !locked ? 'bg-(--ui-bg-quinary)' : 'bg-[color-mix(in_srgb,var(--ui-bg-quinary)_50%,transparent)]' + + // Collapsed = a thin vertical rail: dot, sideways label, count. Still a live + // drop target (drop straight onto the rail); click expands. The dot sits in + // the same h-5 header row as an expanded lane's, so dots align across the + // board regardless of collapse state. + if (collapsed) { + return ( + + ) + } + + return ( +
+
+ + + + {label} + + + {column.tasks.length} + +
+
+ {lanes + ? lanes.map(([assignee, tasks]) => ( +
+
+ {assignee !== UNASSIGNED_LANE && } + {assignee} + {tasks.length} +
+ {tasks.map(task => ( + + ))} +
+ )) + : column.tasks.map(task => ( + + ))} + {/* Jira-style lane add — dashed, faded in on lane hover. Opacity (not + display) so it always holds its slot and never thrashes layout. + Locked lanes get none: you can't create into a system state. */} + {!locked && ( + + )} + {column.tasks.length === 0 && ( +
+ {k.empty} +
+ )} +
+
+ ) +} + +// ── dialogs ────────────────────────────────────────────────────────────────── + +const NO_PARENT = '__none__' +const PARKED = '__parked__' +const WORKSPACE_KINDS = ['scratch', 'worktree', 'dir'] as const + +function Field({ children, label }: { children: ReactNode; label: string }) { + return ( + + ) +} + +function NewTaskDialog({ + onClose, + parents, + target +}: { + onClose: () => void + parents: Array<{ id: string; title: string }> + target: null | string +}) { + const k = useKanban() + const qc = useQueryClient() + const { data: roster } = useQuery({ queryKey: PROFILES_KEY, queryFn: fetchProfiles, staleTime: 60_000 }) + // Title-only creates must RUN: "auto" resolves to the orchestration default + // (ultimately the active profile), applied at create time. Never silently + // unassigned — parking a card is the explicit choice, not the default. + const resolvedDefault = useOrchestration()?.resolved_default_assignee || 'default' + + // Board-level workspace default: a task inherits the current board's + // configured project dir (scratch when unset, worktree in a git repo, else + // dir) unless the operator overrides it below. Set the board default in the + // board switcher's "Board settings…". + const selectedSlug = useValue($boardSlug) + const { data: boards } = useQuery({ queryKey: BOARDS_KEY, queryFn: fetchBoards, staleTime: 30_000 }) + const currentBoard = boards?.boards.find(b => b.slug === (selectedSlug || boards.current)) + const boardDefaultKind = currentBoard?.default_workspace_kind || 'scratch' + const boardDefaultDir = currentBoard?.default_workdir || '' + + const isTriage = target === 'triage' + const [title, setTitle] = useState('') + const [bodyText, setBodyText] = useState('') + const [assignee, setAssignee] = useState('') + const [priority, setPriority] = useState('0') + const [skills, setSkills] = useState('') + const [workspaceKind, setWorkspaceKind] = useState(boardDefaultKind) + // Empty = inherit the board's default project dir (backend resolves it); + // a path here overrides just this task. Only meaningful for dir/worktree. + const [workspacePath, setWorkspacePath] = useState('') + const [parent, setParent] = useState('') + const [modelOverride, setModelOverride] = useState(EMPTY_OVERRIDE) + const [goalMode, setGoalMode] = useState(false) + const [busy, setBusy] = useState(false) + const [error, setError] = useState(null) + const [estimate, setEstimate] = useState(null) + + // Rough effort estimate from the typed title/body (before the task exists), + // via the auto-routed auxiliary model. Makes a model call — explicit action. + const estMut = useMutation({ + mutationFn: () => estimateNew(title.trim(), bodyText.trim()), + onError: err => host.notify({ kind: 'error', message: errText(err) }), + onSuccess: r => { + if (r.ok) { + setEstimate(r) + } else { + host.notify({ kind: 'warning', message: r.reason || k.couldNotEstimate }) + } + } + }) + + // Reset per open — the dialog is externally controlled (open = target set), + // so onOpenChange(true) never fires; key the reset off `target` (and the + // resolved board default, which may arrive after the first open). + useEffect(() => { + if (target) { + setTitle('') + setBodyText('') + setAssignee('') + setPriority('0') + setSkills('') + setWorkspaceKind(boardDefaultKind) + setWorkspacePath('') + setParent('') + setModelOverride(EMPTY_OVERRIDE) + setGoalMode(false) + setError(null) + setBusy(false) + setEstimate(null) + } + }, [target, boardDefaultKind]) + + const submit = async () => { + const trimmed = title.trim() + + if (!trimmed || !target || busy) { + return + } + + setBusy(true) + setError(null) + + try { + const skillList = skills + .split(',') + .map(s => s.trim()) + .filter(Boolean) + + // create() derives status (triage flag → 'triage', else 'ready'); move to + // the requested column when they differ, so a per-column add lands right. + const { task, warning } = await createTask({ + assignee: assignee === PARKED ? undefined : assignee || resolvedDefault, + body: bodyText.trim() || undefined, + goal_mode: goalMode, + parents: parent ? [parent] : undefined, + priority: Number(priority) || 0, + skills: skillList.length ? skillList : undefined, + title: trimmed, + triage: isTriage, + workspace_kind: workspaceKind, + ...overrideCreateFields(modelOverride), + // Empty → backend inherits the board's default project dir. + workspace_path: workspaceKind !== 'scratch' && workspacePath.trim() ? workspacePath.trim() : undefined + }) + + if (task && task.status !== target) { + await patchTask(task.id, { status: target }) + } + + // Dispatcher-presence warning ("this ready task will sit idle") — not an + // error, but the user should know. + if (warning) { + host.notify({ kind: 'warning', message: warning }) + } + + await qc.invalidateQueries({ queryKey: ['kanban', 'board'] }) + onClose() + } catch (err) { + setError(errText(err)) + setBusy(false) + } + } + + return ( + !open && onClose()} open={Boolean(target)}> + {/* `overflow-visible`: DialogContent publishes ITSELF as the portal + container for popovers opened inside it (dialog-portal-context), and + its default `overflow-y-auto` then crops them at the dialog's edge — + the model menu below is born inside that scroll box. This dialog + already owns a scroller on its body div, so the shell's clip is + redundant here and dropping it is safe. The general fix to + DialogContent is in flight as #75600; when that lands this override + becomes a no-op and can go. */} + + + {target ? k.newTaskIn(columnLabel(k, target)) : k.newTask} + +
+ setTitle(event.target.value)} + onKeyDown={event => { + if (event.key === 'Enter') { + event.preventDefault() + void submit() + } + }} + placeholder={isTriage ? k.titlePlaceholderTriage : k.titlePlaceholder} + value={title} + /> +