diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index ed89185acf..7e47b1db69 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -107,7 +107,9 @@ jobs: version: "0.9.28" - name: Set up Python 3.11 (for docker tests) - run: uv python install 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 - name: Install Python dependencies (for docker tests) # ``dev`` extra pulls in pytest, pytest-asyncio — diff --git a/.github/workflows/e2e-desktop.yml b/.github/workflows/e2e-desktop.yml index b951c60ead..8d0c037a37 100644 --- a/.github/workflows/e2e-desktop.yml +++ b/.github/workflows/e2e-desktop.yml @@ -66,8 +66,12 @@ jobs: cache-dependency-glob: | pyproject.toml uv.lock + - name: Set up Python 3.11 - run: uv python install 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 + - name: Install Python dependencies uses: ./.github/actions/retry with: diff --git a/.github/workflows/install-e2e-run.yml b/.github/workflows/install-e2e-run.yml new file mode 100644 index 0000000000..354ff6bd47 --- /dev/null +++ b/.github/workflows/install-e2e-run.yml @@ -0,0 +1,122 @@ +name: Install & Update E2E (reusable) + +# Runs ONE update route against ONE starting commit, in the dev sandbox, with a +# real install (uv, a managed Python, Node, the venv) behind it. +# +# Reusable so callers can fan out over the combinations that matter -- update +# from the tip vs. from an older release, `hermes update` vs. re-running the +# installer -- without duplicating the runner setup. Each leg is independent: +# its own sandbox, its own install, nothing rewound or shared. +# +# Call it: +# +# jobs: +# tip: +# uses: ./.github/workflows/install-e2e-run.yml +# with: +# route: update +# install-ref: refs/heads/main + +on: + workflow_call: + inputs: + route: + description: 'Update path to exercise: update (hermes update) or installer (re-run install.sh).' + required: true + type: string + install-ref: + description: 'What to install before updating: a branch, a tag (v2026.7.7), or a SHA reachable from main.' + required: false + type: string + default: refs/heads/main + runner: + description: 'Runner label.' + required: false + type: string + default: ubuntu-latest + timeout-minutes: + description: 'Job timeout. A cold run installs real toolchains twice.' + required: false + type: number + default: 45 + +permissions: + contents: read + +jobs: + e2e: + name: ${{ inputs.route }} from ${{ inputs.install-ref }} + runs-on: ${{ inputs.runner }} + timeout-minutes: ${{ inputs.timeout-minutes }} + + steps: + # Full history: the sandbox fetches the starting commit and the test + # compares against this commit, so a shallow clone is not enough. + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + fetch-depth: 0 + + # bubblewrap + slirp4netns are what the sandbox is built on; util-linux + # supplies the `unshare` that builds the multi-uid userns for the + # user-level (non-root) install. + - name: Install sandbox dependencies + run: | + set -euo pipefail + sudo apt-get update -qq + sudo apt-get install -y -qq bubblewrap slirp4netns uidmap util-linux + + # Ubuntu 24.04 restricts unprivileged user namespaces through AppArmor, + # which is exactly what bwrap needs. Report the state before touching it + # so a future runner-image change is visible in the log rather than + # silently altering what this job proves. + - name: Permit unprivileged user namespaces + run: | + set -euo pipefail + echo "--- kernel userns settings (before)" + sysctl kernel.unprivileged_userns_clone 2>/dev/null || echo " (sysctl absent)" + sysctl kernel.apparmor_restrict_unprivileged_userns 2>/dev/null || echo " (sysctl absent)" + if sysctl -n kernel.apparmor_restrict_unprivileged_userns >/dev/null 2>&1; then + sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0 + fi + echo "--- subuid/subgid for $(id -un)" + grep "^$(id -un):" /etc/subuid /etc/subgid || echo " (none — sandbox will say so)" + + - name: Run install + update E2E + run: | + set -euo pipefail + tests/install/install-update-e2e.sh \ + --route '${{ inputs.route }}' \ + --install-ref '${{ inputs.install-ref }}' + env: + # Outside the workspace on purpose: the script creates this directory + # up front, and an untracked dir inside the repo makes the worktree + # dirty -- which dev-sandbox reacts to by snapshotting the working + # copy into a fresh fake-main commit on every invocation, moving the + # update target mid-run. + HERMES_E2E_LOG_DIR: ${{ runner.temp }}/e2e-logs + + # Artifact names cannot contain '/', and install-ref may be a full ref + # like refs/heads/main. GitHub Actions expressions have no string-replace + # function, so build the safe name here. Runs even on failure -- that is + # exactly when the logs are wanted. + - name: Build artifact name + if: always() + id: artifact + run: | + set -euo pipefail + safe_ref='${{ inputs.install-ref }}' + safe_ref="${safe_ref//\//-}" + echo "name=install-e2e-${{ inputs.route }}-${safe_ref}" >> "$GITHUB_OUTPUT" + + # The installer's own transcripts say far more than the assertion that + # tripped when a real install breaks. + - name: Upload installer logs + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + # Unique per leg: a matrix over releases runs this workflow several + # times per route, and same-named artifacts collide. + name: ${{ steps.artifact.outputs.name }}-${{ github.sha }} + path: ${{ runner.temp }}/e2e-logs + retention-days: 14 + if-no-files-found: ignore diff --git a/.github/workflows/install-e2e.yml b/.github/workflows/install-e2e.yml new file mode 100644 index 0000000000..03c9a1d0b8 --- /dev/null +++ b/.github/workflows/install-e2e.yml @@ -0,0 +1,110 @@ +name: Install & Update E2E + +# Can a user on a released version get to this commit? +# +# For each release we sample, a leg installs that release through the real +# `curl | install.sh` one-liner (uv, a managed Python, Node, the venv) inside +# scripts/dev-sandbox.sh, then applies one update route and requires the +# checkout to land on this commit with a working `hermes`. +# +# The starting versions are chosen at runtime from the repo's release tags +# (scripts/sandbox/pick-release-tags.sh): newest, oldest, and a spread between. +# A hardcoded list would stop covering the newest release the day after it +# ships, and would pin an "oldest" that nobody still runs. +# +# Triggers: +# * every 12 hours, so upstream drift (a new uv, a Node bump, a PyPI change) +# surfaces on a schedule rather than in someone's review cycle; +# * when a release tag is created -- the moment the set of versions users can +# update FROM changes, and the moment a broken updater would strand them; +# * manually, where you can pick the route and how many releases to sample. +# +# Deliberately NOT on pull_request: a leg takes ~11 minutes of real toolchain +# installation, and the matrix multiplies that. Updating is release-shaped work, +# so it is gated on releases and the clock instead. + +on: + workflow_dispatch: + inputs: + route: + description: 'Which update route to exercise.' + required: false + type: choice + default: both + options: [both, update, installer] + tag-count: + description: 'How many release tags to sample (newest, oldest, and a spread between).' + required: false + type: string + default: '5' + schedule: + # Every 12 hours, off the hour to avoid the top-of-hour runner crunch. + - cron: '20 7,19 * * *' + push: + tags: + # Release tags only: the repo also carries backup/* and one-off tags. + - 'v[0-9]+.[0-9]+.[0-9]+' + - 'v[0-9]+.[0-9]+.[0-9]+.[0-9]+' + +permissions: + contents: read + +concurrency: + group: install-e2e-${{ github.ref }} + cancel-in-progress: true + +jobs: + # Which released versions do we test updating FROM? Resolved once and shared + # by both route matrices, so the two routes cover the same set. + pick-releases: + name: Pick release tags + runs-on: ubuntu-latest + timeout-minutes: 5 + outputs: + tags: ${{ steps.pick.outputs.tags }} + steps: + # This job only reads tag names and runs one script, so take the cheap + # checkout: no blobs (filter), no other files (sparse), but DO fetch tags + # -- they are the whole input, and the default shallow checkout has none. + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + filter: blob:none + fetch-tags: true + sparse-checkout: scripts/sandbox/pick-release-tags.sh + sparse-checkout-cone-mode: false + - id: pick + run: | + set -euo pipefail + tags="$(scripts/sandbox/pick-release-tags.sh --count '${{ inputs.tag-count || 5 }}')" + echo "Testing updates from: $tags" + echo "tags=$tags" >> "$GITHUB_OUTPUT" + + # `hermes update` -- the route most users take. + update: + if: github.event_name != 'workflow_dispatch' || inputs.route != 'installer' + needs: pick-releases + strategy: + # One release breaking is worth knowing about even if another already + # failed, so let every leg report. + fail-fast: false + matrix: + install-ref: ${{ fromJSON(needs.pick-releases.outputs.tags) }} + uses: ./.github/workflows/install-e2e-run.yml + with: + route: update + install-ref: ${{ matrix.install-ref }} + + # Re-running the curl one-liner over an existing checkout: autostash + pull + # rather than the updater's own git handling. + installer: + if: github.event_name != 'workflow_dispatch' || inputs.route != 'update' + needs: pick-releases + strategy: + fail-fast: false + max-parallel: 3 + matrix: + install-ref: ${{ fromJSON(needs.pick-releases.outputs.tags) }} + uses: ./.github/workflows/install-e2e-run.yml + with: + route: installer + install-ref: ${{ matrix.install-ref }} diff --git a/.github/workflows/lint.yml b/.github/workflows/lint.yml index 3ae120f71f..7a49a62203 100644 --- a/.github/workflows/lint.yml +++ b/.github/workflows/lint.yml @@ -163,10 +163,18 @@ jobs: - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - name: Set up Python - uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v5 + - name: Install uv + uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0 with: - python-version: "3.11" + # Pinned: unpinned setup-uv fetches a 'latest' manifest from + # raw.githubusercontent.com every job; transient fetch failures + # fail the job (2026-07-28 incident). Keep in sync with tests.yml. + version: "0.9.28" + + - name: Set up Python 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 - name: Run footgun checker run: python scripts/check-windows-footguns.py --all diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 353e86439e..3888e8d44f 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -90,7 +90,9 @@ jobs: uv.lock - name: Set up Python 3.11 - run: uv python install 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 - name: Install dependencies # `uv sync --locked` installs the exact pinned set from uv.lock (and diff --git a/.gitignore b/.gitignore index 55a2ab8aad..283c8a9164 100644 --- a/.gitignore +++ b/.gitignore @@ -145,6 +145,9 @@ docs/superpowers/* # Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) .hermes-sandbox/ +# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is +# the route name, so each route gets its own tree and two can run at once. +.hermes-sandbox-e2e*/ # Interrupted-update breadcrumb + recovery lock written next to the shared venv # by `hermes update` / launch-time self-heal. Runtime state, never a code change diff --git a/acp_adapter/permissions.py b/acp_adapter/permissions.py index 5f29a96725..b10b2a169e 100644 --- a/acp_adapter/permissions.py +++ b/acp_adapter/permissions.py @@ -158,9 +158,16 @@ def make_approval_callback( try: response = future.result(timeout=timeout) - except (FutureTimeout, Exception) as exc: + except FutureTimeout: future.cancel() - logger.warning("Permission request timed out or failed: %s", exc) + logger.warning("Permission request timed out after %ss", timeout) + # Distinct from an explicit deny: the client never answered. + # tools.approval callers report this as "timed out without user + # response" instead of a user denial. + return "timeout" + except Exception as exc: + future.cancel() + logger.warning("Permission request failed: %s", exc) return "deny" if response is None: diff --git a/agent/agent_init.py b/agent/agent_init.py index db8f026f63..649d5338a9 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -209,7 +209,18 @@ def _context_route_mismatch( if active_route: configured_routes = _provider_default_routes(configured_provider) - return not configured_routes or active_route not in configured_routes + if configured_routes: + return active_route not in configured_routes + # Named/custom providers have no catalog default routes. An empty + # configured URL with a matching provider identity is still the same + # route — agent_init fills base_url from custom_providers before this + # check, but gateway display/hygiene paths historically compared the + # raw empty model.base_url and falsely dropped model.context_length, + # falling through to family defaults (e.g. qwen → 131072) on Discord + # session-reset banners while /status still showed the config pin. + if active_provider and configured_provider == active_provider: + return False + return True return bool( configured_provider and active_provider @@ -1580,6 +1591,17 @@ def init_agent( "reasoning_config": reasoning_config, "max_tokens": max_tokens, } + # Persist a process-scoped --yolo launch into the session row so a later + # `hermes --resume ` can restore the bypass (CLI resume paths read + # model_config.yolo_mode back via SessionDB.session_yolo_enabled). + # Session-scoped /yolo toggles persist separately through + # SessionDB.set_session_yolo at toggle time. + try: + from tools.approval import _YOLO_MODE_FROZEN + if _YOLO_MODE_FROZEN: + agent._session_init_model_config["yolo_mode"] = True + except Exception: + pass # In-memory todo list for task planning (one per agent/session) from tools.todo_tool import TodoStore diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 579314018c..8ddf1b95fd 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -36,7 +36,7 @@ from hermes_cli.timeouts import get_provider_request_timeout from agent.prompt_builder import format_steer_marker from agent.tool_dispatch_helpers import _trajectory_normalize_msg, make_tool_result_message from agent.trajectory import convert_scratchpad_to_think -from agent.credential_pool import STATUS_EXHAUSTED +from agent.credential_pool import STATUS_EXHAUSTED, credential_pool_matches_provider from agent.error_classifier import FailoverReason from agent.turn_context import drop_stale_api_content from utils import base_url_host_matches, base_url_hostname, env_var_enabled, atomic_json_write @@ -52,6 +52,45 @@ logger = logging.getLogger(__name__) _MAX_AUTH_REFRESH_ATTEMPTS = 2 +_REASONING_TAG_NAMES = ("think", "thinking", "reasoning", "REASONING_SCRATCHPAD", "thought") +_TOOL_CALL_TAG_NAMES = ("tool_call", "tool_calls", "tool_result", "function_call", "function_calls") + +_REASONING_BLOCK_PATTERNS = tuple( + re.compile(rf"<{name}>.*?", re.DOTALL | re.IGNORECASE) + for name in _REASONING_TAG_NAMES +) + +_TOOL_CALL_BLOCK_PATTERNS = tuple( + re.compile(rf"<{name}\b[^>]*>.*?", re.DOTALL | re.IGNORECASE) + for name in _TOOL_CALL_TAG_NAMES +) + +# Named blocks — see strip_think_blocks step 1c for the +# full rationale (sentence-boundary lookbehind + tempered-dot body so a plain +# prose mention of "function" is never eaten). +_NAMED_FUNCTION_BLOCK_PATTERN = re.compile( + r'(?:(?<=^)|(?<=[\n\r.!?:]))[ \t]*' + r']*\bname\s*=[^>]*>' + r'(?:(?:(?!).)*)', + re.DOTALL | re.IGNORECASE, +) + +_UNTERMINATED_REASONING_BLOCK_PATTERN = re.compile( + rf'(?:^|\n)[ \t]*<(?:{"|".join(_REASONING_TAG_NAMES)})\b[^>]*>.*$', + re.DOTALL | re.IGNORECASE, +) + +_ORPHAN_REASONING_TAG_PATTERN = re.compile( + rf'\s*', + re.IGNORECASE, +) + +_STRAY_TOOL_CALL_CLOSER_PATTERN = re.compile( + rf'\s*', + re.IGNORECASE, +) + + def _ra(): """Lazy ``run_agent`` reference for test-patch routing.""" import run_agent @@ -826,62 +865,31 @@ def strip_think_blocks(agent, content: str) -> str: # 1. Closed tag pairs — case-insensitive for all variants so # mixed-case tags (, ) don't slip through to # the unterminated-tag pass and take trailing content with them. - content = re.sub(r'.*?', '', content, flags=re.DOTALL | re.IGNORECASE) - content = re.sub(r'.*?', '', content, flags=re.DOTALL | re.IGNORECASE) - content = re.sub(r'.*?', '', content, flags=re.DOTALL | re.IGNORECASE) - content = re.sub(r'.*?', '', content, flags=re.DOTALL | re.IGNORECASE) - content = re.sub(r'.*?', '', content, flags=re.DOTALL | re.IGNORECASE) + for _pattern in _REASONING_BLOCK_PATTERNS: + content = _pattern.sub('', content) # 1b. Tool-call XML blocks (openclaw/openclaw#67318). Handle the # generic tag names first — they have no attribute gating since # a literal in prose is already vanishingly rare. - for _tc_name in ("tool_call", "tool_calls", "tool_result", - "function_call", "function_calls"): - content = re.sub( - rf'<{_tc_name}\b[^>]*>.*?', - '', - content, - flags=re.DOTALL | re.IGNORECASE, - ) + for _pattern in _TOOL_CALL_BLOCK_PATTERNS: + content = _pattern.sub('', content) # 1c. ... — Gemma-style standalone # tool call. Only strip when the tag sits at a block boundary # (start of text, after a newline, or after sentence-ending # punctuation) AND carries a name="..." attribute. This keeps # prose mentions like "Use to declare" safe. - content = re.sub( - r'(?:(?<=^)|(?<=[\n\r.!?:]))[ \t]*' - r']*\bname\s*=[^>]*>' - r'(?:(?:(?!).)*)', - '', - content, - flags=re.DOTALL | re.IGNORECASE, - ) + content = _NAMED_FUNCTION_BLOCK_PATTERN.sub('', content) # 2. Unterminated reasoning block — open tag at a block boundary # (start of text, or after a newline) with no matching close. # Strip from the tag to end of string. Fixes #8878 / #9568 # (MiniMax M2.7 leaking raw reasoning into assistant content). - content = re.sub( - r'(?:^|\n)[ \t]*<(?:think|thinking|reasoning|thought|REASONING_SCRATCHPAD)\b[^>]*>.*$', - '', - content, - flags=re.DOTALL | re.IGNORECASE, - ) + content = _UNTERMINATED_REASONING_BLOCK_PATTERN.sub('', content) # 3. Stray orphan open/close tags that slipped through. - content = re.sub( - r'\s*', - '', - content, - flags=re.IGNORECASE, - ) + content = _ORPHAN_REASONING_TAG_PATTERN.sub('', content) # 3b. Stray tool-call closers. (We do NOT strip bare or # unterminated because a truncated tail # during streaming may still be valuable to the user; matches # OpenClaw's intentional asymmetry.) - content = re.sub( - r'\s*', - '', - content, - flags=re.IGNORECASE, - ) + content = _STRAY_TOOL_CALL_CLOSER_PATTERN.sub('', content) return content @@ -1012,6 +1020,13 @@ def recover_with_credential_pool( } if _credential_id: kwargs["credential_id"] = _credential_id + # Hand the pool the classified semantics, not just the status. A + # billing 403 (OpenRouter "key limit exceeded", xAI spending limit) + # and an edge-throttle 403 are the same number but need opposite + # cooldowns — the pool can only tell them apart if we say which. + # ``effective_reason`` is resolved below; this closure runs after. + if effective_reason is not None: + kwargs["failure_reason"] = effective_reason.value return pool.mark_exhausted_and_rotate(**kwargs) effective_reason = classified_reason @@ -1456,6 +1471,64 @@ def restore_primary_runtime(agent) -> bool: if getattr(agent, "_rate_limited_until", 0) > time.monotonic(): return False # primary still in rate-limit cooldown, stay on fallback + # ── Reset-aware gate ── + # The 60s ``_rate_limited_until`` cooldown covers transient rate limits, + # but subscription-style providers (Claude Pro/Max 5-hour windows, ChatGPT + # weekly limits) report reset times hours or days away. The credential + # pool already stores those timestamps (``last_error_reset_at``); until + # the earliest one elapses, every restore attempt is a *guaranteed* + # failure that costs two prompt-cache invalidations per turn (switch to + # primary, fail, switch back to fallback) and re-marshals the full + # context each way. Skip the restore while the pool says nobody can + # serve, and come back the moment the reset time passes. + # + # Fail-open by design: any error (unreadable auth store, legacy pool + # adapter without ``next_available_at``) falls through to the existing + # every-turn retry. A pool with no reset info returns ``None`` and also + # falls through — this gate only ever *adds* skips for provably + # limited windows, so recovery can never be later than it is today. + # + # When the attached pool belongs to the fallback provider (cross-provider + # fallback rebinds it), the primary pool is loaded here and handed to the + # pool-rebind block below via ``prefetched_primary_pool`` so the load + # happens at most once per restore. + prefetched_primary_pool = None + try: + primary_provider = str( + (agent._primary_runtime or {}).get("provider") or "" + ).strip().lower() + pool = getattr(agent, "_credential_pool", None) + if not credential_pool_matches_provider( + pool, + primary_provider, + base_url=str((agent._primary_runtime or {}).get("base_url") or ""), + ): + from agent.credential_pool import load_pool + + prefetched_primary_pool = ( + load_pool(primary_provider) if primary_provider else None + ) + pool = prefetched_primary_pool + next_at = getattr(pool, "next_available_at", lambda: None)() + if next_at is not None and next_at > time.time(): + if not getattr(agent, "_restore_wait_logged", False): + agent._restore_wait_logged = True + logger.info( + "Primary %s rate-limited until %s; staying on fallback " + "%s/%s until the reset elapses", + primary_provider or "?", + datetime.fromtimestamp(next_at).isoformat(timespec="seconds"), + agent.provider, + agent.model, + ) + return False + except Exception: + logger.debug( + "Reset-aware restore gate failed; falling back to per-turn retry", + exc_info=True, + ) + agent._restore_wait_logged = False + rt = agent._primary_runtime try: # ── Core runtime state ── @@ -1549,9 +1622,14 @@ def restore_primary_runtime(agent) -> bool: agent._credential_pool = None agent._credential_pool_entry_id = None try: - from agent.credential_pool import load_pool + if prefetched_primary_pool is not None: + # Reuse the pool the reset-aware gate already loaded for + # this restore — avoids a second disk read of auth.json. + agent._credential_pool = prefetched_primary_pool + else: + from agent.credential_pool import load_pool - agent._credential_pool = load_pool(primary_provider) + agent._credential_pool = load_pool(primary_provider) except Exception as exc: logger.warning( "Restore could not reload primary credential pool for %s: %s", @@ -1631,6 +1709,7 @@ def restore_primary_runtime(agent) -> bool: # ── Reset fallback chain for the new turn ── agent._fallback_activated = False agent._fallback_index = 0 + agent._rate_limit_backoff_count = 0 # reset exponential backoff counter # Reset the stale-call circuit breaker (#58962): the streak measured # the FALLBACK provider we're leaving; the restored primary deserves @@ -2003,12 +2082,12 @@ def anthropic_prompt_cache_policy( gateway implements the Anthropic cache_control contract (MiniMax, Zhipu GLM, LiteLLM's Anthropic proxy mode all do). - Qwen models on OpenCode and direct Alibaba (DashScope), plus DeepSeek - models on OpenCode, also honour Anthropic-style ``cache_control`` markers - on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi #3393 - documented this for opencode-go Qwen; #24617 reports the same gateway - contract for DeepSeek. Without markers these providers serve zero cache - hits, re-billing the full prompt on every turn. + Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct + Alibaba (DashScope) also honour Anthropic-style ``cache_control`` + markers on OpenAI-wire chat completions. Upstream pi-mono #3392 / + pi #3393 documented this for opencode-go Qwen. Without markers + these providers serve zero cache hits, re-billing the full prompt + on every turn. If the operator has set ``prompt_caching.cache_ttl`` to a falsy value (``false``, ``null``, ``"off"``, etc.) in config.yaml, prompt caching @@ -2135,22 +2214,21 @@ def anthropic_prompt_cache_policy( if is_minimax_provider or is_minimax_host: return True, True - # Qwen on OpenCode (Zen/Go) and native DashScope, plus DeepSeek on - # OpenCode only: OpenAI-wire transports that accept Anthropic-style - # cache_control markers and reward them with real cache hits. Keep direct - # Alibaba specific to Qwen; its catalog does not establish the same - # contract for DeepSeek. + # Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire + # transport that accepts Anthropic-style cache_control markers and + # rewards them with real cache hits. Without this branch + # qwen3.6-plus on opencode-go reports 0% cached tokens and burns + # through the subscription on every turn. + # + # NOTE: DeepSeek models on OpenCode are intentionally excluded. + # OpenCode Zen's relay rejects the Anthropic-style content block + # format that cache markers produce (content becomes a block array + # instead of a plain string), causing HTTP 400 (#77217). model_is_qwen = "qwen" in model_lower - model_is_deepseek = "deepseek" in model_lower - provider_is_opencode = provider_lower in { - "opencode", "opencode-zen", "opencode-go", - } provider_is_alibaba_family = provider_lower in { "opencode", "opencode-zen", "opencode-go", "alibaba", } - if (provider_is_alibaba_family and model_is_qwen) or ( - provider_is_opencode and model_is_deepseek - ): + if provider_is_alibaba_family and model_is_qwen: # Envelope layout (native_anthropic=False): markers on inner # content parts, not top-level tool messages. Matches # pi-mono's "alibaba" cacheControlFormat. diff --git a/agent/anthropic_adapter.py b/agent/anthropic_adapter.py index 7457119921..70a9599323 100644 --- a/agent/anthropic_adapter.py +++ b/agent/anthropic_adapter.py @@ -1334,7 +1334,7 @@ def _resolve_anthropic_pool_token() -> Optional[str]: # to auth.json or trigger a network refresh from a bare resolve. select() # is deliberately NOT used — it runs clear_expired=True, refresh=True, # which would violate this read-only contract. - entries = pool._available_entries(clear_expired=False, refresh=False) + entries, _pending = pool._available_entries(clear_expired=False, refresh=False) except Exception: logger.debug("Failed to read Anthropic credential_pool", exc_info=True) return None @@ -1360,19 +1360,27 @@ def resolve_anthropic_token() -> Optional[str]: Priority: 1. ANTHROPIC_TOKEN env var (OAuth/setup token saved by Hermes) 2. CLAUDE_CODE_OAUTH_TOKEN env var - 3. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json) + 3. ANTHROPIC_API_KEY env var (explicit regular API key) + 4. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json) — with automatic refresh if expired and a refresh token is available - 4. Anthropic credential_pool OAuth entry (~/.hermes/auth.json) - 5. ANTHROPIC_API_KEY env var (regular API key, or legacy fallback) + 5. Anthropic credential_pool OAuth entry (~/.hermes/auth.json) Returns the token string or None. """ - creds = read_claude_code_credentials() + creds: Optional[Dict[str, Any]] = None + creds_loaded = False + + def _read_creds() -> Optional[Dict[str, Any]]: + nonlocal creds, creds_loaded + if not creds_loaded: + creds = read_claude_code_credentials() + creds_loaded = True + return creds # 1. Hermes-managed OAuth/setup token env var token = _getenv("ANTHROPIC_TOKEN").strip() if token: - preferred = _prefer_refreshable_claude_code_token(token, creds) + preferred = _prefer_refreshable_claude_code_token(token, _read_creds()) if preferred: return preferred return token @@ -1380,27 +1388,27 @@ def resolve_anthropic_token() -> Optional[str]: # 2. CLAUDE_CODE_OAUTH_TOKEN (used by Claude Code for setup-tokens) cc_token = _getenv("CLAUDE_CODE_OAUTH_TOKEN").strip() if cc_token: - preferred = _prefer_refreshable_claude_code_token(cc_token, creds) + preferred = _prefer_refreshable_claude_code_token(cc_token, _read_creds()) if preferred: return preferred return cc_token - # 3. Claude Code credential file - resolved_claude_token = _resolve_claude_code_token_from_credentials(creds) - if resolved_claude_token: - return resolved_claude_token - - # 4. Hermes credential_pool OAuth entry. - resolved_pool_token = _resolve_anthropic_pool_token() - if resolved_pool_token: - return resolved_pool_token - - # 5. Regular API key, or a legacy OAuth token saved in ANTHROPIC_API_KEY. - # This remains as a compatibility fallback for pre-migration Hermes configs. + # 3. Regular API key. An explicit user-configured key must not be shadowed + # by auto-discovered Claude Code or credential-pool OAuth credentials. api_key = _getenv("ANTHROPIC_API_KEY").strip() if api_key: return api_key + # 4. Claude Code credential file + resolved_claude_token = _resolve_claude_code_token_from_credentials(_read_creds()) + if resolved_claude_token: + return resolved_claude_token + + # 5. Hermes credential_pool OAuth entry. + resolved_pool_token = _resolve_anthropic_pool_token() + if resolved_pool_token: + return resolved_pool_token + return None @@ -2311,13 +2319,14 @@ def _convert_user_message(content: Any) -> Dict[str, Any]: """Validate and convert a user message to anthropic format.""" if isinstance(content, list): converted_blocks = _convert_content_to_anthropic(content) - if not converted_blocks or all( - (b.get("text") or "").strip() == "" - for b in converted_blocks - if isinstance(b, dict) and b.get("type") == "text" - ): - converted_blocks = [{"type": "text", "text": "(empty message)"}] - return {"role": "user", "content": converted_blocks} + kept_blocks = _fix_blank_text_blocks_in_list( + converted_blocks, + placeholder_text="(empty message)", + msg_index=-1, + role="user", + location="_convert_user_message", + ) + return {"role": "user", "content": kept_blocks} else: if not content or (isinstance(content, str) and not content.strip()): content = "(empty message)" @@ -2620,9 +2629,114 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None: Mirror the Bedrock Converse adapter, which unconditionally prepends a minimal user turn when the first message is not user (convert_messages_to_converse). + + The inserted text block must be non-whitespace: Anthropic separately + rejects any text content block whose text is empty or whitespace-only + ("text content blocks must contain non-whitespace text"), so a single + space here traded the "leading assistant turn" 400 for that one (#69512 + class). Uses the same placeholder as every other synthesized filler + block in this module for consistency. """ if result and result[0].get("role") != "user": - result.insert(0, {"role": "user", "content": [{"type": "text", "text": " "}]}) + result.insert( + 0, {"role": "user", "content": [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]} + ) + + +def _fix_blank_text_blocks_in_list( + blocks: List[Any], + *, + placeholder_text: str, + msg_index: int, + role: Any, + location: str, +) -> List[Any]: + """Drop blank/whitespace-only text blocks from ``blocks``, in place logic. + + Non-text blocks (tool_use, tool_result, image, document, thinking, …) + and the relative order of everything else are left untouched. A + cache_control marker riding on a dropped block is relocated onto the + last surviving text/tool_use block so a breakpoint is never silently + lost. If nothing survives, a single non-blank placeholder text block + takes the dropped blocks' place (carrying the relocated cache_control, + if any) so the message never has empty content. + + Returns a new list; does not mutate ``blocks``. + """ + kept: List[Any] = [] + relocated_cache_control = None + for block_index, blk in enumerate(blocks): + if ( + isinstance(blk, dict) + and blk.get("type") == "text" + and not (isinstance(blk.get("text"), str) and blk["text"].strip()) + ): + if isinstance(blk.get("cache_control"), dict): + relocated_cache_control = blk["cache_control"] + logger.warning( + "Pre-call sanitizer: dropped blank text content block " + "(message_index=%d role=%s location=%s block_index=%d " + "block_type=text)", + msg_index, + role, + location, + block_index, + ) + continue + kept.append(blk) + if not kept: + placeholder: Dict[str, Any] = {"type": "text", "text": placeholder_text} + if relocated_cache_control is not None: + placeholder["cache_control"] = relocated_cache_control + kept.append(placeholder) + elif relocated_cache_control is not None: + _apply_assistant_cache_control_to_last_cacheable_block(kept, relocated_cache_control) + return kept + + +def _scrub_blank_text_blocks(result: List[Dict[str, Any]]) -> None: + """Final provider-boundary guard against blank Anthropic text blocks. + + Anthropic rejects any text content block whose ``text`` is empty or + whitespace-only with HTTP 400 ("text content blocks must contain + non-whitespace text"). ``_convert_assistant_message``, + ``_convert_user_message`` and ``_ensure_leading_user_turn`` already + avoid emitting these for the paths that build them, but this pass runs + last — after every other transform in ``convert_messages_to_anthropic`` + — so a blank block from any current or future producer (including one + nested inside a ``tool_result``'s own content list) never reaches the + wire. Diagnostics are structural only: message index, role, content + location, block index/type. Never logs message text, tool arguments, + tokens, or credentials. Mutates ``result`` in place. + """ + for msg_index, msg in enumerate(result): + if not isinstance(msg, dict): + continue + role = msg.get("role") + content = msg.get("content") + if not isinstance(content, list) or not content: + continue + placeholder_text = _EMPTY_TEXT_PLACEHOLDER if role == "assistant" else "(empty message)" + new_content = _fix_blank_text_blocks_in_list( + content, + placeholder_text=placeholder_text, + msg_index=msg_index, + role=role, + location="content", + ) + for blk in new_content: + if not isinstance(blk, dict) or blk.get("type") != "tool_result": + continue + inner = blk.get("content") + if isinstance(inner, list) and inner: + blk["content"] = _fix_blank_text_blocks_in_list( + inner, + placeholder_text="(no output)", + msg_index=msg_index, + role=role, + location="tool_result", + ) + msg["content"] = new_content def convert_messages_to_anthropic( @@ -2686,6 +2800,7 @@ def convert_messages_to_anthropic( _ensure_leading_user_turn(result) _manage_thinking_signatures(result, base_url, model) _evict_old_screenshots(result) + _scrub_blank_text_blocks(result) return system, result diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 44d7d69683..d66d5d81ac 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -7604,6 +7604,77 @@ def _get_task_extra_body(task: str) -> Dict[str, Any]: return result +# --------------------------------------------------------------------------- +# Per-task concurrency limiting (#23324) +# --------------------------------------------------------------------------- +# Background auxiliary work (title generation, context compression, etc.) can +# spawn unbounded concurrent LLM calls when many sessions are active. During +# provider incidents each call also retries / fans out across the fallback +# chain, multiplying request volume on already-degraded endpoints. A per-task +# semaphore caps in-flight calls so retry amplification stays bounded. + +_aux_sync_semaphores: Dict[str, Tuple[int, threading.BoundedSemaphore]] = {} +_aux_async_semaphores: Dict[Tuple[str, int], Tuple[int, Any]] = {} +_aux_sem_lock = threading.Lock() + + +def _get_task_max_concurrency(task: Optional[str]) -> Optional[int]: + """Return ``auxiliary..max_concurrency`` as a positive int, or None.""" + if not task or task == "vision": + # Vision already uses this key for its encode/resize CPU worker pool; + # its LLM calls deliberately remain concurrent. + return None + raw = _get_auxiliary_task_config(task).get("max_concurrency") + if raw is None: + return None + try: + value = int(raw) + except (TypeError, ValueError): + return None + return value if value > 0 else None + + +def _acquire_sync_aux_semaphore(task: Optional[str]) -> Optional[threading.BoundedSemaphore]: + """Get a per-task sync semaphore, rebuilding it after a config change.""" + limit = _get_task_max_concurrency(task) + if limit is None: + return None + with _aux_sem_lock: + entry = _aux_sync_semaphores.get(task) + if entry is None or entry[0] != limit: + semaphore = threading.BoundedSemaphore(limit) + _aux_sync_semaphores[task] = (limit, semaphore) + return semaphore + return entry[1] + + +def _acquire_async_aux_semaphore(task: Optional[str]): + """Get a per-task, per-event-loop async semaphore after config lookup.""" + limit = _get_task_max_concurrency(task) + if limit is None: + return None + import asyncio + try: + loop = asyncio.get_running_loop() + except RuntimeError: + return None + key = (task, id(loop)) + with _aux_sem_lock: + entry = _aux_async_semaphores.get(key) + if entry is None or entry[0] != limit: + semaphore = asyncio.Semaphore(limit) + _aux_async_semaphores[key] = (limit, semaphore) + return semaphore + return entry[1] + + +def _reset_aux_semaphores() -> None: + """Drop cached semaphores (test helper).""" + with _aux_sem_lock: + _aux_sync_semaphores.clear() + _aux_async_semaphores.clear() + + # --------------------------------------------------------------------------- # Anthropic-compatible endpoint detection + image block conversion # --------------------------------------------------------------------------- @@ -8476,6 +8547,75 @@ def call_llm( api_mode: str = None, stream: bool = False, stream_options: dict = None, +) -> Any: + """Run an auxiliary LLM request, applying the configured task limit.""" + semaphore = _acquire_sync_aux_semaphore(task) + if semaphore is not None: + semaphore.acquire() + try: + response = _call_llm_impl( + task=task, + provider=provider, + model=model, + base_url=base_url, + api_key=api_key, + main_runtime=main_runtime, + messages=messages, + temperature=temperature, + max_tokens=max_tokens, + tools=tools, + timeout=timeout, + extra_body=extra_body, + reasoning_config=reasoning_config, + extra_headers=extra_headers, + api_mode=api_mode, + stream=stream, + stream_options=stream_options, + ) + if stream and semaphore is not None: + stream_semaphore = semaphore + semaphore = None + return _release_sync_semaphore_after_stream(response, stream_semaphore) + return response + finally: + if semaphore is not None: + semaphore.release() + + +def _release_sync_semaphore_after_stream( + stream: Any, semaphore: threading.BoundedSemaphore, +): + """Release a permit only after a streaming response is consumed or closed.""" + try: + yield from stream + finally: + try: + close = getattr(stream, "close", None) + if callable(close): + close() + finally: + semaphore.release() + + +def _call_llm_impl( + task: str = None, + *, + provider: str = None, + model: str = None, + base_url: str = None, + api_key: str = None, + main_runtime: Optional[Dict[str, Any]] = None, + messages: list, + temperature: Optional[float] = None, + max_tokens: int = None, + tools: list = None, + timeout: float = None, + extra_body: dict = None, + reasoning_config: Optional[dict] = None, + extra_headers: Optional[Dict[str, str]] = None, + api_mode: str = None, + stream: bool = False, + stream_options: dict = None, ) -> Any: """Centralized synchronous LLM call. @@ -9239,6 +9379,47 @@ async def async_call_llm( timeout: float = None, extra_body: dict = None, reasoning_config: Optional[dict] = None, +) -> Any: + """Run an asynchronous auxiliary LLM request under the configured limit.""" + semaphore = _acquire_async_aux_semaphore(task) + if semaphore is not None: + await semaphore.acquire() + try: + return await _async_call_llm_impl( + task=task, + provider=provider, + model=model, + base_url=base_url, + api_key=api_key, + main_runtime=main_runtime, + messages=messages, + temperature=temperature, + max_tokens=max_tokens, + tools=tools, + timeout=timeout, + extra_body=extra_body, + reasoning_config=reasoning_config, + ) + finally: + if semaphore is not None: + semaphore.release() + + +async def _async_call_llm_impl( + task: str = None, + *, + provider: str = None, + model: str = None, + base_url: str = None, + api_key: str = None, + main_runtime: Optional[Dict[str, Any]] = None, + messages: list, + temperature: Optional[float] = None, + max_tokens: int = None, + tools: list = None, + timeout: float = None, + extra_body: dict = None, + reasoning_config: Optional[dict] = None, ) -> Any: """Centralized asynchronous LLM call. diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index e6e8ab7fdc..3b7d8b0361 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1712,7 +1712,20 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool current_provider = (getattr(agent, "provider", "") or "").strip().lower() primary_provider = ((agent._primary_runtime or {}).get("provider") or "").strip().lower() if (not fallback_already_active) or (primary_provider and current_provider == primary_provider): - agent._rate_limited_until = time.monotonic() + 60 + # Exponential backoff: keep upstream's 60s first-hit cooldown and + # escalate on CONSECUTIVE rate-limits: 60s → 2m → 4m → 8m → ... → + # 4h cap. The first 429 must NOT bench the primary for half an + # hour — fast primary restore is the common case; escalation only + # punishes providers that keep 429ing. + # Counter is reset by restore_primary_runtime on successful restore. + backoff_count = getattr(agent, "_rate_limit_backoff_count", 0) + agent._rate_limit_backoff_count = backoff_count + 1 + backoff_seconds = min(60 * (2 ** backoff_count), 14400) + agent._rate_limited_until = time.monotonic() + backoff_seconds + logging.info( + "Rate-limit backoff level %d: cooldown %d s (%.1f min, backoff#%d)", + backoff_count, backoff_seconds, backoff_seconds / 60, backoff_count + 1, + ) if agent._fallback_index >= len(agent._fallback_chain): # Chain exhausted. If we actually walked a non-empty chain and the # failure was NOT a rate-limit/billing event (those already armed diff --git a/agent/context_compressor.py b/agent/context_compressor.py index fbb7e6c5e8..59ae723d2b 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -6730,6 +6730,26 @@ This compaction should PRIORITISE preserving all information related to the focu _strip_persistence_markers(compressed) self._last_compression_made_progress = True + # A successful compaction just freed the largest allocation a long + # session ever drops (the compressed-away message dicts), which makes + # this the natural point to hand allocator pages back to the OS. + # #76905's trim lifecycle covers the gateway/TUI housekeeping loops but + # not the CLI compression path, so RSS keeps the pre-compaction + # high-water mark until exit. The helper is glibc-gated, config-gated + # and rate-limited, so this is a safe no-op elsewhere. (#70782) + try: + from hermes_cli.mem_trim import trim_memory + + trim_memory(reason="post-compression") + except Exception as exc: + # debug, not warning: sibling trim sites all log failures at + # debug, and compression must never fail because of a trim. + logger.debug( + "post-compression memory trim failed: %s: %s", + type(exc).__name__, + exc, + ) + # Batch compaction invalidates micro-compaction state: the batch # marker now holds MORE history than the in-memory rolling summary # (it summarized everything in the window, including exchanges micro diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 1884c06489..0515a31331 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -58,6 +58,11 @@ from agent.message_sanitization import ( _strip_images_from_messages, _strip_non_ascii, ) +# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py — kept local +# to avoid importing hermes_state at module load time (its module-level +# DEFAULT_DB_PATH = get_hermes_home() / "state.db" breaks tests that +# monkeypatch get_hermes_home to return a str). +_STALE_MARKER_RE = re.compile(r"^\[[A-Za-z_][A-Za-z0-9_.-]*\]$") from agent.model_metadata import ( MINIMUM_CONTEXT_LENGTH, _estimate_tools_tokens_rough, @@ -1286,6 +1291,14 @@ def run_conversation( agent._last_compression_attempt_recorded = False agent._last_compression_attempt_in_place = None + # Adopt any ~/.hermes/.env credential/base-url edits made since the last + # turn — a Settings save updates .env but not this worker's client, which + # was built at agent init (#67821). No-op when .env is unchanged. + try: + agent._try_refresh_env_client_credentials() + except Exception: + logger.debug("per-turn env credential refresh failed", exc_info=True) + # ── Per-turn setup (the prologue) ── # All once-per-turn setup — stdio guarding, retry-counter resets, user # message sanitization, todo/nudge hydration, system-prompt restore-or- @@ -4229,6 +4242,28 @@ def run_conversation( f" Check which providers support tools: https://openrouter.ai/models/{_model}" ) + # Actionable hint for a bare 404 on a provider whose catalogue + # uses ``vendor/model`` ids. A model id that lost its prefix + # (e.g. ``nemotron-…`` instead of ``nvidia/nemotron-…``) gets + # a content-free "404 page not found" from the provider that + # never names the model, so it reads like an outage or an auth + # failure. Name the real cause and the exact id to use (#78796). + if getattr(api_error, "status_code", None) == 404: + try: + from hermes_cli.model_normalize import suggest_prefixed_model_id + + _suggestion = suggest_prefixed_model_id(_provider, _model) + except Exception: + _suggestion = None + if _suggestion: + agent._buffer_vprint( + f" 💡 Model '{_model}' is not a valid id for provider {_provider} — " + f"it is missing its vendor prefix." + ) + agent._buffer_vprint( + f" Did you mean '{_suggestion}'? Re-pick it with `hermes model`." + ) + # Check for interrupt before deciding to retry if agent._interrupt_requested: # Preserve a pending redirect (mid-stream correction): the @@ -4360,9 +4395,11 @@ def run_conversation( compression_attempts += 1 if compression_attempts <= max_compression_attempts: original_len = len(messages) + # Option A (LCM issue 441): overhead-aware request size so recovery arms on + # the true request (msgs + tools + system), not the tool-blind message count. messages, active_system_prompt = agent._compress_context( messages, system_message, - approx_tokens=approx_tokens, + approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), task_id=effective_task_id, ) conversation_history = conversation_history_after_compression( @@ -4617,8 +4654,11 @@ def run_conversation( original_len = len(messages) original_tokens = estimate_messages_tokens_rough(messages) _overflow_input = messages + # Option A (LCM issue 441): overhead-aware request size so recovery arms on the + # true request (msgs + tools + system), not the tool-blind message count. messages, active_system_prompt = agent._compress_context( - messages, system_message, approx_tokens=approx_tokens, + messages, system_message, + approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), task_id=effective_task_id, ) if messages is _overflow_input and compression_skipped_due_to_lock(agent): @@ -4754,6 +4794,41 @@ def run_conversation( "failed": True, "compression_exhausted": True, } + # Also compress the message history so the output-cap + # retry does not just spin on max_tokens alone. The + # compressor drops the middle window, freeing enough + # tokens for the total to fit inside context_length. + # (#55546) + try: + original_len = len(messages) + original_tokens = estimate_messages_tokens_rough(messages) + _overflow_input = messages + messages, active_system_prompt = agent._compress_context( + messages, system_message, + approx_tokens=request_input_estimate, + task_id=effective_task_id, + ) + if messages is _overflow_input and compression_skipped_due_to_lock(agent): + compression_attempts -= 1 + agent._persist_session(messages, conversation_history) + return _compression_deferred_result( + agent, messages, api_call_count + ) + conversation_history = conversation_history_after_compression( + agent, messages, conversation_history + ) + new_tokens = estimate_messages_tokens_rough(messages) + if len(messages) < original_len: + agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages))) + elif new_tokens > 0 and new_tokens < original_tokens * 0.95: + agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens)) + except Exception: + # Compression must never turn an output-cap error + # fatal — fall through and retry on max_tokens alone. + logger.warning( + "%sOutput-cap compression hit an error; retrying on max_tokens only.", + agent.log_prefix, + ) _retry.restart_with_compressed_messages = True break @@ -4878,8 +4953,13 @@ def run_conversation( original_len = len(messages) original_tokens = estimate_messages_tokens_rough(messages) _overflow_input = messages + # Option A (LCM issue 441): pass the OVERHEAD-AWARE request size (msgs + tool + # schemas + system), not the tool-blind message count, so LCM forced-overflow + # recovery arms on the TRUE request that overflowed. See hermes-lcm engine + # _should_force_overflow_recovery. (approx_tokens stays for the status display.) messages, active_system_prompt = agent._compress_context( - messages, system_message, approx_tokens=approx_tokens, + messages, system_message, + approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), task_id=effective_task_id, ) if messages is _overflow_input and compression_skipped_due_to_lock(agent): @@ -6107,9 +6187,25 @@ def run_conversation( ] assistant_msg = agent._build_assistant_message(assistant_message, finish_reason) - + turn_content = assistant_message.content or "" + # Some local tool-call templates emit a bare bracketed token + # (for example ``[memory]``) as assistant content alongside a + # function call. It is protocol scaffolding, not an answer. + # Persisting or caching it as visible content lets the empty + # post-tool fallback replay that token forever after compaction (#78148). + if ( + assistant_message.tool_calls + and _STALE_MARKER_RE.fullmatch(turn_content.strip()) + ): + logger.warning( + "Discarding bare tool-call marker from assistant content: %s", + turn_content, + ) + turn_content = "" + assistant_msg["content"] = "" + # Classify tools in this turn to determine if they are all housekeeping. # This classification is needed regardless of whether the turn has visible content, # because a substantive tool-only turn must invalidate any older housekeeping fallback. @@ -6373,9 +6469,12 @@ def run_conversation( _clear_warn() agent._safe_print(" ⟳ compacting context…") _post_tool_input = messages + # Route the overhead-aware _real_tokens (computed above) into compression, not + # the bare last_prompt_tokens — which is 0 in the no-usage fallback, hiding the + # true request size from the engine's overflow guard (upstream PR #77169 review). messages, active_system_prompt = agent._compress_context( messages, system_message, - approx_tokens=agent.context_compressor.last_prompt_tokens, + approx_tokens=_real_tokens, task_id=effective_task_id, ) if ( @@ -6662,15 +6761,47 @@ def run_conversation( ) if _truly_empty and (not _has_structured or _prefill_exhausted) and agent._empty_content_retries < 3: agent._empty_content_retries += 1 + wait_time = jittered_backoff( + agent._empty_content_retries, + base_delay=5.0, + max_delay=60.0, + ) logger.warning( "Empty response (no content or reasoning) — " - "retry %d/3 (model=%s)", - agent._empty_content_retries, agent.model, + "retry %d/3 in %.1fs (model=%s)", + agent._empty_content_retries, wait_time, agent.model, ) agent._buffer_status( f"⚠️ Empty response from model — retrying " - f"({agent._empty_content_retries}/3)" + f"({agent._empty_content_retries}/3) in {wait_time:.0f}s" ) + # Sleep in small increments to stay responsive to interrupts + sleep_end = time.time() + wait_time + _backoff_touch_counter = 0 + while time.time() < sleep_end: + if agent._interrupt_requested: + agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during empty-response retry wait, aborting.", force=True) + _interrupt_text = ( + f"Operation interrupted: retrying empty response from model " + f"(retry {agent._empty_content_retries}/3)." + ) + close_interrupted_tool_sequence(messages, _interrupt_text) + agent._persist_session(messages, conversation_history) + agent.clear_interrupt() + return { + "final_response": _interrupt_text, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "interrupted": True, + } + time.sleep(0.2) + _backoff_touch_counter += 1 + if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s + agent._touch_activity( + f"empty response retry backoff ({agent._empty_content_retries}/3), " + f"{int(sleep_end - time.time())}s remaining" + ) continue # ── Exhausted retries — try fallback provider ── diff --git a/agent/credential_pool.py b/agent/credential_pool.py index adb24f75a6..85a6e79a75 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -124,6 +124,17 @@ SUPPORTED_POOL_STRATEGIES = { EXHAUSTED_TTL_401_SECONDS = 5 * 60 # 5 minutes EXHAUSTED_TTL_429_SECONDS = 60 * 60 # 1 hour EXHAUSTED_TTL_DEFAULT_SECONDS = 60 * 60 # 1 hour +# When a pool has no other credential to rotate to (the offending key is the +# sole non-DEAD entry), a 1-hour bench means an hour of hard failures with +# nothing to fall back to. Throttles (429/403/5xx) are transient and reset in +# seconds, so a sole credential cools down briefly instead — same rationale as +# the short 401 cooldown above. Provider-supplied reset_at still overrides. +EXHAUSTED_TTL_SOLE_CREDENTIAL_SECONDS = 60 # 1 minute + +# ``FailoverReason.billing`` as a bare string. The pool stores classified +# failure semantics as plain text (it persists to JSON and must not import +# the classifier), so the value is duplicated here rather than referenced. +FAILURE_REASON_BILLING = "billing" # Throttle window for the "no available entries" INFO line. Credential # selection runs on a hot path (every model call, plus auxiliary tasks like @@ -150,6 +161,13 @@ _EXTRA_KEYS = frozenset({ "token_type", "scope", "client_id", "portal_base_url", "obtained_at", "expires_in", "agent_key_id", "agent_key_expires_in", "agent_key_reused", "agent_key_obtained_at", "tls", "secret_source", "secret_fingerprint", + # Classified failure semantics for the last exhaustion, as decided by + # agent/error_classifier.py. The raw HTTP status is not enough to size a + # cooldown: providers return 403 for both an edge throttle (transient, + # seconds) and a spending/key limit (billing, needs a real fix). Persisted + # with the entry so a restart doesn't downgrade a billing bench back to a + # 60s transient cooldown. + "failure_reason", }) @@ -289,13 +307,39 @@ def _is_manual_source(source: str) -> bool: return normalized == SOURCE_MANUAL or normalized.startswith(f"{SOURCE_MANUAL}:") -def _exhausted_ttl(error_code: Optional[int]) -> int: - """Return cooldown seconds based on the HTTP status that caused exhaustion.""" +def _exhausted_ttl( + error_code: Optional[int], + *, + sole_credential: bool = False, + failure_reason: Optional[str] = None, +) -> int: + """Return cooldown seconds based on the HTTP status that caused exhaustion. + + When *sole_credential* is True the pool has no other entry to rotate to, so + a long bench just blocks the only key. Transient throttles (429 and the + catch-all default, which covers 403/5xx/unknown) are capped to a brief + cooldown so the sole key can recover — mirroring the short 401 path. 401 + keeps its own (already short) TTL. + + *failure_reason* is the classified semantics from + ``agent/error_classifier.py``. The raw status alone can't size the + cooldown: an OpenRouter ``key limit exceeded`` and an xAI spending-limit + block both arrive as **403** but classify as ``billing``, and a 60s retry + on a spent account just re-fails every minute. Billing keeps the full + bench regardless of status; 402 does too, since it is billing by + definition even when nothing classified it. + """ if error_code == 401: return EXHAUSTED_TTL_401_SECONDS - if error_code == 429: - return EXHAUSTED_TTL_429_SECONDS - return EXHAUSTED_TTL_DEFAULT_SECONDS + base = EXHAUSTED_TTL_429_SECONDS if error_code == 429 else EXHAUSTED_TTL_DEFAULT_SECONDS + # Sole credential: shorten only TRANSIENT throttles (429 rate-limit, 403 + # edge-throttle, 5xx server, or unknown). Billing exhaustion — whether + # classified as such or self-evident from a 402 — is a genuine depletion + # where a quick retry can't help, so it keeps the full bench. + is_billing = error_code == 402 or failure_reason == FAILURE_REASON_BILLING + if sole_credential and not is_billing: + return min(base, EXHAUSTED_TTL_SOLE_CREDENTIAL_SECONDS) + return base def _parse_absolute_timestamp(value: Any) -> Optional[float]: @@ -376,14 +420,18 @@ def _normalize_error_context(error_context: Optional[Dict[str, Any]]) -> Dict[st return normalized -def _exhausted_until(entry: PooledCredential) -> Optional[float]: +def _exhausted_until(entry: PooledCredential, *, sole_credential: bool = False) -> Optional[float]: if entry.last_status != STATUS_EXHAUSTED: return None reset_at = _parse_absolute_timestamp(getattr(entry, "last_error_reset_at", None)) if reset_at is not None: return reset_at if entry.last_status_at: - return entry.last_status_at + _exhausted_ttl(entry.last_error_code) + return entry.last_status_at + _exhausted_ttl( + entry.last_error_code, + sole_credential=sole_credential, + failure_reason=getattr(entry, "failure_reason", None), + ) return None @@ -588,7 +636,12 @@ class CredentialPool: self._entries = sorted(entries, key=lambda entry: entry.priority) self._current_id: Optional[str] = None self._strategy = get_pool_strategy(provider) - self._lock = threading.Lock() + # RLock: the mutation primitives below (_replace_entry/_persist) + # self-acquire this lock so the DEFERRED single-use-token refresh + # path (which runs network I/O outside the lock by design) still + # serializes its pool mutations. In-lock callers re-acquire + # reentrantly at negligible cost. + self._lock = threading.RLock() self._active_leases: Dict[str, int] = {} self._max_concurrent = DEFAULT_MAX_CONCURRENT_PER_CREDENTIAL # Monotonic timestamp of the last "no available entries" log, used to @@ -618,7 +671,43 @@ class CredentialPool: # otherwise a status probe here can race a concurrent ``select`` / # rotation and tear ``self._entries`` or double-write auth.json. with self._lock: - return bool(self._available_entries()) + available, _pending = self._available_entries() + return bool(available) + + def next_available_at(self) -> Optional[float]: + """Earliest epoch time (seconds) any entry re-enters rotation. + + Returns ``None`` when at least one entry is available right now, or + when no exhausted entry carries a usable recovery time (empty pool, + or only ``STATUS_DEAD`` entries, which never re-enter via TTL). + Callers must treat ``None`` as "no wait information", not + "unavailable". + + Like :meth:`has_available`, expired cooldowns are left uncleared + (``clear_expired=False``); the only writes are the same + re-auth/token sync paths ``has_available`` already performs — which + is exactly why this must run under ``self._lock`` like every other + ``_available_entries`` caller (see the comment on ``has_available``). + """ + with self._lock: + available, _pending = self._available_entries() + if available: + return None + # Mirror _available_entries: if the pool has no other credential + # to rotate to, the sole entry's transient throttle cools down in + # seconds — next_available_at must report that shorter window too, + # or the fallback restore gate waits an hour for a 60s cooldown. + sole_credential = sum( + 1 for e in self._entries if e.last_status != STATUS_DEAD + ) <= 1 + candidates: List[float] = [] + for entry in self._entries: + if entry.last_status != STATUS_EXHAUSTED: + continue + until = _exhausted_until(entry, sole_credential=sole_credential) + if until is not None: + candidates.append(until) + return min(candidates) if candidates else None def entries(self) -> List[PooledCredential]: with self._lock: @@ -656,18 +745,27 @@ class CredentialPool: return matches[0].id if len(matches) == 1 else None def _replace_entry(self, old: PooledCredential, new: PooledCredential) -> None: - """Swap an entry in-place by id, preserving sort order.""" - for idx, entry in enumerate(self._entries): - if entry.id == old.id: - self._entries[idx] = new - return + """Swap an entry in-place by id, preserving sort order. + + Self-locking (RLock) so the deferred refresh path — which + deliberately runs outside the pool lock — cannot tear + ``self._entries`` against a concurrent select()/rotation. + """ + with self._lock: + for idx, entry in enumerate(self._entries): + if entry.id == old.id: + self._entries[idx] = new + return def _persist(self, *, removed_ids: Optional[List[str]] = None) -> None: - write_credential_pool( - self.provider, - [entry.to_dict() for entry in self._entries], - removed_ids=removed_ids, - ) + # Self-locking (RLock): snapshotting self._entries must not race a + # concurrent rotation when called from the deferred refresh path. + with self._lock: + write_credential_pool( + self.provider, + [entry.to_dict() for entry in self._entries], + removed_ids=removed_ids, + ) def _is_terminal_auth_failure( self, @@ -699,6 +797,7 @@ class CredentialPool: error_context: Optional[Dict[str, Any]] = None, *, persist: bool = True, + failure_reason: Optional[str] = None, ) -> PooledCredential: normalized_error = _normalize_error_context(error_context) # Permanent OAuth failures (token_invalidated, token_revoked, etc.) @@ -712,6 +811,15 @@ class CredentialPool: terminal_status = STATUS_DEAD else: terminal_status = STATUS_EXHAUSTED + # Carry the classifier's verdict onto the entry so the cooldown can be + # sized by what actually failed, not just the HTTP status (a billing + # 403 must not get the sole-credential transient cooldown). Absent a + # classification, clear any stale verdict from a previous failure. + updated_extra = dict(entry.extra) + if failure_reason: + updated_extra["failure_reason"] = failure_reason + else: + updated_extra.pop("failure_reason", None) updated = replace( entry, last_status=terminal_status, @@ -720,6 +828,7 @@ class CredentialPool: last_error_reason=normalized_error.get("reason"), last_error_message=normalized_error.get("message"), last_error_reset_at=normalized_error.get("reset_at"), + extra=updated_extra, ) self._replace_entry(entry, updated) if persist: @@ -1415,17 +1524,22 @@ class CredentialPool: logger.debug( "Failed to clear terminal xAI OAuth state: %s", clear_exc ) - removed_ids = [ - item.id for item in self._entries - if item.source == "device_code" - ] - self._entries = [ - item for item in self._entries - if item.source != "device_code" - ] - if self._current_id == entry.id: - self._current_id = None - self._persist(removed_ids=removed_ids) + # Read-modify-write of self._entries: must be atomic. + # This runs on the DEFERRED refresh path (outside the + # pool lock), so take it here. self._lock is an RLock, + # so the still-locked callers re-enter safely. + with self._lock: + removed_ids = [ + item.id for item in self._entries + if item.source == "device_code" + ] + self._entries = [ + item for item in self._entries + if item.source != "device_code" + ] + if self._current_id == entry.id: + self._current_id = None + self._persist(removed_ids=removed_ids) return None # For openai-codex: same race as xAI/nous — another Hermes process # may have consumed the refresh token between our proactive sync @@ -1485,17 +1599,22 @@ class CredentialPool: logger.debug( "Failed to clear terminal Codex OAuth state: %s", clear_exc ) - removed_ids = [ - item.id for item in self._entries - if item.source == "device_code" - ] - self._entries = [ - item for item in self._entries - if item.source != "device_code" - ] - if self._current_id == entry.id: - self._current_id = None - self._persist(removed_ids=removed_ids) + # Read-modify-write of self._entries: must be atomic. + # This runs on the DEFERRED refresh path (outside the + # pool lock), so take it here. self._lock is an RLock, + # so the still-locked callers re-enter safely. + with self._lock: + removed_ids = [ + item.id for item in self._entries + if item.source == "device_code" + ] + self._entries = [ + item for item in self._entries + if item.source != "device_code" + ] + if self._current_id == entry.id: + self._current_id = None + self._persist(removed_ids=removed_ids) return None # For nous: another process may have consumed the refresh token # between our proactive sync and the HTTP call. Re-sync from @@ -1552,17 +1671,19 @@ class CredentialPool: auth_mod.NOUS_DEVICE_CODE_SOURCE, f"manual:{auth_mod.NOUS_DEVICE_CODE_SOURCE}", } - removed_ids = [ - item.id for item in self._entries - if item.source in singleton_sources - ] - self._entries = [ - item for item in self._entries - if item.source not in singleton_sources - ] - if self._current_id == entry.id: - self._current_id = None - self._persist(removed_ids=removed_ids) + # Atomic read-modify-write; see the note above. + with self._lock: + removed_ids = [ + item.id for item in self._entries + if item.source in singleton_sources + ] + self._entries = [ + item for item in self._entries + if item.source not in singleton_sources + ] + if self._current_id == entry.id: + self._current_id = None + self._persist(removed_ids=removed_ids) return None self._mark_exhausted(entry, None) return None @@ -1646,26 +1767,70 @@ class CredentialPool: return False def select(self) -> Optional[PooledCredential]: - with self._lock: - entry = self._select_unlocked() - if entry is not None: - # A normal (non-recovery) selection starts a fresh episode — - # don't let a leftover unmatched-rotation streak from an old - # failure trip the #70401 bound early next time. - self._unmatched_rotation_streak = 0 + entry, pending_refresh = self._select_under_lock() + if pending_refresh: + self._refresh_pending_entries(pending_refresh) + if entry is not None: + self._unmatched_rotation_streak = 0 return entry + # If no entry was available but we just refreshed some, re-select + # now that the refreshed entries are back in the pool. + if pending_refresh: + entry, _ = self._select_under_lock() + if entry is not None: + self._unmatched_rotation_streak = 0 + return entry - def _available_entries(self, *, clear_expired: bool = False, refresh: bool = False) -> List[PooledCredential]: - """Return entries not currently in exhaustion cooldown. + def _select_under_lock(self) -> Tuple[Optional[PooledCredential], List[tuple]]: + """Run selection under the lock, returning entry + pending refreshes.""" + with self._lock: + return self._select_unlocked() + + def _refresh_pending_entries(self, pending: List[tuple]) -> None: + """Refresh deferred single-use-token entries outside the lock. + + Each entry is refreshed under the cross-process ``_auth_store_lock`` + (which can block for 20+ seconds) and then merged into the pool. + On failure the entry is silently skipped. + """ + for entry, sync_fn in pending: + # _refresh_entry merges the refreshed entry into the pool + # internally. Its mutation primitives (_replace_entry, _persist) + # are self-locking, and the quarantine paths inside + # _refresh_entry_impl take self._lock explicitly around their + # read-modify-write of self._entries — required because this + # call site runs OUTSIDE the pool lock. + self._refresh_entry(entry, force=False) + + def _available_entries( + self, *, clear_expired: bool = False, refresh: bool = False, + ) -> Tuple[List[PooledCredential], List[tuple]]: + """Return (available, pending_refresh) for entries not in cooldown. When *clear_expired* is True, entries whose cooldown has elapsed are reset to STATUS_OK and persisted. When *refresh* is True, entries that need a token refresh are refreshed (skipped on failure). + + Single-use-token refreshes (openai-codex, xai-oauth) are returned as + *pending_refresh* tuples so the caller can execute them outside the + lock, avoiding stalling all pool consumers during cross-process flock + acquisition + OAuth network I/O. """ now = time.time() cleared_any = False entries_to_prune: List[str] = [] available: List[PooledCredential] = [] + # Entries that need an OAuth refresh via a single-use token provider + # (openai-codex, xai-oauth). These require a cross-process file lock + # that can block for 20+ seconds. We collect them under self._lock + # and refresh outside the lock to avoid stalling all pool consumers. + pending_refresh: List[tuple] = [] # (entry, sync_entry_fn) + # DEAD entries never re-enter rotation, so if at most one non-DEAD entry + # exists there is nothing to rotate to: an exhausted sole credential + # should cool down briefly rather than bench the only key for an hour. + sole_credential = sum( + 1 for e in self._entries if e.last_status != STATUS_DEAD + ) <= 1 for entry in self._entries: # Borrowed credentials persist as metadata-only references and are # hydrated from their live source on load. A stale duplicate row @@ -1746,7 +1911,7 @@ class CredentialPool: # the re-auth case for OAuth singletons. continue if entry.last_status == STATUS_EXHAUSTED: - exhausted_until = _exhausted_until(entry) + exhausted_until = _exhausted_until(entry, sole_credential=sole_credential) if exhausted_until is not None and now < exhausted_until: # Codex quota windows can reopen EARLY: the user redeems a # banked rate-limit reset (Codex CLI / ChatGPT UI), upgrades @@ -1774,6 +1939,16 @@ class CredentialPool: entry = cleared cleared_any = True if refresh and self._entry_needs_refresh(entry): + if self.provider in ("openai-codex", "xai-oauth"): + # Defer single-use-token refresh to avoid holding the + # threading lock during cross-process flock + network I/O. + sync_fn = ( + self._sync_codex_entry_from_auth_store + if self.provider == "openai-codex" + else self._sync_xai_oauth_entry_from_pool_store + ) + pending_refresh.append((entry, sync_fn)) + continue refreshed = self._refresh_entry(entry, force=False) if refreshed is None: continue @@ -1784,7 +1959,7 @@ class CredentialPool: self._entries = [e for e in self._entries if e.id not in pruned_ids] if cleared_any: self._persist(removed_ids=entries_to_prune) - return available + return available, pending_refresh def _log_no_available_entries(self) -> None: """Emit the empty-pool INFO line at most once per throttle window. @@ -1800,12 +1975,17 @@ class CredentialPool: self._last_no_entries_log_at = now logger.info("credential pool: no available entries (all exhausted or empty)") - def _select_unlocked(self, *, refresh: bool = True) -> Optional[PooledCredential]: - available = self._available_entries(clear_expired=True, refresh=refresh) + def _select_unlocked(self, *, refresh: bool = True) -> Tuple[Optional[PooledCredential], List[tuple]]: + """Select the best available credential entry. + + Returns ``(entry, pending_refresh)`` where *pending_refresh* contains + single-use-token entries that must be refreshed outside the lock. + """ + available, pending_refresh = self._available_entries(clear_expired=True, refresh=refresh) if not available: self._current_id = None self._log_no_available_entries() - return None + return None, pending_refresh # A successful selection means the pool recovered; re-arm the throttle # so a later re-exhaustion logs immediately rather than being silenced @@ -1815,7 +1995,7 @@ class CredentialPool: if self._strategy == STRATEGY_RANDOM: entry = random.choice(available) self._current_id = entry.id - return entry + return entry, pending_refresh if self._strategy == STRATEGY_LEAST_USED and len(available) > 1: entry = min(available, key=lambda e: e.request_count) @@ -1823,7 +2003,7 @@ class CredentialPool: updated = replace(entry, request_count=entry.request_count + 1) self._replace_entry(entry, updated) self._current_id = entry.id - return updated + return updated, pending_refresh if self._strategy == STRATEGY_ROUND_ROBIN and len(available) > 1: entry = available[0] @@ -1832,11 +2012,11 @@ class CredentialPool: self._entries = [replace(candidate, priority=idx) for idx, candidate in enumerate(rotated)] self._persist() self._current_id = entry.id - return self._current_unlocked() or entry + return self._current_unlocked() or entry, pending_refresh entry = available[0] self._current_id = entry.id - return entry + return entry, pending_refresh def peek(self) -> Optional[PooledCredential]: # Single lock acquisition for the whole read; call the unlocked @@ -1845,7 +2025,7 @@ class CredentialPool: current = self._current_unlocked() if current is not None: return current - available = self._available_entries() + available, _pending = self._available_entries() return available[0] if available else None def mark_exhausted_and_rotate( @@ -1855,6 +2035,7 @@ class CredentialPool: error_context: Optional[Dict[str, Any]] = None, api_key_hint: Optional[str] = None, credential_id: Optional[str] = None, + failure_reason: Optional[str] = None, ) -> Optional[PooledCredential]: with self._lock: entry = None @@ -1894,7 +2075,8 @@ class CredentialPool: # guessing and surface the error (no cooldown is written for # anybody — healthy keys stay available for the next turn). self._unmatched_rotation_streak += 1 - available_count = len(self._available_entries()) + available_count, _ = self._available_entries() + available_count = len(available_count) if self._unmatched_rotation_streak > max(available_count, 1): logger.warning( "credential pool: failed credential identity matched no " @@ -1913,8 +2095,9 @@ class CredentialPool: self.provider, ) self._current_id = None - next_entry = self._select_unlocked() - if next_entry is not None and len(self._available_entries()) == 1: + next_entry, _pending = self._select_unlocked(refresh=False) + avail, _ = self._available_entries() + if next_entry is not None and len(avail) == 1: # A single-entry pool cannot rotate. Returning its only # entry reports a successful recovery without changing # the credential, so the caller retries the same 401 @@ -1927,11 +2110,13 @@ class CredentialPool: # streak is stale (this mark WILL advance pool state). self._unmatched_rotation_streak = 0 if entry is None: - entry = self._current_unlocked() or self._select_unlocked() + entry = self._current_unlocked() or self._select_unlocked(refresh=False)[0] if entry is None: return None _label = entry.label or entry.id[:8] - self._mark_exhausted(entry, status_code, error_context) + self._mark_exhausted( + entry, status_code, error_context, failure_reason=failure_reason + ) # A 402/429/401 is an API-key–level failure: the account is out of # balance, rate-limited, or its key is rejected. The same key can # back more than one pool entry (e.g. an explicit pool entry plus a @@ -1951,7 +2136,11 @@ class CredentialPool: continue if sibling.runtime_api_key == failed_runtime_key: self._mark_exhausted( - sibling, status_code, error_context, persist=False + sibling, + status_code, + error_context, + persist=False, + failure_reason=failure_reason, ) siblings_marked = True if siblings_marked: @@ -1972,7 +2161,7 @@ class CredentialPool: _label, status_code, ) self._current_id = None - next_entry = self._select_unlocked() + next_entry, _pending = self._select_unlocked(refresh=False) if next_entry: _next_label = next_entry.label or next_entry.id[:8] logger.info("credential pool: rotated to %s", _next_label) @@ -1986,15 +2175,32 @@ class CredentialPool: a stable tie-breaker. When every credential is already at the soft cap, still return the least-leased one instead of blocking. """ + chosen_id, pending_refresh = self._acquire_lease_under_lock(credential_id) + if pending_refresh: + self._refresh_pending_entries(pending_refresh) + # Mirror select(): if nothing was leasable but we just refreshed + # deferred single-use-token entries, retry now that they are back + # in rotation. Without this, a pool whose only entries all needed + # a refresh returns None even though the refresh succeeded — the + # caller sees "no credentials available" and fails a request that + # should have gone through. + if chosen_id is None: + chosen_id, _ = self._acquire_lease_under_lock(credential_id) + return chosen_id + + def _acquire_lease_under_lock( + self, credential_id: Optional[str], + ) -> Tuple[Optional[str], List[tuple]]: + """Run lease acquisition under the lock, returning id + pending refreshes.""" with self._lock: if credential_id: self._active_leases[credential_id] = self._active_leases.get(credential_id, 0) + 1 self._current_id = credential_id - return credential_id + return credential_id, [] - available = self._available_entries(clear_expired=True, refresh=True) + available, pending_refresh = self._available_entries(clear_expired=True, refresh=True) if not available: - return None + return None, pending_refresh below_cap = [ entry for entry in available @@ -2007,7 +2213,7 @@ class CredentialPool: ) self._active_leases[chosen.id] = self._active_leases.get(chosen.id, 0) + 1 self._current_id = chosen.id - return chosen.id + return chosen.id, pending_refresh def release_lease(self, credential_id: str) -> None: """Release a previously acquired credential lease.""" @@ -2059,7 +2265,7 @@ class CredentialPool: else: entry = self._current_unlocked() or self._select_unlocked( refresh=False - ) + )[0] if entry is None: return None self._current_id = entry.id @@ -2173,6 +2379,11 @@ def _upsert_entry(entries: List[PooledCredential], provider: str, source: str, p field_updates = {} extra_updates = {} _field_names = {f.name for f in fields(existing)} + token_changed = ( + "access_token" in payload + and payload["access_token"] is not None + and payload["access_token"] != existing.access_token + ) for key, value in payload.items(): if key in {"id", "priority"} or value is None: continue @@ -2184,6 +2395,15 @@ def _upsert_entry(entries: List[PooledCredential], provider: str, source: str, p elif key in _EXTRA_KEYS: if existing.extra.get(key) != value: extra_updates[key] = value + # When the credential token itself changes (key rotation), clear any + # exhaustion/error state — the old status is stale for the new key. + if token_changed and existing.last_status is not None: + field_updates["last_status"] = None + field_updates["last_status_at"] = None + field_updates["last_error_code"] = None + field_updates["last_error_reason"] = None + field_updates["last_error_message"] = None + field_updates["last_error_reset_at"] = None if field_updates or extra_updates: if extra_updates: field_updates["extra"] = {**existing.extra, **extra_updates} @@ -2387,9 +2607,41 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup # env vars (COPILOT_GITHUB_TOKEN / GH_TOKEN). They don't live in # the auth store or credential pool, so we resolve them here. try: - from hermes_cli.copilot_auth import resolve_copilot_token, get_copilot_api_token + from hermes_cli.copilot_auth import ( + COPILOT_ENV_VARS, + resolve_copilot_token, + get_copilot_api_token, + ) + # All-sources suppression gate BEFORE any work — including the + # `gh auth token` subprocess spawn. resolve_copilot_token() + # shells out (~30ms), and the exchange retries 3x with backoff + # (~35s worst case); a user who suppressed every copilot source + # (hermes auth remove copilot gh_cli) must not pay either on + # every pool load (model picker open, /model, agent startup). + # Enumerating the full source space here matches what + # credential_sources._remove_copilot_gh suppresses, so an + # all-suppressed check is stable. + copilot_sources = ["gh_cli"] + [f"env:{v}" for v in COPILOT_ENV_VARS] + if all(_is_suppressed(provider, s) for s in copilot_sources): + return changed, active_sources token, source = resolve_copilot_token() if token: + # ``resolve_copilot_token`` returns exactly "gh auth token" + # for the CLI path; env-sourced tokens return the var name. + # Match exactly — a substring test classifies GH_TOKEN and + # GITHUB_TOKEN as gh_cli, silently bypassing a user's + # per-env-var suppression. + source_name = "gh_cli" if source == "gh auth token" else f"env:{source}" + # Per-source suppression gate (a user may suppress only the + # gh CLI path and keep an env var, or vice versa) BEFORE the + # network exchange. The exchange retries 3x with 10s + # timeouts and 4.5s total backoff (~35s worst case), so a + # source the user already suppressed + # must not burn that dead time just to have the entry + # discarded afterwards. Same early-gate pattern every other + # singleton branch uses. + if _is_suppressed(provider, source_name): + return changed, active_sources api_token, enterprise_base_url = get_copilot_api_token(token) # Observability: get_copilot_api_token falls back to returning # the RAW token when the exchange fails. A raw ~40-char token @@ -2405,27 +2657,25 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup "unavailable); enterprise-only models may 400 with " "model_not_available_for_integrator until exchange recovers." ) - source_name = "gh_cli" if "gh" in source.lower() else f"env:{source}" - if not _is_suppressed(provider, source_name): - active_sources.add(source_name) - pconfig = PROVIDER_REGISTRY.get(provider) - # Use enterprise base URL from token exchange if available, - # otherwise fall back to the provider's default. - effective_base_url = enterprise_base_url or ( - pconfig.inference_base_url if pconfig else "" - ) - changed |= _upsert_entry( - entries, - provider, - source_name, - { - "source": source_name, - "auth_type": AUTH_TYPE_API_KEY, - "access_token": api_token, - "base_url": effective_base_url, - "label": source, - }, - ) + active_sources.add(source_name) + pconfig = PROVIDER_REGISTRY.get(provider) + # Use enterprise base URL from token exchange if available, + # otherwise fall back to the provider's default. + effective_base_url = enterprise_base_url or ( + pconfig.inference_base_url if pconfig else "" + ) + changed |= _upsert_entry( + entries, + provider, + source_name, + { + "source": source_name, + "auth_type": AUTH_TYPE_API_KEY, + "access_token": api_token, + "base_url": effective_base_url, + "label": source, + }, + ) except Exception as exc: logger.debug("Copilot token seed failed: %s", exc) @@ -2571,6 +2821,31 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup return changed, active_sources +# Prefer ~/.hermes/.env over os.environ — the user's config file is the +# authoritative source for Hermes credentials. Stale env vars from parent +# processes (Codex CLI, test scripts, etc.) should not override deliberate +# changes to the .env file. load_env() memoizes on the .env mtime, so +# per-call reads (pool seeding, per-turn credential refresh) cost a stat() +# when the file is unchanged. +def get_env_prefer_dotenv(key: str) -> str: + env_file = load_env() + raw = env_file.get(key, "").strip() + scoped_value = (_get_secret(key, "") or "").strip() + # If .env contains an unresolved op:// reference, prefer the + # already-resolved value supplied by the active secret scope (or by + # os.environ in legacy single-profile mode), set by + # load_hermes_dotenv() -> apply_onepassword_secrets()). The raw + # "op://Vault/Item/field" string would otherwise win and every + # provider auth attempt would receive a URL instead of a key. This + # happens during a partial migration, or when the user wrote op:// + # references straight into .env rather than the secrets.onepassword + # config block. For every non-op:// value the original + # .env-takes-precedence behaviour is preserved unchanged. + if raw.startswith("op://") and scoped_value: + return scoped_value + return raw or scoped_value + + def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool, Set[str]]: changed = False active_sources: Set[str] = set() @@ -2589,27 +2864,10 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool if provider == "copilot": return False, active_sources - # Prefer ~/.hermes/.env over os.environ — the user's config file is the - # authoritative source for Hermes credentials. Stale env vars from parent - # processes (Codex CLI, test scripts, etc.) should not override deliberate - # changes to the .env file. - def _get_env_prefer_dotenv(key: str) -> str: - env_file = load_env() - raw = env_file.get(key, "").strip() - scoped_value = (_get_secret(key, "") or "").strip() - # If .env contains an unresolved op:// reference, prefer the - # already-resolved value supplied by the active secret scope (or by - # os.environ in legacy single-profile mode), set by - # load_hermes_dotenv() -> apply_onepassword_secrets()). The raw - # "op://Vault/Item/field" string would otherwise win and every - # provider auth attempt would receive a URL instead of a key. This - # happens during a partial migration, or when the user wrote op:// - # references straight into .env rather than the secrets.onepassword - # config block. For every non-op:// value the original - # .env-takes-precedence behaviour is preserved unchanged. - if raw.startswith("op://") and scoped_value: - return scoped_value - return raw or scoped_value + # The .env-preferring resolution lives at module level + # (``get_env_prefer_dotenv``) so the pool seeder and the per-turn + # credential refresh share one implementation. + _get_env_prefer_dotenv = get_env_prefer_dotenv # Honour user suppression — `hermes auth remove ` for an # env-seeded credential marks the env: source as suppressed so it diff --git a/agent/curator.py b/agent/curator.py index 750dd7c2eb..ab0adddae1 100644 --- a/agent/curator.py +++ b/agent/curator.py @@ -1923,6 +1923,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]: credential_pool=_credential_pool, request_overrides=_request_overrides, **_agent_kwargs, + enabled_toolsets=["skills", "terminal"], # Umbrella-building over a large skill collection is worth a # high iteration ceiling — the pass typically takes 50-100 # API calls against hundreds of candidate skills. The diff --git a/agent/display.py b/agent/display.py index 3da6e2f24e..d6bea7e54e 100644 --- a/agent/display.py +++ b/agent/display.py @@ -14,6 +14,7 @@ from dataclasses import dataclass, field from difflib import unified_diff from pathlib import Path from typing import Any +from urllib.parse import urlsplit from utils import safe_json_loads from agent.redact import redact_sensitive_text @@ -187,6 +188,15 @@ def _truncate_preview(text: str, max_len: int | None) -> str: return text +@dataclass(frozen=True) +class ToolPreview: + """A compact tool preview plus presentation facts lost to truncation.""" + + text: str + truncated: bool = False + url: str | None = None + + _SHELL_SILENT_HEADS = {"cd", "pushd", "popd", "export", "set", "unset", "source", ".", "true", "false", ":"} _SHELL_PIPE_TAIL_HEADS = {"head", "tail", "wc", "sort", "uniq"} @@ -556,6 +566,35 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - return preview +def prepare_tool_preview( + tool_name: str, + args: dict | None, + *, + fallback: str, + max_len: int, +) -> ToolPreview: + """Build one canonical compact preview before platform formatting. + + The uncapped preview is rebuilt from the tool arguments when possible so + an upstream display cap cannot discard its link target. Platforms then + receive explicit truncation and URL metadata instead of inferring either + fact from the rendered text. + """ + full_text = build_tool_preview(tool_name, args, max_len=0) or fallback + text = _truncate_preview(full_text, max_len) + truncated = text != full_text + url = None + if truncated: + candidate = _display_url(full_text) + try: + parsed = urlsplit(candidate) + except ValueError: + parsed = None + if parsed and parsed.scheme.lower() in {"http", "https"} and parsed.netloc: + url = candidate + return ToolPreview(text=text, truncated=truncated, url=url) + + # ========================================================================= # Friendly tool labels (human-phrased verbs for built-in tools) # @@ -1506,5 +1545,3 @@ def get_cute_tool_message( # ========================================================================= # Honcho session line (one-liner with clickable OSC 8 hyperlink) # ========================================================================= - - diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 8ac0b6c872..92d9fd43ef 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -339,6 +339,32 @@ _MODEL_NOT_FOUND_PATTERNS = [ "no endpoints found that support tool use", ] + +def _model_id_missing_known_prefix(model: str, provider: str) -> bool: + """True when a bare model id is only known to the provider as ``vendor/id``. + + Some providers answer a malformed model id with a naked 404 that names + nothing — NVIDIA NIM returns ``404 page not found`` for a bare + ``nemotron-3-ultra-550b-a55b``, indistinguishable from a bad endpoint + path. Consulting the curated catalogue tells the two apart: if the id + carries no ``/`` but the catalogue has exactly one entry ending in + ``/``, the prefix was dropped and the failure is deterministic. + + Never guesses — an id absent from the catalogue (a local NIM container, + a proxied model) returns False so genuine endpoint problems keep their + retryable ``unknown`` classification. + """ + name = (model or "").strip() + if not name or "/" in name: + return False + try: + from hermes_cli.model_normalize import suggest_prefixed_model_id + + return bool(suggest_prefixed_model_id((provider or "").strip(), name)) + except Exception: + return False + + # Malformed-message-array 400s. Deterministic request-shape rejections that # describe the *transcript* being invalid, not a parameter. The canonical # case: a stream dies mid-response and Hermes persists a content-less @@ -1061,6 +1087,18 @@ def _classify_by_status( retryable=False, should_fallback=True, ) + # A bare id that the provider's catalogue only knows in prefixed form + # is a malformed model id, not a routing glitch — NVIDIA NIM answers + # one with a naked ``404 page not found`` that names nothing, so the + # generic branch below burns three retries and reports what looks + # like an outage (#78796). Deterministic: don't retry, and let the + # model_not_found surface carry the real cause. + if _model_id_missing_known_prefix(model, provider): + return result_fn( + FailoverReason.model_not_found, + retryable=False, + should_fallback=True, + ) # Generic 404 with no "model not found" signal — could be a wrong # endpoint path (common with local llama.cpp / Ollama / vLLM when # the URL is slightly misconfigured), a proxy routing glitch, or diff --git a/agent/insights.py b/agent/insights.py index 9d148a1544..34e78a6ff9 100644 --- a/agent/insights.py +++ b/agent/insights.py @@ -99,6 +99,31 @@ class InsightsEngine: """ self.db = db self._conn = db._conn + # INDEXED BY is a hard dependency (SQLite errors on a missing index). + # A read-only open of a state.db written by an older version skips + # schema init and lacks the partial index — probe once and fall back + # to the unpinned variants (identical rows, optimizer-chosen plan). + try: + self._has_assistant_calls_index = bool( + self._conn.execute( + "SELECT 1 FROM sqlite_master WHERE type='index' AND name=?", + (self._MESSAGES_ASSISTANT_CALLS_INDEX,), + ).fetchone() + ) + except sqlite3.Error: + self._has_assistant_calls_index = False + if not self._has_assistant_calls_index: + _strip = f" INDEXED BY {self._MESSAGES_ASSISTANT_CALLS_INDEX}" + # Loop over every pinned statement so adding a new one can't + # forget its strip line (which would be a hard `no such index` + # crash on read-only DBs — the exact bug this fallback prevents). + for _attr in ( + "_GET_TOOL_CALLS_WITH_SOURCE", + "_GET_TOOL_CALLS_ALL", + "_GET_SKILL_CALLS_WITH_SOURCE", + "_GET_SKILL_CALLS_ALL", + ): + setattr(self, _attr, getattr(self, _attr).replace(_strip, "")) def generate(self, days: int = 30, source: str = None) -> Dict[str, Any]: """ @@ -171,6 +196,21 @@ class InsightsEngine: "top_sessions": top_sessions, } + def get_usage_breakdown(self, days: int = 30, source: str = None) -> Dict[str, Any]: + """Return the analytics-usage payload without running a full generate(). + + Uses the instr()-prefiltered _get_skill_usage query so only messages + that reference skill_view or skill_manage are loaded from SQLite, while + still preserving the per-tool breakdown used by the dashboard route. + """ + cutoff = time.time() - (days * 86400) + tool_usage = self._get_tool_usage(cutoff, source) + skill_usage = self._get_skill_usage(cutoff, source) + return { + "tools": self._compute_tool_breakdown(tool_usage), + "skills": self._compute_skill_breakdown(skill_usage), + } + # ========================================================================= # Data gathering (SQL queries) # ========================================================================= @@ -195,6 +235,53 @@ class InsightsEngine: " ORDER BY started_at DESC" ) + # Assistant ``tool_calls`` scan for tool/skill usage. ``INDEXED BY`` pins + # the partial index ``idx_messages_assistant_calls_by_session`` so the plan + # is deterministic on a freshly initialized state.db (before ANALYZE has + # run) for BOTH the unfiltered and source-filtered branches — without the + # hint the optimizer falls back to ``idx_messages_session_active`` for the + # source-filtered probe and scans each session's non-tool-call rows. + # + # The pin is a HARD dependency: SQLite raises ``no such index`` when the + # named index is absent. That happens in practice — the web dashboard's + # usage analytics open the DB ``read_only=True`` (skipping + # ``_init_schema``), so a state.db created by an older writer has no + # partial index yet. ``__init__`` probes for the index once and falls + # back to the unpinned (still-correct, just optimizer-chosen) variants. + _MESSAGES_ASSISTANT_CALLS_INDEX = "idx_messages_assistant_calls_by_session" + _GET_TOOL_CALLS_WITH_SOURCE = ( + "SELECT m.tool_calls" + f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}" + " JOIN sessions s ON s.id = m.session_id" + " WHERE s.started_at >= ? AND s.source = ?" + " AND m.role = 'assistant' AND m.tool_calls IS NOT NULL" + ) + _GET_TOOL_CALLS_ALL = ( + "SELECT m.tool_calls" + f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}" + " JOIN sessions s ON s.id = m.session_id" + " WHERE s.started_at >= ?" + " AND m.role = 'assistant' AND m.tool_calls IS NOT NULL" + ) + _GET_SKILL_CALLS_WITH_SOURCE = ( + "SELECT m.tool_calls, m.timestamp" + f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}" + " JOIN sessions s ON s.id = m.session_id" + " WHERE s.started_at >= ? AND s.source = ?" + " AND m.role = 'assistant' AND m.tool_calls IS NOT NULL" + " AND (instr(m.tool_calls, 'skill_view') > 0" + " OR instr(m.tool_calls, 'skill_manage') > 0)" + ) + _GET_SKILL_CALLS_ALL = ( + "SELECT m.tool_calls, m.timestamp" + f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}" + " JOIN sessions s ON s.id = m.session_id" + " WHERE s.started_at >= ?" + " AND m.role = 'assistant' AND m.tool_calls IS NOT NULL" + " AND (instr(m.tool_calls, 'skill_view') > 0" + " OR instr(m.tool_calls, 'skill_manage') > 0)" + ) + def _get_sessions(self, cutoff: float, source: str = None) -> List[Dict]: """Fetch sessions within the time window.""" if source: @@ -243,22 +330,10 @@ class InsightsEngine: # (covers CLI sessions where tool_name is NULL on tool responses) if source: cursor2 = self._conn.execute( - """SELECT m.tool_calls - FROM messages m - JOIN sessions s ON s.id = m.session_id - WHERE s.started_at >= ? AND s.source = ? - AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""", - (cutoff, source), + self._GET_TOOL_CALLS_WITH_SOURCE, (cutoff, source) ) else: - cursor2 = self._conn.execute( - """SELECT m.tool_calls - FROM messages m - JOIN sessions s ON s.id = m.session_id - WHERE s.started_at >= ? - AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""", - (cutoff,), - ) + cursor2 = self._conn.execute(self._GET_TOOL_CALLS_ALL, (cutoff,)) tool_calls_counts = Counter() for row in cursor2.fetchall(): @@ -301,22 +376,10 @@ class InsightsEngine: if source: cursor = self._conn.execute( - """SELECT m.tool_calls, m.timestamp - FROM messages m - JOIN sessions s ON s.id = m.session_id - WHERE s.started_at >= ? AND s.source = ? - AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""", - (cutoff, source), + self._GET_SKILL_CALLS_WITH_SOURCE, (cutoff, source) ) else: - cursor = self._conn.execute( - """SELECT m.tool_calls, m.timestamp - FROM messages m - JOIN sessions s ON s.id = m.session_id - WHERE s.started_at >= ? - AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""", - (cutoff,), - ) + cursor = self._conn.execute(self._GET_SKILL_CALLS_ALL, (cutoff,)) for row in cursor.fetchall(): try: diff --git a/agent/memory_provider.py b/agent/memory_provider.py index 4210a4c252..559fc3df6c 100644 --- a/agent/memory_provider.py +++ b/agent/memory_provider.py @@ -34,12 +34,50 @@ Optional hooks (override to opt in): from __future__ import annotations import logging +import re from abc import ABC, abstractmethod from typing import Any, Dict, List, Optional logger = logging.getLogger(__name__) +# Prompts that carry no semantic signal — trivial acknowledgements, greetings, +# slash commands, empty input. Single source of truth shared by the core +# per-turn prefetch gate (agent/turn_context.py, run_agent.py) and provider- +# side classifiers (plugins/memory/honcho) so the two can never drift apart. +# The alternation is anchored and may only be followed by whitespace or +# punctuation, so words that merely START with a trivial word ("k8s", "yolo", +# "note", "hindsight") do NOT match, while trailing-punctuation variants +# ("hi!", "hey.", "thanks :)", "done???") do. +TRIVIAL_PROMPT_RE = re.compile( + r'^(yes|no|ok|okay|sure|thanks|thank you|y|n|yep|nope|yeah|nah|' + r'hi|hey|hello|yo|sup|' + r'continue|go ahead|do it|proceed|got it|cool|nice|great|done|next|lgtm|k)' + r'[\s!?.:;,"' + "'" + r'~\u2018\u2019\u201c\u201d\u2014\u2013\u2026()\[\]{}<>*&^%$#@!+=`\u00a0]*$', + re.IGNORECASE, +) + + +def is_trivial_prompt(text: Optional[str]) -> bool: + """Return True if a user prompt is too trivial to warrant memory recall. + + Empty/whitespace-only input, slash commands, and bare greetings or + acknowledgements (with optional trailing punctuation) all count as + trivial. Callers use this to skip memory-provider prefetch/injection + on turns that carry no semantic signal — saving a blocking network + round-trip and preventing stale user-model context from derailing + one-word replies. + """ + if not text: + return True + stripped = text.strip() + if not stripped: + return True + if stripped.startswith("/"): + return True + return bool(TRIVIAL_PROMPT_RE.match(stripped)) + + class MemoryProvider(ABC): """Abstract base class for memory providers.""" @@ -253,6 +291,10 @@ class MemoryProvider(ABC): required: True if required (default: False) default: default value (optional) choices: list of valid values (optional) + type: text, integer, number, or boolean (optional) + minimum: numeric lower bound for integer/number fields (optional) + maximum: numeric upper bound for integer/number fields (optional) + step: numeric input step for Dashboard rendering (optional) url: URL where user can get this credential (optional) env_var: explicit env var name for secrets (default: auto-generated) diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 26e9523ec3..0e15492a8f 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -12,6 +12,7 @@ import hashlib import logging import re import threading +import time from concurrent.futures import ThreadPoolExecutor, wait as _futures_wait from types import SimpleNamespace from typing import Any @@ -151,6 +152,24 @@ def _redact_trace_accounting(acct: Any) -> Any: ) +# Cold-start caches. A MoA preset switch used to re-resolve the full +# config + preset + every slot's provider runtime on EACH create() call +# (once per tool-loop iteration), serially before the parallel fan-out could +# start — adding 5-30s of "frozen" latency on complex presets +# (#66793). The preset structure is immutable for the life of a turn, so +# cache both the resolved preset and each (provider, model) runtime. +_preset_cache_lock = threading.Lock() +_preset_cache: dict[tuple, Any] = {} + +_runtime_cache_lock = threading.Lock() +_runtime_cache: dict[tuple[str, str], tuple[float, dict[str, Any]]] = {} + +# Runtime entries go stale when providers/credentials change (key rotation, +# base_url edits). Deliberately short-lived: 300s collapses the per-iteration +# re-resolution inside a turn while bounding credential staleness between +# turns — the non-MoA path picks up rotated keys immediately, this path +# within 5 minutes. +_RUNTIME_CACHE_TTL_SECONDS = 300.0 # Upper bound on concurrent reference-model calls. References are independent # advisory calls (no tools, no inter-dependence), so we fan them out the same @@ -322,34 +341,33 @@ def _slot_runtime(slot: dict[str, Any]) -> dict[str, Any]: api_key resolver the CLI, gateway, and delegate_task all use), so the slot gets its provider's real API surface — e.g. MiniMax → anthropic_messages, GPT-5/o-series → max_completion_tokens, custom endpoints → their base_url. - Returns the kwargs to pass through to ``call_llm`` (provider/model plus the resolved base_url/api_key when available). Falls back to the bare provider/model on any resolution error so a misconfigured slot still attempts the call rather than aborting the whole MoA turn. + + The resolved runtime is cached per (provider, model) with a short TTL + (``_RUNTIME_CACHE_TTL_SECONDS``): the resolution does real I/O (catalog + query + config read) that used to run serially per create() call before + the parallel fan-out could start — the dominant source of MoA cold-start + latency (#66793). The TTL bounds credential staleness (key rotation, + base_url edits) instead of caching for the process lifetime. """ provider = str(slot.get("provider") or "").strip() model = str(slot.get("model") or "").strip() + cache_key = (provider, model) + now = time.monotonic() + with _runtime_cache_lock: + entry = _runtime_cache.get(cache_key) + if entry is not None: + stamped_at, cached = entry + if now - stamped_at < _RUNTIME_CACHE_TTL_SECONDS: + return cached out: dict[str, Any] = {"provider": provider, "model": model} try: from hermes_cli.runtime_provider import resolve_runtime_provider rt = resolve_runtime_provider(requested=provider, target_model=model) - # Forward the resolved endpoint through to call_llm unconditionally. - # call_llm's _resolve_task_provider_model() is the single chokepoint that - # decides whether an explicit base_url collapses a call to the generic - # ``custom`` route or keeps the provider's real identity: it preserves - # identity for any first-class provider (via - # _preserve_provider_with_base_url, a provider-catalog capability check), - # so provider branches that add auth refresh / request metadata / - # request-shape adapters — anthropic OAuth (Bearer + anthropic-beta), - # openai-codex Responses wrapping + Cloudflare headers, xai-oauth, - # bedrock SigV4 signing, nous Portal tags — still fire. Those branches - # re-resolve their own credentials by name and ignore a forwarded - # base_url/api_key, so forwarding is safe even for a placeholder key - # (bedrock's "aws-sdk"). We used to maintain a name-preservation set here - # too; that duplicated the chokepoint and drifted out of sync, so the - # single source of truth now lives in call_llm. if rt.get("base_url"): out["base_url"] = rt["base_url"] if rt.get("api_key"): @@ -362,7 +380,14 @@ def _slot_runtime(slot: dict[str, Any]) -> dict[str, Any]: if isinstance(extra_body, dict) and extra_body: out["extra_body"] = dict(extra_body) except Exception as exc: # pragma: no cover - defensive - logger.debug("MoA slot runtime resolution failed for %s: %s", _slot_label(slot), exc) + logger.debug("MoA slot runtime resolution failed for %s: %s", + _slot_label(slot), exc) + # Never cache a fallback-shaped result: a transient resolution error + # (config mid-write, catalog hiccup) would otherwise pin the bare + # provider/model kwargs for a full TTL. + return out + with _runtime_cache_lock: + _runtime_cache[cache_key] = (now, out) return out @@ -1830,11 +1855,35 @@ class MoAChatCompletions: raise TypeError("_moa_prepared_request must be a dict") return self._call_prepared_aggregator(prepared_request, api_kwargs) - from hermes_cli.config import load_config + from hermes_cli.config import get_config_path, load_config from hermes_cli.moa_config import resolve_moa_preset + # Resolve the preset once per (config st_mtime_ns, preset_name). + # resolve_moa_preset re-normalizes + re-validates the whole moa + # config block on every call, and create() runs once per tool-loop + # iteration — a serial cold-start cost before the parallel fan-out + # can begin (#66793). Keyed on the config FILE's mtime_ns (not a + # config-object attribute, which load_config()'s dicts don't carry), + # so a config edit invalidates on the next call. + try: + _cfg_stamp = get_config_path().stat().st_mtime_ns + except OSError: + _cfg_stamp = None + # load_config() is itself (mtime_ns, size)-cached upstream, so this + # read is cheap; the expensive part this cache skips is + # resolve_moa_preset's re-normalization + re-validation. _moa_raw = load_config().get("moa") or {} - preset = resolve_moa_preset(_moa_raw, self.preset_name) + preset_cache_key = (_cfg_stamp, self.preset_name) + preset = None + if _cfg_stamp is not None: + with _preset_cache_lock: + preset = _preset_cache.get(preset_cache_key) + if preset is None: + preset = resolve_moa_preset(_moa_raw, self.preset_name) + if _cfg_stamp is not None: + with _preset_cache_lock: + _preset_cache.clear() # one live config stamp at a time + _preset_cache[preset_cache_key] = preset # Privacy filter mode: '' (off, default) | 'display' | 'full'. See # coerce_privacy_filter / the pattern block at the top of this module. # Remembered on self so _call_prepared_aggregator (which may run on a diff --git a/agent/model_metadata.py b/agent/model_metadata.py index f89d25f4c8..b1163c76c4 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -145,6 +145,96 @@ _ENDPOINT_MODEL_CACHE_TTL = 300 _ENDPOINT_PROBE_TTL_SECONDS = 3600.0 _endpoint_probe_path_cache: Dict[str, tuple] = {} +# A configured endpoint that is routable-but-dead — e.g. a corp LAN address +# while off-VPN — blackholes TCP: the SYN draws no SYN-ACK, no RST and no ICMP +# error, so a probe waits out its full timeout instead of failing fast. Startup +# runs a whole waterfall of such probes across several functions here, and the +# stalls stack into a minute-long hang before the banner renders. +# +# Once ANY probe has actually observed a connect timeout for an endpoint, the +# others have nothing to gain by repeating it. Recording that observation and +# short-circuiting on it performs no network I/O of its own — it adds no probe +# for callers or tests to mock, and it can only ever fire after a real timeout +# has already been paid, so it cannot suppress a probe that would have worked. +_ENDPOINT_BLACKHOLE_TTL_SECONDS = 30.0 +# Values are monotonic timestamps of the last observed connect timeout. +_endpoint_blackhole_cache: Dict[str, float] = {} + + +def _endpoint_host_key(base_url: str) -> Optional[str]: + """Return a ``host:port`` key for ``base_url``, or None if it has no host. + + Keyed on host:port rather than the full URL so every probe path for one + server — ``/v1``-suffixed or not, LM Studio root or API root — shares a + single entry. + """ + normalized = _normalize_base_url(base_url) + if not normalized: + return None + url = normalized if "://" in normalized else f"http://{normalized}" + try: + parsed = urlparse(url) + host = parsed.hostname + port = parsed.port or (443 if parsed.scheme == "https" else 80) + except Exception: + return None + return f"{host}:{port}" if host else None + + +def _note_endpoint_blackholed(base_url: str) -> None: + """Record that a probe to ``base_url`` timed out during TCP connect.""" + key = _endpoint_host_key(base_url) + if key is None: + return + _endpoint_blackhole_cache[key] = time.monotonic() + logger.debug( + "Endpoint %s timed out connecting — skipping further probes for %.0fs", + key, _ENDPOINT_BLACKHOLE_TTL_SECONDS, + ) + + +def _endpoint_blackholed(base_url: str) -> bool: + """True if a recent probe to ``base_url`` timed out during TCP connect. + + Pure cache lookup; never touches the network. The entry expires after + _ENDPOINT_BLACKHOLE_TTL_SECONDS — long enough to collapse one startup's + burst of probes, short enough that bringing the VPN up mid-session is + picked up without a restart. + """ + if _ENDPOINT_BLACKHOLE_TTL_SECONDS <= 0: + return False + key = _endpoint_host_key(base_url) + if key is None: + return False + seen = _endpoint_blackhole_cache.get(key) + if seen is None: + return False + if (time.monotonic() - seen) >= _ENDPOINT_BLACKHOLE_TTL_SECONDS: + del _endpoint_blackhole_cache[key] + return False + return True + + +def _is_connect_timeout(exc: BaseException) -> bool: + """True for connect-phase timeouts raised by httpx or requests. + + Read timeouts are deliberately excluded: those mean the server accepted + the connection, which is the opposite of the blackhole this guards. + """ + try: + import httpx + if isinstance(exc, httpx.ConnectTimeout): + return True + except Exception: + pass + try: + from requests.exceptions import ConnectTimeout + if isinstance(exc, ConnectTimeout): + return True + except Exception: + pass + return False + # ── Disk L2 for local-endpoint probe results ──────────────────────────────── # The in-process caches above die with the process, so every CLI cold start # with a local model re-paid the probe waterfall in AIAgent.__init__: @@ -382,6 +472,7 @@ DEFAULT_CONTEXT_LENGTHS = { "llama": 131072, # Qwen — specific model families before the catch-all. # Official docs: https://help.aliyun.com/zh/model-studio/developer-reference/ + "qwen3.8-max": 1_000_000, # 1M context (OpenRouter & Nous portal, verified 2026-08-03) "qwen3.6-plus": 1048576, # 1M context (DashScope/Alibaba & OpenRouter) "qwen3.7-plus": 1048576, # 1M context (DashScope/Alibaba) "qwen3-coder-plus": 1000000, # 1M context @@ -836,7 +927,10 @@ def _localhost_to_ipv4(url: str) -> str: ``http://localhost...`` (e.g. ``?upstream=http://localhost:11434``) passes through untouched. """ - if not url: + if not url or not isinstance(url, str): + # Non-string values (test doubles, lazily-resolved config objects) + # previously flowed through these call sites untouched — keep that + # contract; re.sub would raise TypeError. return url return re.sub( r"^(https?://)localhost(?=[:/]|$)", @@ -873,6 +967,13 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: if cached is not None and (time.monotonic() - cached[1]) < _ENDPOINT_PROBE_TTL_SECONDS: return cached[0] + # The host already blackholed a connect: skip the waterfall below, each leg + # of which would otherwise burn its full 2s timeout. Deliberately NOT + # written to _endpoint_probe_path_cache — that entry lives for an hour, + # which would pin the endpoint to "undetected" long after it comes back. + if _endpoint_blackholed(server_url): + return None + # Disk L2: a fresh cross-process verdict skips the HTTP waterfall # entirely (back-to-back CLI invocations, cron ticks). disk_hit = _local_probe_disk_get("server_type", server_url) @@ -882,6 +983,16 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: headers = _auth_headers(api_key) + def _probe_failed(exc: Exception) -> None: + """Swallow a probe error — or abort the waterfall if we were blackholed. + + Re-raising propagates out of the ``with`` block to the outer handler, + so the remaining legs are skipped instead of each stalling in turn. + """ + if _is_connect_timeout(exc): + _note_endpoint_blackholed(server_url) + raise exc + result: Optional[str] = None try: with httpx.Client(timeout=2.0, headers=headers) as client: @@ -890,8 +1001,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: r = client.get(f"{lmstudio_url}/api/v1/models") if r.status_code == 200: result = "lm-studio" - except Exception: - pass + except Exception as exc: + _probe_failed(exc) if result is None: # Ollama exposes /api/tags and responds with {"models": [...]} # LM Studio returns {"error": "Unexpected endpoint"} with status 200 @@ -905,8 +1016,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: result = "ollama" except Exception: pass - except Exception: - pass + except Exception as exc: + _probe_failed(exc) if result is None: # llama.cpp exposes /v1/props (older builds used /props without the /v1 prefix) try: @@ -915,8 +1026,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: r = client.get(f"{server_url}/props") # fallback for older builds if r.status_code == 200 and "default_generation_settings" in r.text: result = "llamacpp" - except Exception: - pass + except Exception as exc: + _probe_failed(exc) if result is None: # vLLM: /version try: @@ -925,8 +1036,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: data = r.json() if "version" in data: result = "vllm" - except Exception: - pass + except Exception as exc: + _probe_failed(exc) except Exception: pass @@ -1120,6 +1231,12 @@ def fetch_endpoint_model_metadata( if cached is not None and (time.time() - cached_at) < _ENDPOINT_MODEL_CACHE_TTL: return cached + # Blackholed endpoint: every candidate below would spend its full 5s + # connect budget. Returned empty rather than cached, so the endpoint is + # retried as soon as the blackhole entry expires. + if _endpoint_blackholed(normalized): + return {} + candidates = [normalized] if normalized.endswith("/v1"): alternate = normalized[:-3].rstrip("/") @@ -1182,9 +1299,20 @@ def fetch_endpoint_model_metadata( return cache except Exception as exc: last_error = exc + if _is_connect_timeout(exc): + _note_endpoint_blackholed(normalized) for candidate in candidates: - url = candidate.rstrip("/") + "/models" + # A connect timeout on one candidate condemns the host, not the path: + # the remaining candidates differ only by URL suffix, so trying them + # would repeat the same stall. + if _endpoint_blackholed(normalized): + break + # normalized/candidates stay unrewritten (cache key stability); only + # the outbound request target is IPv4-resolved to skip the multi-second + # dual-stack IPv6 connect timeout (see _localhost_to_ipv4). + request_candidate = _localhost_to_ipv4(candidate) + url = request_candidate.rstrip("/") + "/models" response = None try: response = requests.get( @@ -1230,7 +1358,7 @@ def fetch_endpoint_model_metadata( if is_llamacpp: try: # Try /v1/props first (current llama.cpp); fall back to /props for older builds - base = candidate.rstrip("/").replace("/v1", "") + base = request_candidate.rstrip("/").replace("/v1", "") _verify = _resolve_requests_verify() props_resp = requests.get(base + "/v1/props", headers=headers, timeout=5, verify=_verify) if not props_resp.ok: @@ -1250,6 +1378,8 @@ def fetch_endpoint_model_metadata( return cache except Exception as exc: last_error = exc + if _is_connect_timeout(exc): + _note_endpoint_blackholed(normalized) finally: if response is not None: response.close() @@ -1813,6 +1943,9 @@ def _query_ollama_api_show_uncached(model: str, base_url: str, api_key: str = "" if server_url.endswith("/v1"): server_url = server_url[:-3] + if _endpoint_blackholed(server_url): + return None + headers = _auth_headers(api_key) try: @@ -1844,8 +1977,9 @@ def _query_ollama_api_show_uncached(model: str, base_url: str, api_key: str = "" return ctx except ValueError: pass - except Exception: - pass + except Exception as exc: + if _is_connect_timeout(exc): + _note_endpoint_blackholed(server_url) return None @@ -1933,6 +2067,9 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str server_url = server_url[:-3] lmstudio_url = _localhost_to_ipv4(_lmstudio_server_root(base_url)) + if _endpoint_blackholed(server_url): + return None + headers = _auth_headers(api_key) try: @@ -2003,13 +2140,39 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str if resp.status_code == 200: data = resp.json() models_list = data.get("data", []) + # Match by id; on single-model servers (e.g. llama.cpp) the + # configured name rarely equals the reported id (a GGUF path), + # so fall back to the sole model when nothing matches. + matched = None for m in models_list: if _model_id_matches(m.get("id", ""), model): - ctx = m.get("max_model_len") or m.get("context_length") or m.get("max_tokens") - if ctx and isinstance(ctx, (int, float)): - return int(ctx) - except Exception: - pass + matched = m + break + if matched is None and len(models_list) == 1: + matched = models_list[0] + if matched is not None: + # llama.cpp nests the runtime context under meta.n_ctx; the + # vLLM/OpenAI keys are also checked. Runtime n_ctx is + # preferred over n_ctx_train (the training maximum, which + # can be larger than what the server actually allocates). + for source in (matched, matched.get("meta") or {}): + if not isinstance(source, dict): + continue + for key in ( + "n_ctx", + "context_length", + "context_window", + "max_model_len", + "max_context_length", + "max_tokens", + "n_ctx_train", + ): + val = source.get(key) + if isinstance(val, (int, float)) and val: + return int(val) + except Exception as exc: + if _is_connect_timeout(exc): + _note_endpoint_blackholed(server_url) return None diff --git a/agent/relay_runtime.py b/agent/relay_runtime.py index 533604791a..a0a7315796 100644 --- a/agent/relay_runtime.py +++ b/agent/relay_runtime.py @@ -482,11 +482,9 @@ class RelayTurnContext: default_factory=threading.RLock, repr=False, ) - _token: contextvars.Token[RelayTurnContext | None] | None = field( - default=None, - repr=False, - ) + _previous_turn: RelayTurnContext | None = field(default=None, repr=False) _active_registered: bool = field(default=False, repr=False) + relay_enabled: bool = True closed: bool = False @@ -600,7 +598,28 @@ class RelaySessionCoordinator: if lease.released: raise RuntimeError("Hermes Relay conversation lease is released") turn = RelayTurnContext(lease=lease, turn_id=turn_id, task_id=task_id) - if isinstance(lease.host, RelayRuntime) and lease.session is not None: + key = (lease.profile_key, lease.session_id) + with self._active_turns_lock: + active = self._active_turns.get(key) + if active: + # A Relay session owns one physical scope stack. Concurrent + # Hermes turns would create sibling scopes on that stack, but + # their completion order is not guaranteed to be LIFO. + turn.relay_enabled = False + logger.warning( + "Skipping Relay instrumentation for concurrent Hermes turn " + "%s in session %s", + turn_id, + lease.session_id, + ) + else: + self._active_turns[key] = {id(turn)} + turn._active_registered = True + if ( + turn.relay_enabled + and isinstance(lease.host, RelayRuntime) + and lease.session is not None + ): try: turn.handle = lease.host.run_in_session( lease.session, @@ -617,11 +636,8 @@ class RelaySessionCoordinator: ) except Exception: logger.warning("Hermes Relay turn initialization failed", exc_info=True) - turn._token = _CURRENT_TURN.set(turn) - key = (lease.profile_key, lease.session_id) - with self._active_turns_lock: - self._active_turns.setdefault(key, set()).add(id(turn)) - turn._active_registered = True + turn._previous_turn = _CURRENT_TURN.get() + _CURRENT_TURN.set(turn) return turn def end_turn( @@ -755,16 +771,18 @@ class RelaySessionCoordinator: @staticmethod def _reset_turn_context(turn: RelayTurnContext) -> None: - """Reset the originating ContextVar token when called in that context.""" - if turn._token is None: + """Unwind ``turn`` without disturbing a newer context-local turn.""" + if _CURRENT_TURN.get() is not turn: return - try: - _CURRENT_TURN.reset(turn._token) - except ValueError: - # A copied async/thread context may own terminal cleanup. Keep the - # token so the originating context can clear its stale reference. - return - turn._token = None + previous = turn._previous_turn + seen = {id(turn)} + while previous is not None and previous.closed: + if id(previous) in seen: + previous = None + break + seen.add(id(previous)) + previous = previous._previous_turn + _CURRENT_TURN.set(previous) @staticmethod def release_conversation(lease: ConversationLease) -> None: @@ -793,10 +811,21 @@ def current_turn() -> RelayTurnContext | None: return _CURRENT_TURN.get() +def relay_instrumentation_enabled() -> bool: + """Return whether this inherited turn may create Relay instrumentation.""" + turn = current_turn() + return turn is None or (turn.relay_enabled and not turn.closed) + + def active_turn(session_id: str | None = None) -> RelayTurnContext | None: """Return a live turn only when it belongs to the active profile/session.""" turn = current_turn() - if turn is None or turn.closed or turn.lease.released: + if ( + turn is None + or not turn.relay_enabled + or turn.closed + or turn.lease.released + ): return None if turn.lease.profile_key != current_profile_key(): return None @@ -814,6 +843,11 @@ def resolve_execution_context( session_id: str, ) -> tuple[RelayRuntime | None, RelaySession | None, Any]: """Resolve one active turn/session parent for managed Relay execution.""" + inherited_turn = current_turn() + if inherited_turn is not None and ( + not inherited_turn.relay_enabled or inherited_turn.closed + ): + return None, None, None turn = active_turn(session_id) if ( turn is not None diff --git a/agent/subdirectory_hints.py b/agent/subdirectory_hints.py index ca96c664cb..4e9f7f5ed3 100644 --- a/agent/subdirectory_hints.py +++ b/agent/subdirectory_hints.py @@ -13,6 +13,7 @@ the conversation without modifying the system prompt (preserving prompt caching) Inspired by Block/goose's SubdirectoryHintTracker. """ +import hashlib import logging import os import shlex @@ -45,6 +46,18 @@ _COMMAND_TOOLS = {"terminal"} # Prevents scanning all the way to / for deeply nested paths. _MAX_ANCESTOR_WALK = 5 +# Directory names that never contain authoritative project context. +# Backups, vendored deps, VCS internals, and caches routinely hold *copies* of +# AGENTS.md; loading those duplicates real context and inflates the prompt. +_EXCLUDED_DIR_NAMES = frozenset({ + "node_modules", "venv", ".venv", "__pycache__", + ".git", ".hg", ".svn", + ".Trash", ".cache", ".tox", ".mypy_cache", ".pytest_cache", + "site-packages", "dist-packages", + "backups", "backup", ".backups", + "vendor", "third_party", +}) + def _is_ancestor_or_same(a: Path, b: Path) -> bool: """Check if *a* is the same as or an ancestor of *b* (parent directory check).""" @@ -54,6 +67,7 @@ def _is_ancestor_or_same(a: Path, b: Path) -> bool: except ValueError: return False + class SubdirectoryHintTracker: """Track which directories the agent visits and load hints on first access. @@ -70,8 +84,34 @@ class SubdirectoryHintTracker: def __init__(self, working_dir: Optional[str] = None): self.working_dir = Path(working_dir or os.getcwd()).resolve() self._loaded_dirs: Set[Path] = set() + # Content digests already injected — prevents re-sending the same file + # reachable through symlinks, hardlinks, or duplicated copies. + self._loaded_digests: Set[str] = set() # Pre-mark the working dir as loaded (startup context handles it) self._loaded_dirs.add(self.working_dir) + self._seed_working_dir_digest() + + def _seed_working_dir_digest(self) -> None: + """Record the CWD context file's digest so it is never re-injected. + + ``prompt_builder`` already loads the working directory's context file at + startup. Seeding its digest here means the same content reached through + a different path (a symlink farm, a shared workspace) is recognised as a + duplicate instead of being sent a second time. + """ + for filename in _HINT_FILENAMES: + candidate = self.working_dir / filename + try: + if not candidate.is_file(): + continue + content = candidate.read_text(encoding="utf-8").strip() + except (OSError, UnicodeDecodeError): + continue + if content: + self._loaded_digests.add( + hashlib.sha256(content.encode("utf-8")).hexdigest() + ) + break # first match wins, mirroring startup loading def check_tool_call( self, @@ -193,8 +233,25 @@ class SubdirectoryHintTracker: # check as a best-effort safeguard. if not _is_ancestor_or_same(self.working_dir, path): return False + if self._is_excluded(path): + return False return True + def _is_excluded(self, path: Path) -> bool: + """True when the path sits inside a directory that holds copies, not context. + + Directories the user is deliberately working inside are never excluded — + if ``working_dir`` is itself under ``vendor/``, that segment is legitimate + and only segments *below* the working dir are screened. + """ + try: + rel_parts = path.relative_to(self.working_dir).parts + except ValueError: + # Paths outside the working dir are already rejected by + # _is_valid_subdir before this runs; treat as excluded defensively. + return True + return any(part in _EXCLUDED_DIR_NAMES for part in rel_parts) + def _load_hints_for_directory(self, directory: Path) -> Optional[str]: """Load hint files from a directory. Returns formatted text or None. @@ -230,6 +287,19 @@ class SubdirectoryHintTracker: content = hint_path.read_text(encoding="utf-8").strip() if not content: continue + # Skip content we've already injected. The same AGENTS.md is + # routinely reachable through several paths (symlinked shared + # workspaces, hardlinks, copied backups); re-sending it burns + # context for zero new information. + digest = hashlib.sha256(content.encode("utf-8")).hexdigest() + if digest in self._loaded_digests: + logger.debug( + "Skipping duplicate hint content at %s (digest %s)", + hint_path, + digest[:12], + ) + break + self._loaded_digests.add(digest) # Same security scan as startup context loading content = _scan_context_content(content, filename) if len(content) > _MAX_HINT_CHARS: diff --git a/agent/system_prompt.py b/agent/system_prompt.py index 8b8832ca28..8407a89689 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -11,14 +11,14 @@ Three tiers are joined with ``\\n\\n``: * ``stable`` — identity (SOUL.md or DEFAULT_AGENT_IDENTITY), tool guidance, computer-use guidance, nous subscription block, tool-use - enforcement guidance + per-model operational guidance, skills prompt, + enforcement guidance + per-model operational guidance, alibaba model-name workaround, environment hints, coding guidance, platform hints. * ``context`` — caller-supplied ``system_message`` plus context files (AGENTS.md / .cursorrules / etc.) discovered under ``TERMINAL_CWD``, plus the session's coding-workspace snapshot. -* ``volatile`` — memory snapshot, USER.md profile, external memory - provider block, timestamp/session/model/provider line. +* ``volatile`` — skills index, memory snapshot, USER.md profile, external + memory provider block, timestamp/session/model/provider line. Pure helpers that read the agent's state. AIAgent keeps thin forwarders. """ @@ -158,8 +158,8 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) * ``context`` — the workspace snapshot followed by the remaining session-stable guidance, context files, and caller-supplied system_message. - * ``volatile`` — memory snapshot, user profile, external - memory provider block, timestamp line. + * ``volatile`` — skills index, memory snapshot, user profile, + external memory provider block, timestamp line. Joined into a single string by :func:`build_system_prompt` and cached on ``agent._cached_system_prompt`` for the lifetime of the @@ -325,8 +325,6 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) ) else: skills_prompt = "" - if skills_prompt: - stable_parts.append(skills_prompt) # Alibaba Coding Plan API always returns "glm-4.7" as model name regardless # of the requested model. Inject explicit model identity into the system prompt @@ -497,8 +495,22 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) if context_files_prompt: context_parts.append(context_files_prompt) - # ── Volatile tier (changes per session/turn — never cached) ─── + # ── Volatile tier (most likely to differ on a rebuild; kept last so the stable prefix stays reusable) ── volatile_parts: List[str] = [] + # Skills are runtime-mutable: the agent adds and patches them across a + # session (SKILLS_GUIDANCE tells it to patch a skill the moment it goes + # stale). The built prompt is cached per session and only rebuilt on + # compaction/restore (see build_system_prompt), so a skill change is not + # byte-stable across rebuilds. With the index in the stable band, a rebuild + # that picked up a skill change would bust the cached prefix from the index + # down, taking the whole scaffold with it. Render it at the FRONT of the + # volatile band instead, ahead of the turn-varying memory/timestamp tail: + # on an implicit longest-prefix backend an unchanged index still falls + # inside the reused prefix, and a changed one only re-prefills from here on. + # (No effect for single-block cache_control backends, where the whole + # system message is one cache unit regardless of internal order.) + if skills_prompt: + volatile_parts.append(skills_prompt) if agent._memory_store: if agent._memory_enabled: @@ -556,10 +568,12 @@ def build_system_prompt(agent: Any, system_message: Optional[str] = None) -> str Layers are ordered cache-friendly: stable identity/guidance first, then session-stable context files, then per-call volatile content - (memory, USER profile, timestamp). The whole string is treated as - one cached block — Hermes never rebuilds or reinjects parts of it - mid-session, which is the only way to keep upstream prompt caches - warm across turns. + (skills index, memory, USER profile, timestamp). For explicit + cache_control backends the whole string is one cached block. For + implicit longest-prefix backends the order is what matters: the + content most likely to change is rendered last, so when the prompt is + rebuilt (on compaction/restore) the unchanged stable scaffold ahead of + the change stays in the reused prefix. """ parts = build_system_prompt_parts(agent, system_message=system_message) joined = "\n\n".join(p for p in (parts["stable"], parts["context"], parts["volatile"]) if p) @@ -602,8 +616,8 @@ def reconstruct_static_prefix( Safety: the rebuilt stable tier is used ONLY when the stored prompt literally starts with it (checked here AND re-checked by ``_apply_system_cache_markers``'s ``startswith`` gate). If any - stable-tier input changed since the prompt was persisted (skills - edited, identity changed), the prefix mismatches, the static stays + stable-tier input changed since the prompt was persisted (identity + changed, SOUL.md edited), the prefix mismatches, the static stays None, and requests fall back to the legacy layout with the stored prompt bytes untouched — never a rewritten prompt. diff --git a/agent/tool_dispatch_helpers.py b/agent/tool_dispatch_helpers.py index f7f003f24a..6e76e52081 100644 --- a/agent/tool_dispatch_helpers.py +++ b/agent/tool_dispatch_helpers.py @@ -48,6 +48,7 @@ _PARALLEL_SAFE_TOOLS = frozenset({ "ha_get_state", "ha_list_entities", "ha_list_services", + "image_generate", "read_file", "search_files", "session_search", diff --git a/agent/tool_executor.py b/agent/tool_executor.py index ee2daa07c4..e1a3f3013d 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -93,6 +93,7 @@ def _budget_for_agent(agent) -> BudgetConfig: # Maximum number of concurrent worker threads for parallel tool execution. # Mirrors the constant in ``run_agent`` for tests/imports that look here. _MAX_TOOL_WORKERS = 8 +_DEFAULT_IMAGE_PARALLEL_REQUESTS = 4 # Keep this above the stock auxiliary.web_extract timeout (360s) so the batch # guard does not preempt a slow-but-valid summarization attempt. _DEFAULT_CONCURRENT_TOOL_TIMEOUT_S = 420.0 @@ -159,6 +160,46 @@ def _flush_session_db_after_tool_progress( return False +def _image_generate_parallel_limit() -> int: + """Return the configured image-generation parallelism cap. + + Image-generation calls are slow enough that concurrent execution is useful, + but backend bursts can hit TTFB or rate-limit failures. Keep the default + intentionally conservative while allowing users to tune it per install. + """ + try: + from hermes_cli.config import load_config + + cfg = load_config() or {} + image_gen = cfg.get("image_gen") if isinstance(cfg, dict) else None + value = ( + image_gen.get("max_parallel_requests") + if isinstance(image_gen, dict) + else None + ) + except Exception: + value = None + + try: + limit = int(value) + except (TypeError, ValueError): + limit = _DEFAULT_IMAGE_PARALLEL_REQUESTS + return max(1, min(limit, _MAX_TOOL_WORKERS)) + + +def _max_workers_for_tool_batch(runnable_calls) -> int: + """Return the worker cap for a concurrent tool batch.""" + if not runnable_calls: + return 0 + max_workers = _MAX_TOOL_WORKERS + if any( + (call[2] if len(call) >= 3 else None) == "image_generate" + for call in runnable_calls + ): + max_workers = min(max_workers, _image_generate_parallel_limit()) + return min(len(runnable_calls), max_workers) + + def _ra(): """Lazy reference to ``run_agent`` so patches like ``run_agent._set_interrupt`` work.""" import run_agent @@ -953,7 +994,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe timeout_s = _resolve_concurrent_tool_timeout() deadline = time.monotonic() + timeout_s if timeout_s is not None else None if runnable_calls: - max_workers = min(len(runnable_calls), _MAX_TOOL_WORKERS) + max_workers = _max_workers_for_tool_batch(runnable_calls) # Daemon workers: an interrupted/timed-out batch is abandoned with # shutdown(wait=False), but stdlib ThreadPoolExecutor workers are # non-daemon and registered in concurrent.futures' atexit hook, diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 07f7673673..2572038126 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -9,6 +9,7 @@ which has provider-specific conditionals for max_tokens defaults, reasoning configuration, temperature handling, and extra_body assembly. """ +import json from typing import Any, Dict from agent.lmstudio_reasoning import resolve_lmstudio_effort @@ -18,6 +19,56 @@ from agent.transports.base import ProviderTransport from agent.transports.types import NormalizedResponse, ToolCall, Usage +def _static_prompt_instructions(messages: list[dict[str, Any]]) -> str: + """Return the stable system/developer prefix used for cache routing. + + Chat Completions carries instructions in its message list rather than a + separate ``instructions`` field. Only a leading system/developer message + is static by contract; later messages are conversation state and must not + split a warm prefix bucket on every turn. + """ + if not messages or not isinstance(messages[0], dict): + return "" + first = messages[0] + if first.get("role") not in {"system", "developer"}: + return "" + content = first.get("content") + if isinstance(content, str): + return content + try: + return json.dumps(content, sort_keys=True, ensure_ascii=False, separators=(",", ":")) + except (TypeError, ValueError): + return str(content or "") + + +def _add_prompt_cache_key( + api_kwargs: dict[str, Any], + *, + messages: list[dict[str, Any]], + tools: list[dict[str, Any]] | None, + supports_prompt_cache_key: bool, +) -> None: + """Add a content-addressed key only for an explicitly capable endpoint.""" + if not supports_prompt_cache_key: + return + + # An explicit caller body field is authoritative too. Do not add a + # duplicate top-level field whose SDK merge precedence could overwrite it. + extra_body = api_kwargs.get("extra_body") + if "prompt_cache_key" in api_kwargs or ( + isinstance(extra_body, dict) and "prompt_cache_key" in extra_body + ): + return + + # Reuse the Responses transport's single authoritative hash algorithm so + # equivalent static prefixes route to the same cache bucket across modes. + from agent.transports.codex import _content_cache_key + + cache_key = _content_cache_key(_static_prompt_instructions(messages), tools) + if cache_key: + api_kwargs["prompt_cache_key"] = cache_key + + def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None: """Return the model's wire-compatible reasoning config.""" if not isinstance(reasoning_config, dict): @@ -112,6 +163,24 @@ def _is_gemini_openai_compat_base_url(base_url: Any) -> bool: return normalized.endswith("/openai") +def _is_openai_api_base_url(base_url: Any) -> bool: + """True only for api.openai.com itself (exact host). + + OpenAI documents ``prompt_cache_key`` as a first-class body field and + GPT-5.6+ docs recommend it for reliable cache routing, so the flag is + implied for the real endpoint. Deliberately NOT a substring match: + Azure OpenAI and strict OpenAI-compat endpoints may reject unknown + fields and must stay opt-in via ``supports_prompt_cache_key``. + """ + try: + from urllib.parse import urlparse + + host = (urlparse(str(base_url or "").strip()).hostname or "").lower() + except Exception: + return False + return host == "api.openai.com" + + def _model_consumes_thought_signature(model: Any) -> bool: """True when the outgoing model is a Gemini family model that requires ``extra_content`` (thought_signature) to be replayed on tool calls. @@ -327,6 +396,8 @@ class ChatCompletionsTransport(ProviderTransport): # Claude on OpenRouter/Nous max output anthropic_max_output: int | None extra_body_additions: dict | None + supports_prompt_cache_key: bool — explicit endpoint capability for + the top-level Chat Completions request field; defaults off. """ # Codex sanitization: drop reasoning_items / call_id / response_item_id. # Pass model so the Gemini thought_signature (extra_content) is kept for @@ -507,6 +578,14 @@ class ChatCompletionsTransport(ProviderTransport): if overrides: api_kwargs.update(overrides) + _add_prompt_cache_key( + api_kwargs, + messages=sanitized, + tools=api_kwargs.get("tools"), + supports_prompt_cache_key=bool(params.get("supports_prompt_cache_key")) + or _is_openai_api_base_url(params.get("base_url")), + ) + return api_kwargs def _build_kwargs_from_profile(self, profile, model, sanitized, tools, params): @@ -649,6 +728,13 @@ class ChatCompletionsTransport(ProviderTransport): if extra_body: api_kwargs["extra_body"] = extra_body + _add_prompt_cache_key( + api_kwargs, + messages=sanitized, + tools=api_kwargs.get("tools"), + supports_prompt_cache_key=bool(profile.supports_prompt_cache_key), + ) + return api_kwargs def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse: diff --git a/agent/transports/codex.py b/agent/transports/codex.py index de71c2b500..c4c901c259 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -28,6 +28,52 @@ def _bounded_prompt_cache_key(value: Any) -> Optional[str]: return f"pck_{digest}" +# Wire-name used when Hermes keeps client-side web_search on xAI Responses. +# A function literally named ``web_search`` collides with Grok's native +# server-side tool (incomplete hang or HTTP 400 duplicate names); this alias +# avoids that while still dispatching through Hermes's configured provider +# (Firecrawl / Tavily / …). Mapped back to ``web_search`` in normalize_response. +_XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search" + + +def _xai_prefers_native_web_search() -> bool: + """True when xAI Responses should use Grok's native ``web_search`` built-in. + + Delegates to the web-search registry's provider resolution (which reads + ``web.search_backend`` / ``web.backend`` from config) and checks whether + the resolved provider is xAI. Falls back to the legacy ``_get_search_backend`` + probe when the registry has no providers loaded. On any resolution failure, + returns True (fail-closed to native — preserves the #48108 incomplete-hang + fix rather than risk reintroducing it). + """ + try: + from agent.web_search_registry import get_active_search_provider + + provider = get_active_search_provider() + if provider is not None: + return getattr(provider, "name", None) == "xai" + + from tools.web_tools import _get_search_backend + + return (_get_search_backend() or "").strip().lower() == "xai" + except Exception: + # Fail closed to native — same behavior as pre-fix main. + return True + + +def _rename_client_web_search_for_xai(response_tools: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """Rename client ``web_search`` → alias so xAI won't hijack it server-side.""" + rewritten: List[Dict[str, Any]] = [] + for tool in response_tools: + if isinstance(tool, dict) and tool.get("name") == "web_search": + aliased = dict(tool) + aliased["name"] = _XAI_CLIENT_WEB_SEARCH_ALIAS + rewritten.append(aliased) + else: + rewritten.append(tool) + return rewritten + + _EXTENDED_PROMPT_CACHE_MODELS = ( "gpt-5.5-pro", "gpt-5.5", @@ -232,63 +278,41 @@ class ResponsesApiTransport(ProviderTransport): response_tools = _responses_tools(tools) - # xAI server-side web search. + # xAI server-side web search vs Hermes web providers. # - # grok models on xAI's /v1/responses surface (notably - # grok-composer-2.5-fast on SuperGrok OAuth) have a *native*, - # server-executed web search. When the model is handed a - # client-side function literally named ``web_search``, it routes - # the intent to that native engine — but because the tool is - # declared as a plain ``function`` rather than xAI's first-class - # ``{"type": "web_search"}`` built-in, the server-side search is - # dispatched but never reconciled: the response streams reasoning - # + ``web_search_call`` progress items, the searches never reach - # ``status="completed"`` in the assembled output, no final - # message is emitted, and ``_normalize_codex_response`` correctly - # sees reasoning-with-no-answer and reports ``incomplete``. The - # turn then burns 3 continuation retries and fails with "Codex - # response remained incomplete after 3 continuation attempts". - # Verified live against grok-composer-2.5-fast (2026-06). + # grok models on xAI's /v1/responses surface have a *native*, + # server-executed web search. A client-side function literally named + # ``web_search`` collides with that engine: declared as a plain + # ``function`` rather than ``{"type": "web_search"}``, the search + # dispatches but never reconciles → incomplete turn + 3 retries. + # Verified live against grok-composer-2.5-fast (2026-06); see #48108. # - # Fix: when the agent HAS a client-side ``web_search`` function (i.e. - # the user enabled the web toolset), declare xAI's native - # ``web_search`` built-in instead so the search actually runs to - # completion server-side and the model streams a real answer. The - # Responses API rejects two tools sharing the name ``web_search`` - # (HTTP 400 "Duplicate tool names"), so we drop the client-side - # ``web_search`` function for the xAI path and let the native tool - # satisfy it. All other client-side tools (read_file, terminal, - # web_extract, MCP tools, …) are untouched and continue to dispatch - # through Hermes's agent loop. + # Two modes, chosen by the user's web-search backend config: # - # Scope: we ONLY swap in the native built-in when the client - # ``web_search`` was actually present. We do NOT force-enable Grok - # server-side search on turns where the user never had web enabled — - # that would silently route around Hermes's web-provider config and - # tool-trace/citation plumbing for every xai-oauth turn. The swap is - # a 1:1 replacement of an already-requested capability, not an - # additive grant. - # - # NOTE: for the swapped case this routes ``web_search`` to Grok's - # native search engine for xAI sessions instead of Hermes's - # configured web provider (Tavily/etc.), and those results bypass - # Hermes's tool-trace / citation plumbing (they arrive baked into the - # model's answer rather than as a tool result the loop observes). - # Scoped to ``is_xai_responses`` deliberately; narrow to specific - # models if a future grok variant should keep the client-side - # function. + # 1. **Native** (active/configured backend is ``xai``, or resolution + # fails): drop the client ``web_search`` function and declare + # xAI's built-in instead. 1:1 swap only when client ``web_search`` + # was already present — never an additive grant. + # 2. **Client** (Firecrawl / Tavily / Exa / … configured or resolved): + # keep Hermes dispatch so ``web.backend`` / ``web.search_backend`` + # is honored, but rename the wire tool to + # ``hermes_web_search`` so Grok cannot hijack the name. The alias + # is mapped back to ``web_search`` in ``normalize_response``. if is_xai_responses and response_tools: has_client_web_search = any( isinstance(t, dict) and t.get("name") == "web_search" for t in response_tools ) if has_client_web_search: - filtered = [ - t for t in response_tools - if not (isinstance(t, dict) and t.get("name") == "web_search") - ] - filtered.append({"type": "web_search"}) - response_tools = filtered + if _xai_prefers_native_web_search(): + filtered = [ + t for t in response_tools + if not (isinstance(t, dict) and t.get("name") == "web_search") + ] + filtered.append({"type": "web_search"}) + response_tools = filtered + else: + response_tools = _rename_client_web_search_for_xai(response_tools) # ``tools`` MUST be omitted entirely when there are no functions to # expose: the openai SDK's ``responses.stream()`` / ``responses.parse()`` @@ -487,9 +511,14 @@ class ResponsesApiTransport(ProviderTransport): provider_data["call_id"] = tc.call_id if hasattr(tc, "response_item_id") and tc.response_item_id: provider_data["response_item_id"] = tc.response_item_id + name = tc.function.name if hasattr(tc, "function") else getattr(tc, "name", "") + # Undo the xAI client-path wire alias so Hermes dispatches + # the real ``web_search`` tool (Firecrawl / etc.). + if name == _XAI_CLIENT_WEB_SEARCH_ALIAS: + name = "web_search" tool_calls.append(ToolCall( - id=tc.id if hasattr(tc, "id") else (tc.function.name if hasattr(tc, "function") else None), - name=tc.function.name if hasattr(tc, "function") else getattr(tc, "name", ""), + id=tc.id if hasattr(tc, "id") else (name or None), + name=name, arguments=tc.function.arguments if hasattr(tc, "function") else getattr(tc, "arguments", "{}"), provider_data=provider_data or None, )) diff --git a/agent/transports/codex_app_server_session.py b/agent/transports/codex_app_server_session.py index e2ace753bc..384a7de8ab 100644 --- a/agent/transports/codex_app_server_session.py +++ b/agent/transports/codex_app_server_session.py @@ -1259,6 +1259,8 @@ def _approval_choice_to_codex_decision(choice: str) -> str: return "accept" if choice in {"session", "always"}: return "acceptForSession" + # "deny" and "timeout" both map to decline — codex has no wire value for + # "prompt expired"; the Hermes-side messaging already distinguishes them. return "decline" diff --git a/agent/turn_context.py b/agent/turn_context.py index e080d6a5d9..def497b158 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -41,6 +41,7 @@ from agent.conversation_compression import ( from agent.context_engine import automatic_compaction_status_message from agent.iteration_budget import IterationBudget from agent.memory_manager import build_memory_context_block +from agent.memory_provider import is_trivial_prompt from agent.model_metadata import ( estimate_messages_tokens_rough, estimate_request_tokens_rough, @@ -1152,11 +1153,15 @@ def build_turn_context( pass # External memory provider: prefetch once before the tool loop. + # + # Skip prefetch on trivial prompts (greetings, acknowledgements) to + # prevent memory-context injection on turns that carry no semantic signal. ext_prefetch_cache = "" if agent._memory_manager: try: _query = original_user_message if isinstance(original_user_message, str) else "" - ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or "" + if not is_trivial_prompt(_query): + ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or "" except Exception: pass diff --git a/apps/desktop/e2e/right-pane.spec.ts b/apps/desktop/e2e/right-pane.spec.ts new file mode 100644 index 0000000000..5447955f45 --- /dev/null +++ b/apps/desktop/e2e/right-pane.spec.ts @@ -0,0 +1,138 @@ +import { test, expect } from './test' + +import { + type MockBackendFixture, + setupMockBackend, + waitForAppReady, +} from './fixtures' + +let fixture: MockBackendFixture | null = null + +test.beforeAll(async () => { + fixture = await setupMockBackend() + await waitForAppReady(fixture, 120_000) +}) + +test.afterAll(async () => { + await fixture?.cleanup() + fixture = null +}) + +test('persistent terminal overlay follows the pane after split dragging', async () => { + const page = fixture!.page + + await page.keyboard.press('Control+`') + await page.locator('[data-terminal-slot]').waitFor({ state: 'visible', timeout: 30_000 }) + await page.locator('[data-persistent-terminal] .xterm').waitFor({ state: 'visible', timeout: 30_000 }) + + const result = await page.evaluate(async () => { + const slot = document.querySelector('[data-terminal-slot]') + const overlay = document.querySelector('[data-persistent-terminal]') + + if (!slot || !overlay) { + return { drift: -1, moved: 0, target: false } + } + + const before = slot.getBoundingClientRect() + const target = [...document.querySelectorAll('[role="separator"]')] + .map(element => { + const box = element.getBoundingClientRect() + const horizontal = box.width > box.height + const center = horizontal + ? (box.top + box.bottom) / 2 + : (box.left + box.right) / 2 + const sides = horizontal + ? [before.top, before.bottom] + : [before.left, before.right] + + return { + element, + box, + horizontal, + score: Math.min(...sides.map(side => Math.abs(center - side))), + } + }) + .filter(item => item.box.width > 0 && item.box.height > 0) + .sort((a, b) => a.score - b.score)[0] + + if (!target) { + return { drift: -1, moved: 0, target: false } + } + + const x = target.box.left + target.box.width / 2 + const y0 = target.box.top + target.box.height / 2 + const nearestSide = target.horizontal + ? Math.abs(y0 - before.top) < Math.abs(y0 - before.bottom) + ? 'top' + : 'bottom' + : Math.abs(x - before.left) < Math.abs(x - before.right) + ? 'left' + : 'right' + const deltaX = nearestSide === 'left' ? -1 : nearestSide === 'right' ? 1 : 0 + const deltaY = nearestSide === 'top' ? -1 : nearestSide === 'bottom' ? 1 : 0 + let currentX = x + let y = y0 + const pointer = { + bubbles: true, + cancelable: true, + pointerId: 71, + pointerType: 'mouse', + isPrimary: true, + button: 0, + buttons: 1, + } + + target.element.dispatchEvent( + new PointerEvent('pointerdown', { ...pointer, clientX: x, clientY: y }), + ) + + for (let index = 0; index < 24; index += 1) { + currentX += deltaX + y += deltaY + window.dispatchEvent( + new PointerEvent('pointermove', { + ...pointer, + clientX: currentX, + clientY: y, + }), + ) + await new Promise(resolve => requestAnimationFrame(() => resolve())) + } + + window.dispatchEvent( + new PointerEvent('pointerup', { + ...pointer, + buttons: 0, + clientX: currentX, + clientY: y, + }), + ) + await new Promise(resolve => setTimeout(resolve, 350)) + await new Promise(resolve => + requestAnimationFrame(() => requestAnimationFrame(() => resolve())), + ) + + const next = slot.getBoundingClientRect() + const fixed = overlay.getBoundingClientRect() + + return { + drift: Math.max( + Math.abs(next.top - fixed.top), + Math.abs(next.left - fixed.left), + Math.abs(next.width - fixed.width), + Math.abs(next.height - fixed.height), + ), + moved: Math.max( + Math.abs(next.top - before.top), + Math.abs(next.left - before.left), + Math.abs(next.width - before.width), + Math.abs(next.height - before.height), + ), + target: true, + } + }) + + expect(result.target).toBe(true) + expect(result.moved).toBeGreaterThan(10) + expect(result.drift).toBeLessThanOrEqual(1) +}) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 834a67ece3..a082e2ba60 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -5410,6 +5410,12 @@ function buildApplicationMenu() { { role: 'cut' }, { role: 'copy' }, { role: 'paste' }, + // ⌘⇧V is only wired up by this item existing: an accelerator with no menu + // entry is never translated into an editor command, so the chord was a + // no-op in every input in the app. The composer inserts plain text on + // every paste anyway, so this is the same result as ⌘V there — it's the + // terminal, preview, and other editable surfaces that need the strip. + { role: 'pasteAndMatchStyle' }, { role: 'delete' }, { role: 'selectAll' } ] @@ -10327,7 +10333,7 @@ ipcMain.handle('hermes:notify', (_event, payload) => { // kind+session can arrive here twice. Collapse it at this single choke point. // Return true (not false): a notification for the event IS being shown by the // first caller, so the settings "send test" success probe stays honest. - if (isDuplicateNotification(`${payload?.kind ?? ''}:${payload?.sessionId ?? ''}`)) { + if (isDuplicateNotification(`${payload?.kind ?? ''}:${payload?.sessionId ?? payload?.tag ?? ''}`)) { return true } @@ -10504,6 +10510,22 @@ ipcMain.handle('hermes:writeClipboard', (_event, text) => { return true }) +// Native save-location picker (profile export etc.) — the write itself happens +// elsewhere (the backend, for profile archives); this only picks the path. +ipcMain.handle('hermes:selectSavePath', async (_event, options: any = {}) => { + const result = await dialog.showSaveDialog(mainWindow, { + title: options?.title || 'Save', + defaultPath: options?.defaultPath ? String(options.defaultPath) : undefined, + filters: Array.isArray(options?.filters) ? options.filters : undefined + }) + + if (result.canceled || !result.filePath) { + return null + } + + return result.filePath +}) + // Paired reader for the GUI terminal's paste chord: the renderer's // navigator.clipboard.readText() throws "Document is not focused" whenever a // portaled overlay has focus, and there's no way to route a read through the diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index 11483d9ce8..dd9537b26f 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -115,6 +115,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { }, readFileText: filePath => ipcRenderer.invoke('hermes:readFileText', filePath), selectPaths: options => ipcRenderer.invoke('hermes:selectPaths', options), + selectSavePath: options => ipcRenderer.invoke('hermes:selectSavePath', options), writeClipboard: text => ipcRenderer.invoke('hermes:writeClipboard', text), readClipboard: () => ipcRenderer.invoke('hermes:readClipboard'), saveImageFromUrl: url => ipcRenderer.invoke('hermes:saveImageFromUrl', url), diff --git a/apps/desktop/scripts/perf/README.md b/apps/desktop/scripts/perf/README.md index 986595f8b2..58c1ca5bc1 100644 --- a/apps/desktop/scripts/perf/README.md +++ b/apps/desktop/scripts/perf/README.md @@ -53,6 +53,7 @@ directly via `window.__PERF_DRIVE__`, so no LLM credits are spent. | `transcript` | ci | large-transcript mount + paint cost | (new) | | `render-churn` | ci | per-component render attribution + store churn while N tabs stream | (new) | | `idle-cost` | report | busy-but-silent tiles: idle commit rate, + fps while resizing / typing | (new) | +| `right-pane` | report | file tree + persistent xterm tabs under chat/terminal output and split dragging | (new) | | `cold-start` | cold | launch → CDP → driver → first paint (fresh spawn/run) | (new) | | `first-token` | backend | Enter → first assistant token painted (TTFT) | (new) | | `submit` | backend | Enter → cleared → user msg painted, scroll jump | measure-submit, measure-jump | diff --git a/apps/desktop/scripts/perf/scenarios/index.mjs b/apps/desktop/scripts/perf/scenarios/index.mjs index d03ad457fd..01ca3be384 100644 --- a/apps/desktop/scripts/perf/scenarios/index.mjs +++ b/apps/desktop/scripts/perf/scenarios/index.mjs @@ -8,6 +8,7 @@ import keystroke from './keystroke.mjs' import multitab from './multitab.mjs' import profileSwitch from './profile-switch.mjs' import renderChurn from './render-churn.mjs' +import rightPane from './right-pane.mjs' import sessionLoad from './session-load.mjs' import sessionSwitch from './session-switch.mjs' import stream from './stream.mjs' @@ -22,6 +23,7 @@ export const SCENARIOS = { [transcript.name]: transcript, [multitab.name]: multitab, [renderChurn.name]: renderChurn, + [rightPane.name]: rightPane, [idleCost.name]: idleCost, [coldStart.name]: coldStart, [firstToken.name]: firstToken, diff --git a/apps/desktop/scripts/perf/scenarios/right-pane.mjs b/apps/desktop/scripts/perf/scenarios/right-pane.mjs new file mode 100644 index 0000000000..411b283c9a --- /dev/null +++ b/apps/desktop/scripts/perf/scenarios/right-pane.mjs @@ -0,0 +1,315 @@ +// File-tree + terminal workspace stress. This is the regression scene for the +// desktop symptom where opening the project tree/terminal made the whole page +// hitch while chat and PTY output continued. +// +// It mounts a real project tree, one PTY plus multiple persistent xterm tabs, +// streams chat and terminal output together, mutates Git decoration state, and +// drags the terminal split. The debug probe records the specific work we care +// about rather than inferring it from CPU alone: +// - fixed-overlay measurements +// - active/hidden xterm fits +// - ProjectTree + per-path row renders +// - frame pacing / slow frames +// +// npm run perf -- right-pane --spawn --prod --runs 3 + +import { dirname, resolve } from 'node:path' +import { fileURLToPath } from 'node:url' + +import { sleep } from '../lib/cdp.mjs' +import { frameHistogram, percentile } from '../lib/stats.mjs' + +const DEFAULT_CWD = resolve(dirname(fileURLToPath(import.meta.url)), '../../..') + +const RECORDERS = ` + (() => { + window.__RP_FRAME_GEN__ = (window.__RP_FRAME_GEN__ || 0) + 1 + const generation = window.__RP_FRAME_GEN__ + window.__RP_FRAMES__ = { times: [], stop: false } + let last = performance.now() + const tick = () => { + if (window.__RP_FRAME_GEN__ !== generation || window.__RP_FRAMES__.stop) return + const now = performance.now() + window.__RP_FRAMES__.times.push(now - last) + last = now + requestAnimationFrame(tick) + } + requestAnimationFrame(tick) + + window.__RP_LONG__ = { entries: [], stop: false } + try { + const observer = new PerformanceObserver(list => { + if (window.__RP_LONG__.stop) return + for (const entry of list.getEntries()) { + window.__RP_LONG__.entries.push({ duration: entry.duration, startTime: entry.startTime }) + } + }) + observer.observe({ entryTypes: ['longtask'] }) + window.__RP_LONG__.observer = observer + } catch {} + return 'armed' + })() +` + +const COLLECT_RECORDERS = ` + (() => { + window.__RP_FRAMES__.stop = true + window.__RP_LONG__.stop = true + try { window.__RP_LONG__.observer && window.__RP_LONG__.observer.disconnect() } catch {} + return JSON.stringify({ frames: window.__RP_FRAMES__.times, longtasks: window.__RP_LONG__.entries }) + })() +` + +const START_COUNTERS = `window.__RIGHT_PANE_PERF__.start(); 'recording'` +const SNAPSHOT_COUNTERS = ` + (() => { + window.__RIGHT_PANE_PERF__.stop() + return JSON.stringify(window.__RIGHT_PANE_PERF__.snapshot()) + })() +` + +const DRAG_TERMINAL_SPLIT = ` + (async () => { + const slot = document.querySelector('[data-terminal-slot]') + const overlay = document.querySelector('[data-persistent-terminal]') + if (!slot || !overlay) return JSON.stringify({ target: 'none', drift: -1, moved: 0 }) + + const slotBox = slot.getBoundingClientRect() + const candidates = [...document.querySelectorAll('[role="separator"]')] + .map(element => ({ element, box: element.getBoundingClientRect() })) + .filter(item => item.box.width > item.box.height * 3) + .sort((a, b) => + Math.abs((a.box.top + a.box.bottom) / 2 - slotBox.top) - + Math.abs((b.box.top + b.box.bottom) / 2 - slotBox.top) + ) + const target = candidates[0] + if (!target) return JSON.stringify({ target: 'none', drift: -1, moved: 0 }) + + const x = target.box.left + target.box.width / 2 + const y0 = target.box.top + target.box.height / 2 + let y = y0 + const pointer = { + bubbles: true, cancelable: true, pointerId: 91, pointerType: 'mouse', + isPrimary: true, button: 0, buttons: 1 + } + target.element.dispatchEvent(new PointerEvent('pointerdown', { ...pointer, clientX: x, clientY: y })) + + for (let i = 0; i < 24; i += 1) { + y -= 1 + window.dispatchEvent(new PointerEvent('pointermove', { ...pointer, clientX: x, clientY: y })) + await new Promise(resolve => requestAnimationFrame(resolve)) + } + for (let i = 0; i < 24; i += 1) { + y += 1 + window.dispatchEvent(new PointerEvent('pointermove', { ...pointer, clientX: x, clientY: y })) + await new Promise(resolve => requestAnimationFrame(resolve)) + } + window.dispatchEvent(new PointerEvent('pointerup', { ...pointer, buttons: 0, clientX: x, clientY: y })) + // Track-size transitions continue briefly after pointerup. Wait through + // that animation, then give the overlay its normal two-frame calibration. + await new Promise(resolve => setTimeout(resolve, 350)) + await new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve))) + + const a = slot.getBoundingClientRect() + const b = overlay.getBoundingClientRect() + const drift = Math.max( + Math.abs(a.top - b.top), + Math.abs(a.left - b.left), + Math.abs(a.width - b.width), + Math.abs(a.height - b.height) + ) + return JSON.stringify({ target: 'horizontal-separator', drift, moved: 24 }) + })() +` + +async function waitFor(cdp, expression, label, timeoutMs = 20000) { + const deadline = Date.now() + timeoutMs + + while (Date.now() < deadline) { + if (await cdp.eval(expression)) { + return + } + + await sleep(100) + } + + throw new Error(`right-pane timed out waiting for ${label}`) +} + +const trimWarmup = (frames, warmupMs = 300) => { + const kept = [] + let elapsed = 0 + + for (const frame of frames) { + elapsed += frame + + if (elapsed >= warmupMs) { + kept.push(frame) + } + } + + return kept +} + +export default { + name: 'right-pane', + tier: 'report', + description: 'Project tree + persistent terminal tabs under chat/terminal output and split dragging.', + async run(cdp, opts = {}) { + const cwd = resolve(String(opts.cwd ?? DEFAULT_CWD)) + const terminalCount = Math.max(2, Number(opts.terminals ?? 3)) + const tokens = Number(opts.tokens ?? 90) + const outputChunks = Number(opts.outputChunks ?? 160) + + await cdp.send('Runtime.enable') + + const ready = await cdp.eval( + `!!(window.__PERF_DRIVE__?.rightPaneSetup && window.__RIGHT_PANE_PERF__ && window.__HERMES_LAYOUT_TREE__)` + ) + + if (!ready) { + throw new Error('right-pane needs a dev renderer or a production build with VITE_PERF_PROBE=1.') + } + + let setup + + try { + setup = await cdp.eval( + `window.__PERF_DRIVE__.rightPaneSetup(${JSON.stringify({ cwd, terminals: terminalCount })})` + ) + await cdp.eval(`window.__HERMES_LAYOUT_TREE__.reveal('files'); window.__HERMES_LAYOUT_TREE__.reveal('terminal')`) + await waitFor(cdp, `!!document.querySelector('[data-project-tree]')`, 'project tree') + await waitFor( + cdp, + `!!document.querySelector('[data-terminal-slot]') && !!document.querySelector('[data-persistent-terminal]')`, + 'persistent terminal' + ) + await waitFor( + cdp, + `document.querySelectorAll('[data-terminal] .xterm').length >= ${terminalCount}`, + `${terminalCount} mounted xterms`, + 30000 + ) + await sleep(1200) + + // Activate every keep-alive tab once, then return to the output tab. + // Each activation should restore exactly one fit; inactive tabs must stay + // at zero even while another tab resizes or writes output. + await cdp.eval(START_COUNTERS) + + for (const id of setup.terminalIds) { + await cdp.eval(`window.__PERF_DRIVE__.rightPaneSelect(${JSON.stringify(id)})`) + await sleep(180) + } + + const activation = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS)) + + await cdp.eval(RECORDERS) + + // Chat DOM churn is deliberately measured in its own counter window: + // terminal positioning should receive no wakeups from transcript changes. + await cdp.eval(START_COUNTERS) + await cdp.eval( + `window.__PERF_DRIVE__.stream({ + chunk: 'Right pane streaming sentence with **bold** and \`code\`.\\n\\n', + intervalMs: 16, + totalTokens: ${tokens}, + flushMinMs: 33 + })` + ) + await cdp.eval(` + (() => { + let n = 0 + window.__RP_OUTPUT_TIMER__ = setInterval(() => { + window.__PERF_DRIVE__.rightPaneWrite( + ${JSON.stringify(setup.procId)}, + 'terminal output line ' + n + ' ........................................\\r\\n' + ) + n += 1 + if (n >= ${outputChunks}) clearInterval(window.__RP_OUTPUT_TIMER__) + }, 16) + return 'writing' + })() + `) + await sleep(Math.max(tokens, outputChunks) * 16 + 900) + const stream = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS)) + + // An unrelated Git status publication should render neither the tree root + // nor any visible row. A status for one visible path should touch only it. + await cdp.eval(START_COUNTERS) + await cdp.eval(`window.__PERF_DRIVE__.rightPaneGit('__right_pane_unrelated__.txt', 'modified')`) + await sleep(250) + const unrelatedGit = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS)) + + const visiblePath = await cdp.eval( + `document.querySelector('[data-project-tree] [title]')?.getAttribute('title') || ''` + ) + let affectedGit = { counts: { 'project-tree-render': 0, 'project-tree-row-render': 0 }, rows: {} } + + if (visiblePath) { + const relative = String(visiblePath).startsWith(`${cwd}/`) + ? String(visiblePath).slice(cwd.length + 1) + : String(visiblePath) + await cdp.eval(START_COUNTERS) + await cdp.eval(`window.__PERF_DRIVE__.rightPaneGit(${JSON.stringify(relative)}, 'modified')`) + await sleep(250) + affectedGit = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS)) + } + + await cdp.eval(START_COUNTERS) + const drag = JSON.parse(await cdp.eval(DRAG_TERMINAL_SPLIT)) + const dragCounters = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS)) + const recorded = JSON.parse(await cdp.eval(COLLECT_RECORDERS)) + const frames = trimWarmup(recorded.frames) + const longtasks = recorded.longtasks.map(entry => entry.duration) + const streamCounts = stream.counts + const activationCounts = activation.counts + const unrelatedCounts = unrelatedGit.counts + const affectedRows = Object.values(affectedGit.rows).reduce((sum, count) => sum + count, 0) + const affectedPaths = Object.keys(affectedGit.rows).length + + if (drag.target === 'none') { + throw new Error('right-pane found no horizontal terminal split separator.') + } + + return { + metrics: { + chat_terminal_measures: streamCounts['terminal-measure'], + hidden_terminal_fits: activationCounts['terminal-fit-hidden'] + streamCounts['terminal-fit-hidden'], + activation_fit_mismatch: Math.abs(activationCounts['terminal-fit-active'] - setup.terminalIds.length), + unrelated_tree_renders: unrelatedCounts['project-tree-render'], + unrelated_row_renders: unrelatedCounts['project-tree-row-render'], + affected_tree_renders: affectedGit.counts['project-tree-render'], + affected_row_path_excess: Math.max(0, affectedPaths - 1), + terminal_drift_px: Math.round(drag.drift * 10) / 10, + frame_p95_ms: Math.round(percentile(frames, 0.95) * 10) / 10, + frame_p99_ms: Math.round(percentile(frames, 0.99) * 10) / 10, + slow_frames_33: frames.filter(frame => frame > 33).length, + longtask_max_ms: Math.round((longtasks.length ? Math.max(...longtasks) : 0) * 10) / 10 + }, + detail: { + cwd, + terminals: setup.terminalIds.length, + activation, + stream, + unrelatedGit, + affectedGit, + affectedRows, + drag, + dragCounters, + frameHistogram: frameHistogram(frames), + frames: frames.length + } + } + } finally { + await cdp.eval(` + (() => { + clearInterval(window.__RP_OUTPUT_TIMER__) + window.__RIGHT_PANE_PERF__?.stop() + window.__PERF_DRIVE__?.reset() + return 'cleaned' + })() + `) + } + } +} diff --git a/apps/desktop/src/app/agents/index.tsx b/apps/desktop/src/app/agents/index.tsx index fe392e8461..eb74a2d2a7 100644 --- a/apps/desktop/src/app/agents/index.tsx +++ b/apps/desktop/src/app/agents/index.tsx @@ -3,6 +3,7 @@ import { type ReactNode, useEffect, useMemo, useState } from 'react' import { useElapsedSeconds } from '@/components/chat/activity-timer' import { ActivityTimerText } from '@/components/chat/activity-timer-text' +import { usePaneVisible } from '@/components/pane-shell/pane-visibility' import { Codicon } from '@/components/ui/codicon' import { FadeText } from '@/components/ui/fade-text' import { GlyphSpinner } from '@/components/ui/glyph-spinner' @@ -189,15 +190,17 @@ function SubagentTree({ tree }: { tree: SubagentNode[] }) { const tokens = flat.reduce((sum, n) => sum + (n.inputTokens ?? 0) + (n.outputTokens ?? 0), 0) const cost = flat.reduce((sum, n) => sum + (n.costUsd ?? 0), 0) + const visible = usePaneVisible() + useEffect(() => { - if (active <= 0 || typeof window === 'undefined') { + if (active <= 0 || !visible || typeof window === 'undefined') { return } const id = window.setInterval(() => setNowMs(Date.now()), 500) return () => window.clearInterval(id) - }, [active]) + }, [active, visible]) if (tree.length === 0) { return ( diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index 3767a4ad4e..28dbcf23a6 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -446,18 +446,12 @@ export function ChatBar({ const handlePaste = (event: ClipboardEvent) => { const imageBlobs = extractClipboardImageBlobs(event.clipboardData) - if (imageBlobs.length > 0) { - event.preventDefault() + if (imageBlobs.length > 0 && onAttachImageBlob) { + triggerHaptic('selection') - if (onAttachImageBlob) { - triggerHaptic('selection') - - for (const blob of imageBlobs) { - void onAttachImageBlob(blob) - } + for (const blob of imageBlobs) { + void onAttachImageBlob(blob) } - - return } // Trim surrounding whitespace so a copy that dragged along leading/trailing @@ -469,6 +463,10 @@ export function ChatBar({ if (!pastedText) { event.preventDefault() + if (imageBlobs.length > 0) { + return + } + // Under WSL2/WSLg the Windows host clipboard doesn't bridge *images* to // the Linux clipboard the DOM paste event reads, so a host screenshot // arrives as an empty paste (no blobs, no text). Fall back to the main diff --git a/apps/desktop/src/app/chat/composer/text-utils.test.ts b/apps/desktop/src/app/chat/composer/text-utils.test.ts index a2e54b2c07..ffca3ba642 100644 --- a/apps/desktop/src/app/chat/composer/text-utils.test.ts +++ b/apps/desktop/src/app/chat/composer/text-utils.test.ts @@ -212,6 +212,47 @@ describe('extractClipboardImageBlobs', () => { expect(extractClipboardImageBlobs(clipboard)).toEqual([image]) }) + + // A rich-text copy (Discord thread, web page, doc) carries prose plus whatever + // inline images the page decorated it with. That is a TEXT paste: attaching the + // page's placeholder graphics as composer images while the text vanished is the + // "blank attachments, no message" bug. + it('ignores inline HTML images when the copy carries its own text', () => { + const clipboard = { + files: { length: 0, item: () => null }, + getData: (type: string) => + type === 'text/html' + ? `

hello from the thread

` + : 'hello from the thread', + items: [] + } as unknown as DataTransfer + + expect(extractClipboardImageBlobs(clipboard)).toEqual([]) + }) + + it('keeps inline HTML images when the copy is image-only', () => { + const clipboard = { + files: { length: 0, item: () => null }, + getData: (type: string) => + type === 'text/html' ? `` : '', + items: [] + } as unknown as DataTransfer + + const blobs = extractClipboardImageBlobs(clipboard) + + expect(blobs).toHaveLength(1) + expect(blobs[0]?.type).toBe('image/png') + }) + + it('drops sub-thumbnail inline images — spacers, trackers, blurhash placeholders', () => { + const clipboard = { + files: { length: 0, item: () => null }, + getData: (type: string) => (type === 'text/html' ? `` : ''), + items: [] + } as unknown as DataTransfer + + expect(extractClipboardImageBlobs(clipboard)).toEqual([]) + }) }) describe('blobDedupeKey', () => { diff --git a/apps/desktop/src/app/chat/composer/text-utils.ts b/apps/desktop/src/app/chat/composer/text-utils.ts index e8828fe890..42af807c39 100644 --- a/apps/desktop/src/app/chat/composer/text-utils.ts +++ b/apps/desktop/src/app/chat/composer/text-utils.ts @@ -70,6 +70,11 @@ const SLASH_INLINE_TRIGGER_RE = /[\s\uFFFC](\/)([a-zA-Z][\w-]*)?$/ // `:` or `:D` smiley doesn't open a popover the user didn't ask for. const EMOJI_TRIGGER_RE = /(?:^|[\s\uFFFC])(:)([a-zA-Z0-9_+-]{2,})$/ +const INLINE_IMAGE_SRC_RE = /]*?\bsrc\s*=\s*["'](data:image\/[^"']+)["']/gi +// Below this, an inline data URL is chrome rather than content — a spacer, a +// 1×1 tracker, or a blurhash placeholder. Real pasted artwork clears it easily. +const MIN_INLINE_IMAGE_BYTES = 4096 + /** Stable key for paste dedupe — `items` and `files` often mirror the same image as different objects. */ export function blobDedupeKey(blob: Blob): string { if (blob instanceof File) { @@ -125,16 +130,22 @@ export function extractClipboardImageBlobs(clipboard: DataTransfer): Blob[] { if (DATA_IMAGE_URL_RE.test(text)) { push(dataUrlToBlob(text)) + + return blobs } - if (blobs.length === 0) { - const html = clipboard.getData('text/html') + // Inline `` in the clipboard's HTML — but only for a copy + // that carried no text of its own. A rich-text copy WITH prose is a text + // paste that happens to contain images, and its data URLs are the page's + // decorations rather than content: Discord ships a 32×5 blurhash placeholder + // beside every image embed, so copying a thread attached a blank thumbnail + // and (because an image paste swallows the event) dropped the text entirely. + if (!text) { + for (const match of clipboard.getData('text/html').matchAll(INLINE_IMAGE_SRC_RE)) { + const blob = dataUrlToBlob(match[1]) - if (html) { - const matches = html.matchAll(/]*?\bsrc\s*=\s*["'](data:image\/[^"']+)["']/gi) - - for (const match of matches) { - push(dataUrlToBlob(match[1])) + if (blob && blob.size >= MIN_INLINE_IMAGE_BYTES) { + push(blob) } } } diff --git a/apps/desktop/src/app/chat/index.test.tsx b/apps/desktop/src/app/chat/index.test.tsx new file mode 100644 index 0000000000..8caace4cc9 --- /dev/null +++ b/apps/desktop/src/app/chat/index.test.tsx @@ -0,0 +1,164 @@ +import { QueryClient, QueryClientProvider } from '@tanstack/react-query' +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { useState } from 'react' +import { MemoryRouter } from 'react-router' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { assistantTextPart, type ChatMessage } from '@/lib/chat-messages' +import { + $activeSessionId, + $awaitingResponse, + $busy, + $contextSuggestions, + $currentCwd, + $currentModel, + $currentProvider, + $freshDraftReady, + $gatewayState, + $messages, + $selectedStoredSessionId, + $sessions +} from '@/store/session' + +const threadRenderCount = vi.hoisted(() => ({ current: 0 })) + +vi.mock('@/components/assistant-ui/thread', async () => { + const React = await import('react') + + return { + Thread: () => { + threadRenderCount.current += 1 + + return React.createElement('div', { 'data-testid': 'thread' }) + } + } +}) + +vi.mock('@/components/Backdrop', async () => { + const React = await import('react') + + return { Backdrop: () => React.createElement('div', { 'data-testid': 'backdrop' }) } +}) + +vi.mock('@/components/prompt-overlays', () => ({ PromptOverlays: () => null })) +vi.mock('@/components/chat/vibe-hearts', () => ({ COMPOSER_HEART_CONFIG: {}, HeartField: () => null })) +vi.mock('@/lib/model-options', () => ({ + modelOptionsQueryKey: (...parts: unknown[]) => ['model-options', ...parts], + requestModelOptions: vi.fn(async () => ({ models: [] })) +})) +vi.mock('./chat-drop-overlay', () => ({ ChatDropOverlay: () => null })) +vi.mock('./chat-swap-overlay', () => ({ ChatSwapOverlay: () => null })) +vi.mock('./composer', () => ({ ChatBar: () => null, ChatBarFallback: () => null })) +vi.mock('./hooks/use-file-drop-zone', () => ({ + useFileDropZone: () => ({ dragKind: null, dropHandlers: {} }) +})) +vi.mock('./sidebar/session-actions-menu', async () => { + const React = await import('react') + + return { + SessionActionsMenu: ({ children }: { children: React.ReactNode }) => + React.createElement('div', { 'data-testid': 'session-actions-menu' }, children) + } +}) + +const { ChatView } = await import('./index') + +function assistantMessage(id: string, text: string): ChatMessage { + return { + id, + parts: [assistantTextPart(text)], + role: 'assistant' + } +} + +describe('ChatView render isolation', () => { + beforeEach(() => { + threadRenderCount.current = 0 + $activeSessionId.set('runtime-1') + $awaitingResponse.set(false) + $busy.set(false) + $contextSuggestions.set([]) + $currentCwd.set('/work') + $currentModel.set('test-model') + $currentProvider.set('test-provider') + $freshDraftReady.set(false) + $gatewayState.set('closed') + $messages.set([assistantMessage('assistant-1', 'Stable historical answer')]) + $selectedStoredSessionId.set('stored-1') + $sessions.set([{ id: 'stored-1', message_count: 1, title: 'Stable chat' } as never]) + }) + + afterEach(() => { + cleanup() + vi.restoreAllMocks() + $activeSessionId.set(null) + $awaitingResponse.set(false) + $busy.set(false) + $contextSuggestions.set([]) + $currentCwd.set('') + $currentModel.set('') + $currentProvider.set('') + $freshDraftReady.set(false) + $gatewayState.set('idle') + $messages.set([]) + $selectedStoredSessionId.set(null) + $sessions.set([]) + }) + + it('does not re-render chat history when an unrelated parent idle tick updates', () => { + const props = { + gateway: null, + maxVoiceRecordingSeconds: 120, + onAddContextRef: vi.fn(), + onAddUrl: vi.fn(), + onAttachDroppedItems: vi.fn(), + onAttachImageBlob: vi.fn(), + onBranchInNewChat: vi.fn(), + onCancel: vi.fn(), + onDeleteSelectedSession: vi.fn(), + onEdit: vi.fn(), + onPasteClipboardImage: vi.fn(), + onPickFiles: vi.fn(), + onPickFolders: vi.fn(), + onPickImages: vi.fn(), + onReload: vi.fn(), + onRemoveAttachment: vi.fn(), + onRetryResume: vi.fn(), + onSteer: vi.fn(), + onSubmit: vi.fn(), + onThreadMessagesChange: vi.fn(), + onToggleSelectedPin: vi.fn(), + onTranscribeAudio: vi.fn() + } + + const queryClient = new QueryClient({ + defaultOptions: { queries: { retry: false } } + }) + + function ParentTickHarness() { + const [tick, setTick] = useState(0) + + return ( + + + + + + + ) + } + + render() + + expect(screen.getByTestId('thread')).toBeTruthy() + expect(threadRenderCount.current).toBe(1) + + fireEvent.click(screen.getByRole('button', { name: /parent tick/i })) + + // memo(ChatView) with stable props must absorb the parent's idle tick — + // the transcript (Thread) must not re-render. This is PR #38470's contract. + expect(threadRenderCount.current).toBe(1) + }) +}) diff --git a/apps/desktop/src/app/chat/index.tsx b/apps/desktop/src/app/chat/index.tsx index 3d72bf3410..1bb98203b5 100644 --- a/apps/desktop/src/app/chat/index.tsx +++ b/apps/desktop/src/app/chat/index.tsx @@ -3,11 +3,12 @@ import { useStore } from '@nanostores/react' import { useQuery } from '@tanstack/react-query' import type { ReadableAtom } from 'nanostores' import type * as React from 'react' -import { Suspense, useCallback, useEffect, useMemo, useState } from 'react' +import { memo, Suspense, useCallback, useEffect, useMemo, useState } from 'react' import { useLocation } from 'react-router' import type { SubmitTextOptions } from '@/app/session/hooks/use-prompt-actions/utils' import { Thread } from '@/components/assistant-ui/thread' +import { TranscriptWindowProvider } from '@/components/assistant-ui/thread/transcript-window' import { Backdrop } from '@/components/Backdrop' import { COMPOSER_HEART_CONFIG, HeartField } from '@/components/chat/vibe-hearts' import { usePaneVisible } from '@/components/pane-shell/pane-visibility' @@ -63,6 +64,7 @@ import { ScrollToBottomButton } from './scroll-to-bottom-button' import { useSessionView } from './session-view' import { SessionActionsMenu } from './sidebar/session-actions-menu' import { threadLoadingState } from './thread-loading' +import { selectTranscriptWindow } from './transcript-window' interface ChatViewProps extends Omit, 'onSubmit'> { gateway: HermesGateway | null @@ -220,9 +222,34 @@ function ChatRuntimeBoundary({ onThreadMessagesChange, suppressMessages }: ChatRuntimeBoundaryProps) { - const storeMessages = useMessagesWhileVisible(useSessionView().$messages) + const view = useSessionView() + const runtimeId = useStore(view.$runtimeId) + const storeMessages = useMessagesWhileVisible(view.$messages) const messages = suppressMessages ? NO_MESSAGES : storeMessages - const runtimeMessageRepository = useRuntimeMessageRepository(messages) + + const [windowPages, setWindowPages] = useState(1) + const [windowSessionKey, setWindowSessionKey] = useState(runtimeId) + + // Reset the window on session swap during RENDER, so a large expand from the + // previous chat can't leak into the next one's first paint (#55191). + if (windowSessionKey !== runtimeId) { + setWindowSessionKey(runtimeId) + setWindowPages(1) + } + + const { messages: windowedMessages, windowed } = useMemo( + () => selectTranscriptWindow(messages, windowPages), + [messages, windowPages] + ) + + const runtimeMessageRepository = useRuntimeMessageRepository(windowedMessages) + + const expandWindow = useCallback(() => setWindowPages(pages => pages + 1), []) + + const transcriptWindow = useMemo( + () => ({ olderAvailable: windowed, expandWindow }), + [expandWindow, windowed] + ) const runtime = useIncrementalExternalStoreRuntime({ messageRepository: runtimeMessageRepository, @@ -237,10 +264,17 @@ function ChatRuntimeBoundary({ onReload }) - return {children} + return ( + + {children} + + ) } -export function ChatView({ +// Memoized: the tile caller (session-tile.tsx) and the contrib surface re-render +// on idle ticks unrelated to the chat; with stable callback props (hoisted to +// useCallback at the call sites) memo() lets the whole chat shell skip those. +export const ChatView = memo(function ChatView({ className, gateway, modelMenuContent, @@ -596,4 +630,4 @@ export function ChatView({ ) -} +}) diff --git a/apps/desktop/src/app/chat/perf-probe.tsx b/apps/desktop/src/app/chat/perf-probe.tsx index 987d89fb42..56b540f8a4 100644 --- a/apps/desktop/src/app/chat/perf-probe.tsx +++ b/apps/desktop/src/app/chat/perf-probe.tsx @@ -1,7 +1,18 @@ import { Profiler, type ProfilerOnRenderCallback, type ReactNode } from 'react' +import { $terminalTakeover, setTerminalTakeover } from '@/app/right-sidebar/store' +import { writeAgentTerminalChunk } from '@/app/right-sidebar/terminal/agent-terminal-stream' +import { + $activeTerminalId, + $terminals, + createTerminal, + ensureAgentTerminal, + selectTerminal, + type TerminalEntry +} from '@/app/right-sidebar/terminal/terminals' +import { $repoStatusByCwd } from '@/store/coding-status' import { $gateway } from '@/store/gateway' -import { $messages, setBusy, setMessages } from '@/store/session' +import { $currentCwd, $messages, setBusy, setCurrentCwdTransient, setMessages } from '@/store/session' type Sample = { id: string @@ -38,6 +49,12 @@ declare global { * backend) doesn't contaminate frame-pacing numbers. */ connected: () => boolean + /** Mount files + multiple xterms for the synthetic right-pane scenario. */ + rightPaneSetup: (opts: { cwd: string; terminals?: number }) => { procId: string; terminalIds: string[] } + rightPaneGit: (path: string, kind?: 'added' | 'conflicted' | 'modified') => void + rightPaneReset: () => void + rightPaneSelect: (id: string) => void + rightPaneWrite: (procId: string, chunk: string) => void reset: () => void snapshotMsgs: () => number } @@ -102,11 +119,32 @@ if (typeof window !== 'undefined' && !window.__PERF_DRIVE__) { let baseline: ReturnType | null = null let activeHandle: SyntheticDriverHandle | null = null + let rightPaneBaseline: null | { + activeTerminalId: null | string + cwd: string + repoStatusByCwd: ReturnType + takeover: boolean + terminals: readonly TerminalEntry[] + } = null + const stop = () => { activeHandle = null setBusy(false) } + const resetRightPane = () => { + if (!rightPaneBaseline) { + return + } + + setTerminalTakeover(rightPaneBaseline.takeover) + $terminals.set(rightPaneBaseline.terminals) + $activeTerminalId.set(rightPaneBaseline.activeTerminalId) + $repoStatusByCwd.set(rightPaneBaseline.repoStatusByCwd) + setCurrentCwdTransient(rightPaneBaseline.cwd) + rightPaneBaseline = null + } + // One synthetic turn's worth of mixed markdown — prose, a list, a fenced // code block, inline code, a link, and a short table — so a loaded transcript // exercises the same render cost (Streamdown blocks, code cards) a real one @@ -166,6 +204,69 @@ if (typeof window !== 'undefined' && !window.__PERF_DRIVE__) { return false } }, + rightPaneGit: (path, kind = 'modified') => { + const file = { + conflicted: kind === 'conflicted', + path, + staged: false, + unstaged: kind === 'modified', + untracked: kind === 'added' + } + + const cwd = $currentCwd.get().trim() + $repoStatusByCwd.set({ + ...$repoStatusByCwd.get(), + [cwd]: { + added: 0, + ahead: 0, + behind: 0, + branch: 'perf', + changed: 1, + conflicted: kind === 'conflicted' ? 1 : 0, + defaultBranch: 'main', + detached: false, + files: [file], + removed: 0, + staged: 0, + unstaged: kind === 'modified' ? 1 : 0, + untracked: kind === 'added' ? 1 : 0 + } + }) + }, + rightPaneReset: resetRightPane, + rightPaneSelect: selectTerminal, + rightPaneSetup: ({ cwd, terminals = 3 }) => { + resetRightPane() + rightPaneBaseline = { + activeTerminalId: $activeTerminalId.get(), + cwd: $currentCwd.get(), + repoStatusByCwd: $repoStatusByCwd.get(), + takeover: $terminalTakeover.get(), + terminals: $terminals.get() + } + + setCurrentCwdTransient(cwd) + const terminalIds = [createTerminal(cwd)] + let procId = '' + + for (let index = 1; index < Math.max(1, terminals); index += 1) { + procId = `right-pane-perf-${Date.now()}-${index}` + const id = ensureAgentTerminal(procId, `perf output ${index}`) + + if (id) { + terminalIds.push(id) + } + } + + if (procId) { + selectTerminal(terminalIds.at(-1) ?? terminalIds[0]) + } + + setTerminalTakeover(true) + + return { procId, terminalIds } + }, + rightPaneWrite: (procId, chunk) => writeAgentTerminalChunk(procId, chunk), loadTranscript: (turns = 200) => { if (!baseline) { baseline = $messages.get() @@ -190,6 +291,7 @@ if (typeof window !== 'undefined' && !window.__PERF_DRIVE__) { }, reset: () => { activeHandle?.stop() + resetRightPane() if (baseline) { setMessages(baseline) diff --git a/apps/desktop/src/app/chat/runtime-repository.test.ts b/apps/desktop/src/app/chat/runtime-repository.test.ts index ae0d4bd469..de7a97d39c 100644 --- a/apps/desktop/src/app/chat/runtime-repository.test.ts +++ b/apps/desktop/src/app/chat/runtime-repository.test.ts @@ -51,4 +51,32 @@ describe('useRuntimeMessageRepository', () => { expect(feedToRepository(result.current).map(item => item.id)).toEqual(['user-1', 'assistant-stream-1', 'user-2']) }) + + it('anchors a branch group to its fork point, and a windowed cut keeps it', () => { + // Branch groups record their fork parent the first time they are seen. A + // window that started mid-group would anchor the survivors to whatever + // preceded them instead — selectTranscriptWindow aligns the cut so the + // whole group arrives together (#55191). + const branch = (id: string): ChatMessage => ({ + ...text(id, 'assistant', 'branch'), + branchGroupId: 'group-1' + }) + + const messages = [text('user-1', 'user', 'hi'), branch('a-1'), branch('a-2'), text('user-2', 'user', 'more')] + + const { result } = renderHook(() => useRuntimeMessageRepository(messages)) + + const parents = new Map(result.current.messages.map(item => [item.message.id, item.parentId])) + + expect(parents.get('a-1')).toBe('user-1') + expect(parents.get('a-2')).toBe('user-1') + + // The same group fed as a window that begins AT the group start keeps the + // fork intact (parent becomes null: the group is now the transcript root). + const { result: windowed } = renderHook(() => useRuntimeMessageRepository(messages.slice(1))) + + const windowedParents = new Map(windowed.current.messages.map(item => [item.message.id, item.parentId])) + + expect(windowedParents.get('a-1')).toBe(windowedParents.get('a-2')) + }) }) diff --git a/apps/desktop/src/app/chat/session-status-dot.tsx b/apps/desktop/src/app/chat/session-status-dot.tsx index 09fb7773a8..c84a1cae0d 100644 --- a/apps/desktop/src/app/chat/session-status-dot.tsx +++ b/apps/desktop/src/app/chat/session-status-dot.tsx @@ -1,5 +1,6 @@ import { useStore } from '@nanostores/react' +import { StatusPulse } from '@/components/ui/status-pulse' import { type Translations, useI18n } from '@/i18n' import { useStoreSelector } from '@/lib/use-session-slice' import { cn } from '@/lib/utils' @@ -17,6 +18,10 @@ import { type SessionDotState, sessionDotState } from './sidebar/session-row-sta type DotVariant = { ariaLabel?: (r: Translations['sidebar']['row']) => string className: string + pulse?: { + className: string + opacity: number + } role?: 'status' title?: (r: Translations['sidebar']['row']) => string } @@ -24,12 +29,6 @@ type DotVariant = { // Shared base for every active dot; idle is smaller and uses its own class. const DOT_BASE = 'relative size-1.5 rounded-full' -// Pseudo-element ping ring that scales outward and fades — shared scaffold for -// the two pulsing dots. The `before:bg-*` color is written inline per variant -// (NOT interpolated here): Tailwind only generates utilities it can see as -// complete static strings, so a `before:bg-${color}` template never emits. -const PING = "before:absolute before:inset-0 before:animate-ping before:rounded-full before:content-['']" - const DOT_VARIANTS: Record = { // Amber steady — a clarify/approval is blocking the turn. Steady (not // pulsing) reads as "your turn", distinct from the accent pulse of a turn. @@ -42,14 +41,22 @@ const DOT_VARIANTS: Record = { // Accent pulse — the LLM turn is actively running. working: { ariaLabel: r => r.sessionRunning, - className: `${DOT_BASE} bg-(--ui-accent) shadow-[0_0_0.625rem_color-mix(in_srgb,var(--ui-accent)_55%,transparent)] ${PING} before:bg-(--ui-accent) before:opacity-70`, + className: `${DOT_BASE} bg-(--ui-accent) shadow-[0_0_0.625rem_color-mix(in_srgb,var(--ui-accent)_55%,transparent)]`, + pulse: { + className: 'absolute inset-0 rounded-full bg-(--ui-accent) opacity-0', + opacity: 0.7 + }, role: 'status' }, // Quiet accent pulse — the turn is still authoritative-running, but no // stream activity has arrived for the watchdog window. stalled: { ariaLabel: r => r.sessionRunning, - className: `${DOT_BASE} bg-(--ui-accent) opacity-70 ${PING} before:bg-(--ui-accent) before:opacity-40`, + className: `${DOT_BASE} bg-(--ui-accent) opacity-70`, + pulse: { + className: 'absolute inset-0 rounded-full bg-(--ui-accent) opacity-0', + opacity: 0.4 + }, role: 'status', title: r => r.sessionRunning }, @@ -58,7 +65,11 @@ const DOT_VARIANTS: Record = { // than muted-foreground so it's visible against the surface. background: { ariaLabel: r => r.backgroundRunning, - className: `${DOT_BASE} bg-muted-foreground/80 ${PING} before:bg-muted-foreground/80 before:opacity-60`, + className: `${DOT_BASE} bg-muted-foreground/80`, + pulse: { + className: 'absolute inset-0 rounded-full bg-muted-foreground/80 opacity-0', + opacity: 0.6 + }, role: 'status', title: r => r.backgroundRunning }, @@ -123,6 +134,7 @@ export function SessionStatusDot({ storedSessionId, session, branchStem, classNa const hasBackground = useStoreSelector($backgroundRunningSessionIds, ids => ids.includes(storedSessionId)) const dotState = sessionDotState({ hasBackground, isStalled, isUnread, isWorking, needsInput }) + const variant = DOT_VARIANTS[dotState] return ( @@ -135,11 +147,20 @@ export function SessionStatusDot({ storedSessionId, session, branchStem, classNa ) diff --git a/apps/desktop/src/app/chat/session-tile.tsx b/apps/desktop/src/app/chat/session-tile.tsx index 27954abc2a..9c30a632f9 100644 --- a/apps/desktop/src/app/chat/session-tile.tsx +++ b/apps/desktop/src/app/chat/session-tile.tsx @@ -104,6 +104,13 @@ function buildTileView(storedSessionId: string): SessionView { } } +// Module-level constants so these ChatView props are referentially stable — +// tiles have no pin/delete affordance, and transcription needs no per-tile state. +const noop = () => undefined + +const tileTranscribeAudio = async (audio: Blob) => + (await transcribeAudio(await blobToDataUrl(audio), audio.type)).transcript + function TileChat({ runtimeId, storedSessionId, @@ -144,6 +151,29 @@ function TileChat({ scope: { add: attachments.add, remove: attachments.remove, target: scope.target } }) + // ChatView is memo()d — every callback prop must be referentially stable or + // the memo never holds and each tile-level render (idle ticks, unrelated + // store updates) re-renders the whole chat shell. The individual composer + // functions are useCallback'd inside useComposerActions, so hoisting these + // wrappers onto them keeps identity stable across renders. + const { addContextRefAttachment, pasteClipboardImage, pickContextPaths, pickImages, removeAttachment } = composer + + const onAddUrl = useCallback( + (url: string) => addContextRefAttachment(`@url:${formatRefValue(url)}`, url), + [addContextRefAttachment] + ) + + const onPasteClipboardImage = useCallback( + (opts?: { silent?: boolean }) => pasteClipboardImage(opts), + [pasteClipboardImage] + ) + + const onPickFiles = useCallback(() => void pickContextPaths('file'), [pickContextPaths]) + const onPickFolders = useCallback(() => void pickContextPaths('folder'), [pickContextPaths]) + const onPickImages = useCallback(() => void pickImages(), [pickImages]) + const onRemoveAttachment = useCallback((id: string) => void removeAttachment(id), [removeAttachment]) + const onRetryResume = useCallback(() => patchSessionTile(storedSessionId, { error: undefined }), [storedSessionId]) + // Per-tile model menu — rendered under this tile's SessionView so the pill // + switch target THIS runtime, not the primary (which may be mid-turn). const modelMenuContent = useMemo( @@ -165,27 +195,27 @@ function TileChat({ composer.addContextRefAttachment(`@url:${formatRefValue(url)}`, url)} + onAddContextRef={addContextRefAttachment} + onAddUrl={onAddUrl} onAttachDroppedItems={composer.attachDroppedItems} onAttachImageBlob={composer.attachImageBlob} onCancel={actions.cancelRun} - onDeleteSelectedSession={() => undefined} + onDeleteSelectedSession={noop} onDismissError={actions.dismissError} onEdit={actions.editMessage} - onPasteClipboardImage={opts => composer.pasteClipboardImage(opts)} - onPickFiles={() => void composer.pickContextPaths('file')} - onPickFolders={() => void composer.pickContextPaths('folder')} - onPickImages={() => void composer.pickImages()} + onPasteClipboardImage={onPasteClipboardImage} + onPickFiles={onPickFiles} + onPickFolders={onPickFolders} + onPickImages={onPickImages} onReload={actions.reloadFromMessage} - onRemoveAttachment={id => void composer.removeAttachment(id)} + onRemoveAttachment={onRemoveAttachment} onRestoreToMessage={actions.restoreToMessage} - onRetryResume={() => patchSessionTile(storedSessionId, { error: undefined })} + onRetryResume={onRetryResume} onSteer={actions.steerPrompt} onSubmit={actions.submitText} onThreadMessagesChange={actions.handleThreadMessagesChange} - onToggleSelectedPin={() => undefined} - onTranscribeAudio={async audio => (await transcribeAudio(await blobToDataUrl(audio), audio.type)).transcript} + onToggleSelectedPin={noop} + onTranscribeAudio={tileTranscribeAudio} /> diff --git a/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx b/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx index dd6988ce91..fea0d8dff0 100644 --- a/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx +++ b/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx @@ -1,6 +1,7 @@ import { useStore } from '@nanostores/react' import { useEffect, useMemo, useState } from 'react' +import { usePaneVisible } from '@/components/pane-shell/pane-visibility' import { ActionsContextMenu, type MenuKit, renderActionItem } from '@/components/ui/actions-menu' import { Codicon } from '@/components/ui/codicon' import { DisclosureCaret } from '@/components/ui/disclosure-caret' @@ -92,17 +93,19 @@ export function SidebarCronJobsSection({ // Rows revealed so far; starts compact, grows in steps via "load more". const [visibleCount, setVisibleCount] = useState(INITIAL_VISIBLE_JOBS) + const visible = usePaneVisible() + // One clock for the whole section (rows are pure) so the countdowns tick - // without re-rendering the rest of the sidebar. Only runs while expanded. + // without re-rendering the rest of the sidebar. Only runs while expanded and visible. useEffect(() => { - if (!open) { + if (!open || !visible) { return } const id = window.setInterval(() => setNowMs(Date.now()), 1000) return () => window.clearInterval(id) - }, [open]) + }, [open, visible]) // Upcoming first (soonest next run), jobs with no next run sink to the bottom, // then alphabetical for stability. @@ -328,6 +331,7 @@ function CronJobSidebarRuns({ jobId, onOpenRun }: { jobId: string; onOpenRun: (s const changeEventsAvailable = useStore($changeEventsAvailable) const cronChangeTick = useStore($cronChangeTick) const [runs, setRuns] = useState(null) + const visible = usePaneVisible() useEffect(() => { let cancelled = false @@ -345,6 +349,15 @@ function CronJobSidebarRuns({ jobId, onOpenRun }: { jobId: string; onOpenRun: (s } }) + // Hidden pane: skip the peek entirely — no initial load, no interval. + // `visible` is in the dep array, so becoming visible re-runs this effect + // and starts the load + timer fresh (same shape as the section clock). + if (!visible) { + return () => { + cancelled = true + } + } + void load() const intervalId = window.setInterval( @@ -361,7 +374,7 @@ function CronJobSidebarRuns({ jobId, onOpenRun }: { jobId: string; onOpenRun: (s window.clearInterval(intervalId) } // cronChangeTick: a fired run reloads the peek immediately. - }, [changeEventsAvailable, cronChangeTick, jobId]) + }, [changeEventsAvailable, cronChangeTick, jobId, visible]) return (
diff --git a/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx b/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx index 1bd5ad3c1d..4a178f15c5 100644 --- a/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx +++ b/apps/desktop/src/app/chat/sidebar/profile-switcher.tsx @@ -59,6 +59,7 @@ import { setShowAllProfiles, sortByProfileOrder } from '@/store/profile' +import { runExportProfileFlow, runImportProfileFlow } from '@/store/profile-share' import type { ProfileInfo } from '@/types/hermes' import { CreateProfileDialog } from '../../profiles/create-profile-dialog' @@ -264,6 +265,7 @@ export function ProfileRail() { profiles={named} /> setCreateOpen(true)} /> +
) : (
setCreateOpen(true)} /> +
)} @@ -435,6 +438,24 @@ function AddProfileButton({ label, onClick }: { label: string; onClick: () => vo ) } +// Import-archive door beside the "+": adopt a shared profile bundle (theme, +// skills, layout) as a new profile. Same chrome as AddProfileButton; the whole +// flow (picker → import → apply overlay → switch) lives in the store. +function ImportProfileButton({ label }: { label: string }) { + return ( + + + + ) +} + // The condensed rail: every named profile in one compact select. The trigger // shows the active profile (tinted initial + name); on default/all scope it // falls back to the placeholder since the left toggle pill carries that state. @@ -692,6 +713,10 @@ function ProfileSquare({ {p.editSoul} + void runExportProfileFlow(label)}> + + {p.exportProfile} + ({ t: { common: { cancel: 'Cancel', close: 'Close', delete: 'Delete', save: 'Save' }, sidebar: { - projects: { menuAppearance: 'Appearance', noColor: 'No color' }, + projects: { + menuAppearance: 'Appearance', + moveFailed: 'Could not move session', + moveNoProjects: 'No other projects', + movedTo: (name: string) => `Moved to ${name}`, + moveToProject: 'Move to project', + noColor: 'No color' + }, row: { archive: 'Archive', branchFrom: 'Branch from here', @@ -50,6 +57,12 @@ vi.mock('@/lib/profile-color', () => ({ PROFILE_SWATCHES: [] })) vi.mock('@/lib/session-export', () => ({ exportSession: vi.fn() })) vi.mock('@/store/gateway', () => ({ activeGateway: vi.fn(() => null) })) vi.mock('@/store/notifications', () => ({ notify: vi.fn(), notifyError: vi.fn() })) +vi.mock('@/store/projects', () => ({ + $projectTree: atom([]), + moveSessionToProject: vi.fn(), + projectIdForCwd: vi.fn(() => null), + projectRootCwd: vi.fn(() => '') +})) vi.mock('@/store/session', () => ({ $activeSessionId: atom(null), $selectedStoredSessionId: atom(null), diff --git a/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx b/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx index 73f741a905..0a32e1881f 100644 --- a/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx +++ b/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx @@ -7,6 +7,7 @@ import { closeAllTreeTabs, closeOtherTreeTabs, closeTreeTabsToRight, + reloadTreePane, treeTabCloseTargets } from '@/components/pane-shell/tree/store' import { @@ -29,6 +30,7 @@ import { PROFILE_SWATCHES } from '@/lib/profile-color' import { exportSession } from '@/lib/session-export' import { activeGateway } from '@/store/gateway' import { notify, notifyError } from '@/store/notifications' +import { $projectTree, moveSessionToProject, projectIdForCwd, projectRootCwd } from '@/store/projects' import { $activeSessionId, $selectedStoredSessionId, @@ -132,6 +134,44 @@ function SessionColorSwatches({ sessionId }: { sessionId: string }) { ) } +// The project list inside the session menu's "Move to project" submenu. Its own +// component so only an OPEN submenu subscribes to the stores (same reasoning as +// SessionColorSwatches). Re-homes the session's workspace at the target +// project's root — the fix for a chat created in the wrong folder. The current +// owner and folderless projects (the Home bucket) are excluded: there is +// nothing to move into. +function MoveToProjectItems({ kit, sessionId, profile }: { kit: MenuKit; sessionId: string; profile?: string }) { + const { t } = useI18n() + const p = t.sidebar.projects + const tree = useStore($projectTree) + const session = useStore($sessions).find(s => sessionMatchesStoredId(s, sessionId)) + const cwd = session?.cwd?.trim() || '' + const currentProjectId = cwd ? projectIdForCwd(cwd) : null + const targets = tree.filter(node => node.id !== currentProjectId && !node.isNoProject && projectRootCwd(node)) + + if (targets.length === 0) { + return {p.moveNoProjects} + } + + return ( + <> + {targets.map(node => ( + { + triggerHaptic('selection') + moveSessionToProject(sessionId, node.id, profile) + .then(() => notify({ durationMs: 2_000, kind: 'success', message: p.movedTo(node.label) })) + .catch(err => notifyError(err, p.moveFailed)) + }} + > + {node.label} + + ))} + + ) +} + function useSessionActions({ sessionId, title, @@ -239,12 +279,24 @@ function useSessionActions({ }) ] - // TAB — close verbs that act on the strip (tabs only; a row isn't a tab). + // TAB — verbs that act on the strip (tabs only; a row isn't a tab). const closeTargets = surface === 'tab' && tabPaneId ? treeTabCloseTargets(tabPaneId) : null - const tabCloseItems: ActionItemSpec[] = + const tabItems: ActionItemSpec[] = surface === 'tab' ? [ + ...(tabPaneId + ? [ + spec({ + icon: 'refresh', + label: t.zones.reload, + onSelect: () => { + triggerHaptic('selection') + reloadTreePane(tabPaneId) + } + }) + ] + : []), ...(onClose ? [ spec({ @@ -342,10 +394,19 @@ function useSessionActions({ /> {workItems.map(item => renderActionItem(kit, item))} - {tabCloseItems.length > 0 && ( + + + + {t.sidebar.projects.moveToProject} + + + + + + {tabItems.length > 0 && ( <> - {tabCloseItems.map(item => renderActionItem(kit, item))} + {tabItems.map(item => renderActionItem(kit, item))} )} diff --git a/apps/desktop/src/app/chat/sidebar/sessions-section.test.tsx b/apps/desktop/src/app/chat/sidebar/sessions-section.test.tsx new file mode 100644 index 0000000000..5fd286e846 --- /dev/null +++ b/apps/desktop/src/app/chat/sidebar/sessions-section.test.tsx @@ -0,0 +1,184 @@ +import { cleanup, render } from '@testing-library/react' +import type * as React from 'react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import type { SessionInfo } from '@/hermes' + +import { SidebarSessionsSection, VIRTUALIZE_THRESHOLD } from './sessions-section' +import type { VirtualSessionListProps } from './virtual-session-list' + +afterEach(cleanup) + +vi.mock('@/i18n', () => ({ + useI18n: () => ({ + t: { + sidebar: { + dateDivider: { + earlierThisMonth: 'Earlier this month', + lastMonth: 'Last month', + lastWeek: 'Last week', + older: 'Older', + today: 'Today', + yesterday: 'Yesterday' + } + } + } + }) +})) + +const mockVirtualListPropsHistory: VirtualSessionListProps[] = [] + +vi.mock('./virtual-session-list', () => ({ + VirtualSessionList: (props: VirtualSessionListProps) => { + mockVirtualListPropsHistory.push(props) + + return
Virtual List ({props.rows.length} rows)
+ } +})) + +vi.mock('./session-row', () => ({ + SidebarSessionRow: ({ session }: { session: SessionInfo }) => ( +
{session.id}
+ ) +})) + +function makeSession(id: string, startedAt = 1000): SessionInfo { + return { + handoff_platform: null, + handoff_state: null, + id, + last_active: startedAt, + profile: 'default', + started_at: startedAt + } as unknown as SessionInfo +} + +function generateSessions(count: number): SessionInfo[] { + return Array.from({ length: count }, (_, i) => makeSession(`session-${i + 1}`, 10000 - i * 100)) +} + +const noop = () => {} + +describe('SidebarSessionsSection memoization & virtualizer stability', () => { + it('memoizes flatRows and passes the exact same rows array reference across parent re-renders', () => { + mockVirtualListPropsHistory.length = 0 + + const sessions = generateSessions(VIRTUALIZE_THRESHOLD + 5) + + const { rerender } = render( + Empty} + label="Sessions" + onArchiveSession={noop} + onDeleteSession={noop} + onResumeSession={noop} + onToggle={noop} + onTogglePin={noop} + open={true} + pinned={false} + sessions={sessions} + workingSessionIdSet={new Set()} + /> + ) + + expect(mockVirtualListPropsHistory.length).toBe(1) + const initialRowsRef = mockVirtualListPropsHistory[0].rows + expect(initialRowsRef.length).toBeGreaterThan(VIRTUALIZE_THRESHOLD) + + // Re-render parent with the exact same sessions array and props + rerender( + Empty} + label="Sessions" + onArchiveSession={noop} + onDeleteSession={noop} + onResumeSession={noop} + onToggle={noop} + onTogglePin={noop} + open={true} + pinned={false} + sessions={sessions} + workingSessionIdSet={new Set()} + /> + ) + + expect(mockVirtualListPropsHistory.length).toBe(2) + const nextRowsRef = mockVirtualListPropsHistory[1].rows + + // Confirm that the flatRows array reference remains strictly identical across renders (useMemo proof) + expect(nextRowsRef).toBe(initialRowsRef) + }) + + it('re-computes flatRows reference when dateGrouped or sessions change', () => { + mockVirtualListPropsHistory.length = 0 + + const initialSessions = generateSessions(VIRTUALIZE_THRESHOLD + 2) + + const { rerender } = render( + Empty} + label="Sessions" + onArchiveSession={noop} + onDeleteSession={noop} + onResumeSession={noop} + onToggle={noop} + onTogglePin={noop} + open={true} + pinned={false} + sessions={initialSessions} + workingSessionIdSet={new Set()} + /> + ) + + const firstRowsRef = mockVirtualListPropsHistory[0].rows + + // Change dateGrouped to true + rerender( + Empty} + label="Sessions" + onArchiveSession={noop} + onDeleteSession={noop} + onResumeSession={noop} + onToggle={noop} + onTogglePin={noop} + open={true} + pinned={false} + sessions={initialSessions} + workingSessionIdSet={new Set()} + /> + ) + + const secondRowsRef = mockVirtualListPropsHistory[1].rows + expect(secondRowsRef).not.toBe(firstRowsRef) + + // Change sessions array identity + const updatedSessions = generateSessions(VIRTUALIZE_THRESHOLD + 4) + rerender( + Empty} + label="Sessions" + onArchiveSession={noop} + onDeleteSession={noop} + onResumeSession={noop} + onToggle={noop} + onTogglePin={noop} + open={true} + pinned={false} + sessions={updatedSessions} + workingSessionIdSet={new Set()} + /> + ) + + const thirdRowsRef = mockVirtualListPropsHistory[2].rows + expect(thirdRowsRef).not.toBe(secondRowsRef) + }) +}) diff --git a/apps/desktop/src/app/chat/sidebar/sessions-section.tsx b/apps/desktop/src/app/chat/sidebar/sessions-section.tsx index 35174a8370..684d217856 100644 --- a/apps/desktop/src/app/chat/sidebar/sessions-section.tsx +++ b/apps/desktop/src/app/chat/sidebar/sessions-section.tsx @@ -1,6 +1,6 @@ import type { useSensors } from '@dnd-kit/core' import type * as React from 'react' -import { useMemo } from 'react' +import { useCallback, useMemo } from 'react' import { SidebarPanelLabel } from '@/app/shell/sidebar-label' import { DisclosureCaret } from '@/components/ui/disclosure-caret' @@ -225,52 +225,79 @@ export function SidebarSessionsSection({ [sessions, preserveInputOrder] ) - const renderRow = (session: SessionInfo, draggable: boolean, branchStem?: string) => { - const rowProps = { - branchStem, - isPinned: pinned, - isSelected: session.id === activeSessionId, - isWorking: workingSessionIdSet.has(session.id), - onArchive: () => onArchiveSession(session.id), - onBranch: onBranchSession ? () => onBranchSession(session.id, session.profile) : undefined, - onDelete: () => onDeleteSession(session.id), - onPin: () => onTogglePin(sessionPinId(session)), - onResume: () => onResumeSession(session.id), - reorderable: draggable && !branchStem, - session, - showProfile: showProfileTags - } + const renderRow = useCallback( + (session: SessionInfo, draggable: boolean, branchStem?: string) => { + const rowProps = { + branchStem, + isPinned: pinned, + isSelected: session.id === activeSessionId, + isWorking: workingSessionIdSet.has(session.id), + onArchive: () => onArchiveSession(session.id), + onBranch: onBranchSession ? () => onBranchSession(session.id, session.profile) : undefined, + onDelete: () => onDeleteSession(session.id), + onPin: () => onTogglePin(sessionPinId(session)), + onResume: () => onResumeSession(session.id), + reorderable: draggable && !branchStem, + session, + showProfile: showProfileTags + } - return draggable && !branchStem ? ( - - ) : ( - - ) - } + return draggable && !branchStem ? ( + + ) : ( + + ) + }, + [ + activeSessionId, + onArchiveSession, + onBranchSession, + onDeleteSession, + onResumeSession, + onTogglePin, + pinned, + showProfileTags, + workingSessionIdSet + ] + ) // A single flat/virtual/lane list row — either a date divider or a session. - const renderListRow = (row: SidebarListRow, draggable: boolean) => - row.kind === 'divider' ? ( - - ) : ( - renderRow(row.entry.session, draggable, row.entry.branchStem) - ) + const renderListRow = useCallback( + (row: SidebarListRow, draggable: boolean) => + row.kind === 'divider' ? ( + + ) : ( + renderRow(row.entry.session, draggable, row.entry.branchStem) + ), + [dividerLabels, renderRow] + ) // Sessions inside repos/worktrees are date-ordered and static. - const renderRows = (items: SessionInfo[]) => - flattenSessionsWithBranches(items).map(({ branchStem, session }) => renderRow(session, false, branchStem)) + const renderRows = useCallback( + (items: SessionInfo[]) => + flattenSessionsWithBranches(items).map(({ branchStem, session }) => renderRow(session, false, branchStem)), + [renderRow] + ) // Same as `renderRows`, but with date dividers folded in — used for // entered-project lanes so a lane spanning multiple days reads // chronologically, matching the flat recents list. - const renderRowsDated = (items: SessionInfo[]) => { - const entries = flattenSessionsWithBranches(items) + const renderRowsDated = useCallback( + (items: SessionInfo[]) => { + const entries = flattenSessionsWithBranches(items) - return (dateGrouped ? groupEntriesByRecency(entries) : toSessionRows(entries)).map(row => renderListRow(row, false)) - } + return (dateGrouped ? groupEntriesByRecency(entries) : toSessionRows(entries)).map(row => + renderListRow(row, false) + ) + }, + [dateGrouped, renderListRow] + ) // Flat recents as list rows: grouped by recency when enabled, plain otherwise. - const flatRows: SidebarListRow[] = dateGrouped ? groupEntriesByRecency(displayEntries) : toSessionRows(displayEntries) + const flatRows: SidebarListRow[] = useMemo( + () => (dateGrouped ? groupEntriesByRecency(displayEntries) : toSessionRows(displayEntries)), + [dateGrouped, displayEntries] + ) const flatVirtualized = !showEmptyState && diff --git a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx index d9cfd6c004..90e10ddc38 100644 --- a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx +++ b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx @@ -27,7 +27,7 @@ interface SessionRowCommonProps { showProfile?: boolean } -interface VirtualSessionListProps { +export interface VirtualSessionListProps { activeSessionId: null | string className?: string rows: SidebarListRow[] diff --git a/apps/desktop/src/app/chat/transcript-window.test.ts b/apps/desktop/src/app/chat/transcript-window.test.ts new file mode 100644 index 0000000000..4726724f68 --- /dev/null +++ b/apps/desktop/src/app/chat/transcript-window.test.ts @@ -0,0 +1,144 @@ +import { describe, expect, it } from 'vitest' + +import type { ChatMessage } from '@/lib/chat-messages' +import { RENDER_WEIGHT_CHARS } from '@/lib/render-weight' + +import { + alignToBranchGroup, + selectTranscriptWindow, + TRANSCRIPT_WINDOW_BUDGET, + TRANSCRIPT_WINDOW_MIN_MESSAGES +} from './transcript-window' + +const message = (id: string, chars: number, branchGroupId?: string): ChatMessage => ({ + id, + parts: [{ type: 'text', text: 'x'.repeat(chars) }], + role: id.startsWith('u') ? 'user' : 'assistant', + ...(branchGroupId ? { branchGroupId } : {}) +}) + +/** Messages of `chars` each, newest last. */ +const transcript = (count: number, chars: number): ChatMessage[] => + Array.from({ length: count }, (_, i) => message(`m-${i}`, chars)) + +describe('selectTranscriptWindow', () => { + it('does not window a transcript that fits the budget', () => { + const messages = transcript(50, 100) + + const window = selectTranscriptWindow(messages) + + expect(window.windowed).toBe(false) + // Reference identity preserved — a fresh array would re-render the runtime. + expect(window.messages).toBe(messages) + }) + + it('windows a HEAVY-but-SHORT transcript that a message-count cap would miss', () => { + // 40 messages, each a big tool result. Well under any sane count cap, but + // this is the shape that exhausts the renderer heap (#55191). + const messages = transcript(40, RENDER_WEIGHT_CHARS * 400) + + const window = selectTranscriptWindow(messages) + + expect(window.windowed).toBe(true) + expect(window.messages.length).toBeLessThan(messages.length) + expect(window.messages.at(-1)).toBe(messages.at(-1)) + }) + + it('keeps far MORE messages when they are light than when they are heavy', () => { + // The contract is weight, not count: a message-count cap would treat these + // two identically. 500 tiny messages are cheaper than 500 tool results, so + // many more of them survive the same budget. + const light = selectTranscriptWindow(transcript(500, 20)) + const heavy = selectTranscriptWindow(transcript(500, RENDER_WEIGHT_CHARS * 40)) + + expect(light.messages.length).toBeGreaterThan(heavy.messages.length * 10) + }) + + it('leaves a long transcript whole when the whole thing is cheap', () => { + const messages = transcript(600, 20) + + const window = selectTranscriptWindow(messages) + + expect(window.windowed).toBe(false) + expect(window.messages).toBe(messages) + }) + + it('keeps a floor of messages when single turns are enormous', () => { + const messages = transcript(80, RENDER_WEIGHT_CHARS * TRANSCRIPT_WINDOW_BUDGET) + + const window = selectTranscriptWindow(messages) + + expect(window.messages.length).toBeGreaterThanOrEqual(TRANSCRIPT_WINDOW_MIN_MESSAGES) + }) + + it('grows by one budget page per expand and eventually covers everything', () => { + const messages = transcript(400, RENDER_WEIGHT_CHARS * 40) + + const first = selectTranscriptWindow(messages, 1) + const second = selectTranscriptWindow(messages, 2) + + expect(first.windowed).toBe(true) + expect(second.messages.length).toBeGreaterThan(first.messages.length) + + let pages = 1 + let window = selectTranscriptWindow(messages, pages) + + while (window.windowed && pages < 100) { + window = selectTranscriptWindow(messages, ++pages) + } + + // Paging terminates at the full transcript — never a dead end. + expect(window.windowed).toBe(false) + expect(window.messages).toHaveLength(messages.length) + }) + + it('never cuts inside a branch group, so branches keep their fork point', () => { + const heavy = RENDER_WEIGHT_CHARS * 200 + + // A branch group sits right where a weight-only cut would land. + const messages: ChatMessage[] = [ + ...transcript(20, heavy), + message('a-branch-1', heavy, 'group-1'), + message('a-branch-2', heavy, 'group-1'), + message('a-branch-3', heavy, 'group-1'), + ...transcript(20, heavy).map(m => ({ ...m, id: `tail-${m.id}` })) + ] + + for (let pages = 1; pages <= 6; pages++) { + const kept = selectTranscriptWindow(messages, pages).messages + const groupMembers = kept.filter(m => m.branchGroupId === 'group-1') + + // Either the whole group survives or none of it does — never a partial + // group, which would re-parent the surviving branches. + expect([0, 3]).toContain(groupMembers.length) + } + }) + + it('handles an empty transcript', () => { + const messages: ChatMessage[] = [] + + expect(selectTranscriptWindow(messages)).toEqual({ messages, windowed: false }) + }) +}) + +describe('alignToBranchGroup', () => { + const messages = [ + message('u-1', 10), + message('a-1', 10, 'g'), + message('a-2', 10, 'g'), + message('u-2', 10) + ] + + it('widens a cut that lands mid-group back to the group start', () => { + expect(alignToBranchGroup(messages, 2)).toBe(1) + }) + + it('leaves a cut on a non-branch message alone', () => { + expect(alignToBranchGroup(messages, 3)).toBe(3) + }) + + it('clamps out-of-range indices', () => { + expect(alignToBranchGroup(messages, -5)).toBe(0) + expect(alignToBranchGroup(messages, 99)).toBe(messages.length) + }) +}) diff --git a/apps/desktop/src/app/chat/transcript-window.ts b/apps/desktop/src/app/chat/transcript-window.ts new file mode 100644 index 0000000000..29d2d1c9d2 --- /dev/null +++ b/apps/desktop/src/app/chat/transcript-window.ts @@ -0,0 +1,111 @@ +import type { ChatMessage } from '@/lib/chat-messages' +import { messageRenderWeight } from '@/lib/render-weight' + +/** + * Bound what reaches assistant-ui at all. + * + * Rendering the full transcript of an oversized session rebuilds an unbounded + * runtime repository on every store update and exhausts the renderer's V8 heap + * (#55191). The DOM budget in `thread/list.tsx` already bounds what PAINTS, but + * every message still gets normalized into the repository first — so a session + * only has to be heavy, not visible, to crash the window. + * + * The window spends the same currency as the DOM budget: render weight, not + * message count. A count cap gets this wrong in both directions — measured on a + * real 1,175-session store, a 400-message cap would disable itself on 37 + * sessions that are heavy but short (one was 133 messages / 1.05MB) while + * engaging on 92 long-but-light sessions that were never at risk. + */ + +/** + * One window page, in render-weight units. + * + * Four DOM pages (the `RENDER_BUDGET` of 300 in `thread/list.tsx`). "Show + * earlier" spends the DOM budget first, so the user pages through the + * already-materialized window three times before this asks the store for more + * — and the reported crash shape (~231K tokens ≈ 2,260 units) is windowed + * rather than handed to the repository whole. + */ +export const TRANSCRIPT_WINDOW_BUDGET = 1200 + +/** + * Floor on messages kept regardless of weight. A transcript of enormous turns + * must still render the turn the user is having; without this a single + * multi-megabyte tool result could window everything after it away. + */ +export const TRANSCRIPT_WINDOW_MIN_MESSAGES = 30 + +export interface TranscriptWindow { + /** The tail assistant-ui is allowed to materialize. */ + messages: ChatMessage[] + /** Store holds older messages than this window. */ + windowed: boolean +} + +/** + * Widen a cut backwards so it never lands inside an assistant branch group. + * + * `useRuntimeMessageRepository` records a group's fork point the first time it + * sees the group (`branchParentByGroup`). A cut through the middle of a group + * therefore anchors the surviving branches to whatever message happens to + * precede them in the window — silently re-parenting a branch. Include the + * whole group or none of it. + */ +export function alignToBranchGroup(messages: readonly ChatMessage[], start: number): number { + if (start <= 0 || start >= messages.length) { + return Math.max(0, Math.min(start, messages.length)) + } + + const group = messages[start].branchGroupId + + if (!group) { + return start + } + + let aligned = start + + while (aligned > 0 && messages[aligned - 1].branchGroupId === group) { + aligned-- + } + + return aligned +} + +/** + * Select the tail of the transcript that fits one window, grown by `pages`. + * + * Walks newest-first accumulating weight until the budget is met, keeps at + * least MIN messages, then aligns the cut off a branch-group boundary. + */ +export function selectTranscriptWindow( + messages: readonly ChatMessage[], + pages = 1 +): TranscriptWindow { + const budget = TRANSCRIPT_WINDOW_BUDGET * Math.max(1, Math.floor(pages)) + + if (messages.length === 0) { + return { messages: messages as ChatMessage[], windowed: false } + } + + let start = messages.length + let weight = 0 + + for (let i = messages.length - 1; i >= 0; i--) { + weight += messageRenderWeight(messages[i].parts) + start = i + + if (weight >= budget && messages.length - i >= TRANSCRIPT_WINDOW_MIN_MESSAGES) { + break + } + } + + start = alignToBranchGroup(messages, start) + + if (start <= 0) { + // Preserve reference identity when the whole transcript fits: handing React + // a fresh array of the same messages re-renders the runtime for nothing. + return { messages: messages as ChatMessage[], windowed: false } + } + + return { messages: messages.slice(start), windowed: true } +} diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index da4ff828cb..b643a31c1b 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -39,7 +39,7 @@ import { useContributions } from '@/contrib/react/use-contributions' import { registry } from '@/contrib/registry' import { discoverRuntimePlugins } from '@/contrib/runtime-loader' import { sessionTitle as storedSessionTitle } from '@/lib/chat-runtime' -import { FileText, LayoutDashboard, PanelBottom, Terminal, Zap } from '@/lib/icons' +import { Download, FileText, LayoutDashboard, PanelBottom, Terminal, Upload, Zap } from '@/lib/icons' import { type KeybindContribution, KEYBINDS_AREA } from '@/lib/keybinds/actions' import { setYoloEnabled } from '@/lib/yolo-session' import { pruneComposerPopoutZones } from '@/store/composer-popout' @@ -56,6 +56,7 @@ import { SIDEBAR_MAX_WIDTH } from '@/store/layout' import { $previewOpenRequest, $previewTabs, closeRightRail } from '@/store/preview' +import { runExportProfileFlow, runImportProfileFlow } from '@/store/profile-share' import { $reviewOpen, closeReview, openReview, REVIEW_PANE_ID } from '@/store/review' import { $currentCwd, $selectedStoredSessionId, $sessions, $yoloActive, sessionMatchesStoredId } from '@/store/session' import { watchSessionPins } from '@/store/session-pin-sync' @@ -314,6 +315,31 @@ registry.registerMany([ keywords: ['keybinds', 'shortcuts', 'hotkeys', 'keyboard'], run: () => window.dispatchEvent(new CustomEvent('hermes:open-keybinds')) } satisfies PaletteContribution + }, + // Profile sharing: bundle the active profile (config, skills, theme, layout) + // into a portable archive, or adopt someone else's. Both open native dialogs, + // so the palette closing on select is correct. + { + id: 'profile.export', + area: PALETTE_AREA, + data: { + id: 'profile.export', + label: 'Export profile…', + icon: Upload, + keywords: ['profile', 'export', 'share', 'bundle', 'theme', 'settings', 'backup'], + run: () => void runExportProfileFlow() + } satisfies PaletteContribution + }, + { + id: 'profile.import', + area: PALETTE_AREA, + data: { + id: 'profile.import', + label: 'Import profile…', + icon: Download, + keywords: ['profile', 'import', 'share', 'bundle', 'archive', 'restore'], + run: () => void runImportProfileFlow() + } satisfies PaletteContribution } ]) diff --git a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts index f522e71035..1297690584 100644 --- a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts +++ b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts @@ -77,7 +77,8 @@ describe('useSessionTileDelegate resumeTile', () => { expect(requestGateway).toHaveBeenCalledWith('session.resume', { session_id: 'stored-x', cols: 96, - profile: 'ai-engineer' + profile: 'ai-engineer', + omit_messages: true }) }) @@ -94,7 +95,8 @@ describe('useSessionTileDelegate resumeTile', () => { expect(requestGateway).toHaveBeenCalledWith('session.resume', { session_id: 'stored-y', cols: 96, - profile: 'default' + profile: 'default', + omit_messages: true }) }) }) diff --git a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts index ba1107e9e0..57bfa92bfa 100644 --- a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts +++ b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts @@ -79,6 +79,7 @@ export function useSessionTileDelegate({ requestGateway('session.resume', { session_id: storedSessionId, cols: 96, + omit_messages: true, ...(profile ? { profile } : {}) }) ]) diff --git a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts index f02ee69f5e..99ccc0564c 100644 --- a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts +++ b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts @@ -5,6 +5,7 @@ import type { HermesConnection } from '@/global' import { HermesGateway } from '@/hermes' import { translateNow } from '@/i18n' import { desktopDefaultCwd } from '@/lib/desktop-fs' +import { reconnectBackoffDelayMs } from '@/lib/reconnect-backoff' import { $desktopBoot, applyDesktopBootProgress, @@ -42,12 +43,16 @@ import type { RpcEvent } from '@/types/hermes' import { stashGatewaySurvivor, survivorIsStale, takeGatewaySurvivor } from './gateway-hmr-survivor' -// After this many consecutive failed reconnects (≈45s with the 1→15s backoff) -// raise a recoverable boot error. Otherwise a dropped remote gateway loops the -// backoff forever behind the fullscreen CONNECTING overlay with no way to reach -// Settings / sign in / switch to local — the "lost connection breaks the app" -// dead end. The next successful reconnect clears it. -const RECONNECT_ESCALATE_AFTER = 6 +// After the reconnect loop has been failing for this long, raise a recoverable +// boot error. Otherwise a dropped remote gateway loops the backoff forever +// behind the fullscreen CONNECTING overlay with no way to reach Settings / +// sign in / switch to local — the "lost connection breaks the app" dead end. +// The next successful reconnect clears it. Time-based (not attempt-count) +// because the full-jitter backoff makes attempt counts a meaningless clock: +// six jittered attempts can elapse in ~9s, while the old deterministic +// 1→15s ladder took ~45s to reach six failures — this threshold keeps that +// original ~45s calibration. +const RECONNECT_ESCALATE_AFTER_MS = 45_000 interface GatewayBootOptions { beforeConnectionSwitch: () => void @@ -114,13 +119,18 @@ export function useGatewayBoot({ let reconnecting = false let reconnectTimer: ReturnType | null = null let reconnectAttempt = 0 + // Wall-clock start of the current disconnect episode (first failed + // reconnect attempt); null while healthy. Drives the time-based + // escalation below. Reset on a clean open or a manual/wake reconnect. + let reconnectFailingSince: number | null = null // Surface "sign in again" once per disconnect episode, not on every backoff // tick — a stale OAuth ticket fails every attempt and would otherwise stack // identical error toasts (and their haptics). Reset on the next clean open. let reauthNotified = false - // Raised once the reconnect loop crosses RECONNECT_ESCALATE_AFTER so the - // recovery overlay replaces the dead-end CONNECTING screen. Reset on a clean - // open or a manual/wake-driven reconnect. + // Raised once the reconnect loop has been failing for + // RECONNECT_ESCALATE_AFTER_MS so the recovery overlay replaces the + // dead-end CONNECTING screen. Reset on a clean open or a manual/ + // wake-driven reconnect. let escalated = false // Wrap the live getter in a call so TS control-flow analysis doesn't narrow @@ -173,6 +183,7 @@ export function useGatewayBoot({ } reconnectAttempt = 0 + reconnectFailingSince = null // A respawned backend re-mints (recycles) runtime ids, so any tile's // bound runtime id is now stale — drop them so each tile re-resumes. resetTileRuntimeBindings() @@ -192,7 +203,11 @@ export function useGatewayBoot({ reconnecting = false if (!cancelled && !gatewayOpen() && !$gatewaySwitching.get()) { - if (reconnectAttempt >= RECONNECT_ESCALATE_AFTER && !escalated) { + if (reconnectFailingSince === null) { + reconnectFailingSince = Date.now() + } + + if (Date.now() - reconnectFailingSince >= RECONNECT_ESCALATE_AFTER_MS && !escalated) { escalated = true failDesktopBoot(translateNow('boot.errors.gatewayConnectionLost')) } @@ -207,8 +222,11 @@ export function useGatewayBoot({ return } - // 1s, 2s, 4s … capped at 15s. - const delay = Math.min(15_000, 1_000 * 2 ** Math.min(reconnectAttempt, 4)) + // Full-jitter exponential backoff (300ms base, 15s cap) so a gateway + // restart doesn't get redialed by every desktop client in lockstep — + // an immediate-retry reconnect storm can exhaust the gateway's file + // descriptors while it's still coming back up. + const delay = reconnectBackoffDelayMs(reconnectAttempt) reconnectAttempt += 1 reconnectTimer = setTimeout(() => { reconnectTimer = null @@ -223,6 +241,7 @@ export function useGatewayBoot({ clearReconnectTimer() reconnectAttempt = 0 + reconnectFailingSince = null escalated = false reconnectSecondaryGateways() @@ -269,6 +288,7 @@ export function useGatewayBoot({ $gatewaySwitching.set(true) clearReconnectTimer() reconnectAttempt = 0 + reconnectFailingSince = null escalated = false reauthNotified = false callbacksRef.current.beforeConnectionSwitch() @@ -371,6 +391,7 @@ export function useGatewayBoot({ if (st === 'open') { reconnectAttempt = 0 + reconnectFailingSince = null reauthNotified = false escalated = false clearReconnectTimer() diff --git a/apps/desktop/src/app/hooks/use-keybinds.ts b/apps/desktop/src/app/hooks/use-keybinds.ts index 32d9caaba2..fa41cad238 100644 --- a/apps/desktop/src/app/hooks/use-keybinds.ts +++ b/apps/desktop/src/app/hooks/use-keybinds.ts @@ -42,7 +42,7 @@ import { } from '@/store/profile' import { openFolderAsProject, requestNewWorktree } from '@/store/projects' import { toggleReview } from '@/store/review' -import { setModelPickerOpen } from '@/store/session' +import { $selectedStoredSessionId, setModelPickerOpen } from '@/store/session' import { reopenLastClosedTile } from '@/store/session-states' import { $switcherOpen, @@ -62,11 +62,13 @@ import { useTheme } from '@/themes/context' import { requestComposerFocus, requestModelMenuToggle, requestVoiceToggle } from '../chat/composer/focus' import { openSession } from '../open-session' import { + $workspaceIsPage, AGENTS_ROUTE, ARTIFACTS_ROUTE, CRON_ROUTE, MESSAGING_ROUTE, navigateToWorkspacePage, + NEW_CHAT_ROUTE, PROFILES_ROUTE, sessionRoute, SETTINGS_ROUTE, @@ -99,11 +101,29 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { const profileSwitchHandlers: HandlerMap = {} + // A tab key that lands on the WORKSPACE tab while a full page (skills / + // messaging / artifacts / a plugin route) covers it must also route back to + // the chat: the workspace pane is already the zone's active tab behind the + // page, so fronting it alone changes nothing on screen and the key reads + // dead. Mirrors `openSession`'s full-page rule — only a route change puts + // the chat back. + const leavePageForWorkspaceChat = (paneId: null | string) => { + if (paneId === 'workspace' && $workspaceIsPage.get()) { + const selected = $selectedStoredSessionId.get() + + navigate(selected ? sessionRoute(selected) : NEW_CHAT_ROUTE) + } + } + for (let slot = 1; slot <= PROFILE_SLOT_COUNT; slot += 1) { // ⌘1…⌘9 switch the FOCUSED zone's tab when it's a real tab strip; only a // single-pane (or unfocused) layout falls through to the profile switch. profileSwitchHandlers[`profile.switch.${slot}`] = () => { - if (!activateTreeTabSlot(slot)) { + const pane = activateTreeTabSlot(slot) + + if (pane) { + leavePageForWorkspaceChat(pane) + } else { switchProfileToSlot(slot) } } @@ -132,6 +152,19 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { goToSession(openOrAdvanceSwitcher(direction)) } + // ⌃Tab cycles the focused session/main tab strip; only a non-tabbed focus + // falls through to the recent-session switcher. Landing on the workspace + // under a full page routes back to the chat (same as ⌘1). + const cycleTab = (direction: 1 | -1) => { + const pane = cycleTreeTabInFocusedZone(direction) + + if (pane) { + leavePageForWorkspaceChat(pane) + } else { + stepSession(direction) + } + } + const showFiles = () => { setFileBrowserOpen(true) setTerminalTakeover(false) @@ -170,10 +203,8 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { }, 'session.newTab': () => deps.openNewSessionTab(), 'session.newWindow': () => void openNewWindow(), - // ⌃Tab cycles the focused session/main tab strip; only a non-tabbed focus - // falls through to the recent-session switcher. - 'session.next': () => void (cycleTreeTabInFocusedZone(1) || stepSession(1)), - 'session.prev': () => void (cycleTreeTabInFocusedZone(-1) || stepSession(-1)), + 'session.next': () => cycleTab(1), + 'session.prev': () => cycleTab(-1), ...sessionSlotHandlers, 'session.focusSearch': requestSessionSearchFocus, 'session.togglePin': deps.toggleSelectedPin, diff --git a/apps/desktop/src/app/right-sidebar/files/tree-sizing.test.ts b/apps/desktop/src/app/right-sidebar/files/tree-sizing.test.ts new file mode 100644 index 0000000000..04e8a8bb47 --- /dev/null +++ b/apps/desktop/src/app/right-sidebar/files/tree-sizing.test.ts @@ -0,0 +1,29 @@ +import { describe, expect, it, vi } from 'vitest' + +import { projectTreeViewportSize } from './tree' + +describe('projectTreeViewportSize', () => { + it('uses ResizeObserver contentRect without forcing another layout read', () => { + const element = document.createElement('div') + const getBoundingClientRect = vi.spyOn(element, 'getBoundingClientRect') + const contentRect = { height: 480, width: 320 } as DOMRectReadOnly + + expect( + projectTreeViewportSize([{ contentRect, target: element } as unknown as ResizeObserverEntry], element) + ).toEqual({ + height: 480, + width: 320 + }) + expect(getBoundingClientRect).not.toHaveBeenCalled() + }) + + it('falls back to a rect when ResizeObserver is unavailable', () => { + const element = document.createElement('div') + vi.spyOn(element, 'getBoundingClientRect').mockReturnValue({ + height: 240, + width: 160 + } as DOMRect) + + expect(projectTreeViewportSize([], element)).toEqual({ height: 240, width: 160 }) + }) +}) diff --git a/apps/desktop/src/app/right-sidebar/files/tree.tsx b/apps/desktop/src/app/right-sidebar/files/tree.tsx index 97d96f267b..b20bfcaf4b 100644 --- a/apps/desktop/src/app/right-sidebar/files/tree.tsx +++ b/apps/desktop/src/app/right-sidebar/files/tree.tsx @@ -1,12 +1,14 @@ import { useStore } from '@nanostores/react' import { type KeyboardEvent as ReactKeyboardEvent, useCallback, useEffect, useRef, useState } from 'react' +import { useMemo } from 'react' import { type NodeApi, type NodeRendererProps, type RowRendererProps, Tree, type TreeApi } from 'react-arborist' import { TreeSkeleton } from '@/components/chat/skeletons' import { Codicon } from '@/components/ui/codicon' +import { markRightPanePerf } from '@/debug/right-pane-events' import { useResizeObserver } from '@/hooks/use-resize-observer' import { cn } from '@/lib/utils' -import { $repoChangeByPath, type RepoChangeKind } from '@/store/coding-status' +import { type RepoChangeKind, repoChangeKindForPath } from '@/store/coding-status' import { $renamingPath, beginInlineRename } from '@/store/file-actions' import { $revealInTreeRequest } from '@/store/layout' @@ -55,19 +57,20 @@ export function ProjectTree({ onPreviewFile, openState }: ProjectTreeProps) { + markRightPanePerf('project-tree-render') + const containerRef = useRef(null) const treeRef = useRef | null>(null) const [size, setSize] = useState({ height: 0, width: 0 }) - const changeByPath = useStore($repoChangeByPath) - const syncTreeSize = useCallback(() => { + const syncTreeSize = useCallback((entries: readonly ResizeObserverEntry[]) => { const el = containerRef.current if (!el) { return } - const { height, width } = el.getBoundingClientRect() + const { height, width } = projectTreeViewportSize(entries, el) setSize(prev => { if (prev.height === height && prev.width === width) { @@ -175,7 +178,12 @@ export function ProjectTree({ }, []) return ( -
+
{size.height > 0 && size.width > 0 ? ( childrenAccessor={node => (node?.isDirectory ? (node.children ?? []) : null)} @@ -200,7 +208,6 @@ export function ProjectTree({ {props => ( item.target === element) + const box = entry?.contentRect ?? element.getBoundingClientRect() + + return { height: box.height, width: box.width } +} + function TreeSizingState() { return } @@ -244,7 +261,6 @@ const CHANGE_TINT: Record = { } function ProjectTreeRow({ - changeKind, dragHandle, node, onAttachFile, @@ -253,13 +269,17 @@ function ProjectTreeRow({ relativeTo, style }: NodeRendererProps & { - changeKind?: RepoChangeKind onAttachFile: (path: string) => void onAttachFolder: (path: string) => void onPreviewFile?: (path: string) => void relativeTo?: null | string }) { const renamingPath = useStore($renamingPath) + const path = node.data?.id ?? '' + const changeStore = useMemo(() => repoChangeKindForPath(path), [path]) + const changeKind: RepoChangeKind | undefined = useStore(changeStore) + + markRightPanePerf('project-tree-row-render', path) if (!node.data) { return
diff --git a/apps/desktop/src/app/right-sidebar/terminal/active-resize.test.ts b/apps/desktop/src/app/right-sidebar/terminal/active-resize.test.ts new file mode 100644 index 0000000000..27909c2cb0 --- /dev/null +++ b/apps/desktop/src/app/right-sidebar/terminal/active-resize.test.ts @@ -0,0 +1,159 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { observeActiveTerminalResize } from './active-resize' + +afterEach(() => { + vi.unstubAllGlobals() +}) + +function installRaf() { + let nextId = 1 + const frames = new Map() + + vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => { + const id = nextId++ + frames.set(id, callback) + + return id + }) + vi.stubGlobal('cancelAnimationFrame', (id: number) => frames.delete(id)) + + return { + flush() { + const pending = [...frames.entries()] + frames.clear() + pending.forEach(([, callback]) => callback(0)) + }, + pending: () => frames.size + } +} + +describe('observeActiveTerminalResize', () => { + it('fits once on activation and coalesces later resize bursts', () => { + const raf = installRaf() + const resize = { current: null as ResizeObserverCallback | null } + const disconnect = vi.fn() + + vi.stubGlobal( + 'ResizeObserver', + class { + constructor(callback: ResizeObserverCallback) { + resize.current = callback + } + + disconnect = disconnect + observe = vi.fn((target: Element) => { + resize.current?.([{ target } as ResizeObserverEntry], this as unknown as ResizeObserver) + }) + unobserve = vi.fn() + } as unknown as typeof ResizeObserver + ) + + const onFit = vi.fn() + const onActivate = vi.fn() + const host = document.createElement('div') + const dispose = observeActiveTerminalResize(host, { onActivate, onFit }) + + // ResizeObserver's initial delivery is absorbed by the activation frame. + expect(raf.pending()).toBe(1) + raf.flush() + expect(onFit).toHaveBeenCalledTimes(1) + expect(onActivate).toHaveBeenCalledTimes(1) + + resize.current?.([], {} as ResizeObserver) + resize.current?.([], {} as ResizeObserver) + resize.current?.([], {} as ResizeObserver) + expect(raf.pending()).toBe(1) + raf.flush() + expect(onFit).toHaveBeenCalledTimes(2) + + dispose() + expect(disconnect).toHaveBeenCalledTimes(1) + }) + + it('cancels activation without fitting when hidden before the first frame', () => { + const raf = installRaf() + + vi.stubGlobal( + 'ResizeObserver', + class { + disconnect = vi.fn() + observe = vi.fn() + unobserve = vi.fn() + } as unknown as typeof ResizeObserver + ) + + const onFit = vi.fn() + + const dispose = observeActiveTerminalResize(document.createElement('div'), { + onActivate: vi.fn(), + onFit + }) + + dispose() + + raf.flush() + expect(onFit).not.toHaveBeenCalled() + }) + + it('absorbs a real browser-style initial resize delivered after activation', () => { + const raf = installRaf() + const resize = { current: null as ResizeObserverCallback | null } + + vi.stubGlobal( + 'ResizeObserver', + class { + constructor(callback: ResizeObserverCallback) { + resize.current = callback + } + + disconnect = vi.fn() + observe = vi.fn() + unobserve = vi.fn() + } as unknown as typeof ResizeObserver + ) + + const onFit = vi.fn() + observeActiveTerminalResize(document.createElement('div'), { onActivate: vi.fn(), onFit }) + + raf.flush() + expect(onFit).toHaveBeenCalledTimes(1) + + // Browser initial delivery: the activation fit already covered this size. + resize.current?.([], {} as ResizeObserver) + expect(raf.pending()).toBe(0) + + // A later real resize schedules exactly one fit. + resize.current?.([], {} as ResizeObserver) + expect(raf.pending()).toBe(1) + raf.flush() + expect(onFit).toHaveBeenCalledTimes(2) + }) + + it('reuses a first-mount fit without fitting again on activation', () => { + const raf = installRaf() + + vi.stubGlobal( + 'ResizeObserver', + class { + disconnect = vi.fn() + observe = vi.fn() + unobserve = vi.fn() + } as unknown as typeof ResizeObserver + ) + + const onActivate = vi.fn() + const onFit = vi.fn() + + observeActiveTerminalResize(document.createElement('div'), { + fitOnActivate: false, + onActivate, + onFit + }) + + raf.flush() + + expect(onActivate).toHaveBeenCalledOnce() + expect(onFit).not.toHaveBeenCalled() + }) +}) diff --git a/apps/desktop/src/app/right-sidebar/terminal/active-resize.ts b/apps/desktop/src/app/right-sidebar/terminal/active-resize.ts new file mode 100644 index 0000000000..ea68ceda1d --- /dev/null +++ b/apps/desktop/src/app/right-sidebar/terminal/active-resize.ts @@ -0,0 +1,78 @@ +interface ActiveTerminalResizeOptions { + fitOnActivate?: boolean + onActivate: () => void + onFit: () => void +} + +/** + * Observe one visible xterm host. + * + * Inactive terminals never call this helper, so their preserved DOM/PTY stays + * mounted without paying for ResizeObserver delivery or FitAddon work. The + * first frame owns activation and ignores the observer's initial delivery; + * later resize bursts are coalesced to one fit per animation frame. + */ +export function observeActiveTerminalResize( + host: HTMLElement, + { fitOnActivate = true, onActivate, onFit }: ActiveTerminalResizeOptions +): () => void { + let activated = false + let frame = 0 + let initialResizeDelivered = false + let stopped = false + + const scheduleFit = () => { + if (!activated || stopped || frame !== 0) { + return + } + + frame = window.requestAnimationFrame(() => { + frame = 0 + + if (!stopped) { + onFit() + } + }) + } + + const observer = new ResizeObserver(() => { + // ResizeObserver's initial delivery is asynchronous in browsers and may + // arrive before OR after the activation rAF. Activation already fits the + // current box, so absorb that first delivery in either ordering. + if (!initialResizeDelivered) { + initialResizeDelivered = true + + return + } + + scheduleFit() + }) + + observer.observe(host) + + frame = window.requestAnimationFrame(() => { + frame = 0 + + if (stopped) { + return + } + + activated = true + + if (fitOnActivate) { + onFit() + } + + onActivate() + }) + + return () => { + stopped = true + observer.disconnect() + + if (frame !== 0) { + window.cancelAnimationFrame(frame) + frame = 0 + } + } +} diff --git a/apps/desktop/src/app/right-sidebar/terminal/persistent.test.tsx b/apps/desktop/src/app/right-sidebar/terminal/persistent.test.tsx index c0f718e546..77eeb480d6 100644 --- a/apps/desktop/src/app/right-sidebar/terminal/persistent.test.tsx +++ b/apps/desktop/src/app/right-sidebar/terminal/persistent.test.tsx @@ -2,6 +2,8 @@ import { act, type ReactNode } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { $paneStates } from '@/store/panes' + import { PersistentTerminal, TerminalSlot } from './persistent' vi.mock('./terminals', () => ({ @@ -212,7 +214,8 @@ describe('PersistentTerminal rect tracking', () => { render() - expect(mutationObserveCalls.some(call => call.options?.subtree === true)).toBe(true) + expect(mutationObserveCalls.length).toBeGreaterThan(0) + expect(mutationObserveCalls.every(call => call.options?.subtree === false)).toBe(true) act(() => { raf.runNext() @@ -244,6 +247,27 @@ describe('PersistentTerminal rect tracking', () => { expect(raf.pending()).toBe(0) }) + it('remeasures from an explicit pane-layout state change', () => { + const raf = installRaf() + const before = $paneStates.get() + vi.spyOn(HTMLElement.prototype, 'getBoundingClientRect').mockReturnValue(rect(10, 20, 200, 100)) + + render() + raf.runNext() + expect(raf.pending()).toBe(0) + + act(() => { + $paneStates.set({ ...before, __terminal_rect_test__: { open: true } }) + }) + + expect(raf.pending()).toBe(1) + + act(() => { + raf.runNext() + $paneStates.set(before) + }) + }) + it('does not schedule rect RAFs while the Electron window is paused, then resumes when visible', () => { const raf = installRaf() vi.spyOn(HTMLElement.prototype, 'getBoundingClientRect').mockReturnValue(rect(10, 20, 200, 100)) diff --git a/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx b/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx index b5e75978b0..d3a7b0c829 100644 --- a/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx +++ b/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx @@ -2,7 +2,10 @@ import { useStore } from '@nanostores/react' import { atom } from 'nanostores' import { type CSSProperties, useEffect, useLayoutEffect, useRef, useState } from 'react' +import { $layoutTree } from '@/components/pane-shell/tree/store' +import { markRightPanePerf } from '@/debug/right-pane-events' import { createRendererLoopPauseController } from '@/lib/renderer-loop-pause' +import { $paneStates } from '@/store/panes' import { $terminalTakeover } from '../store' @@ -39,7 +42,7 @@ export function TerminalSlot({ className = SLOT_CLASS }: { className?: string }) } }, []) - return
+ return
} interface PersistentTerminalProps { @@ -86,6 +89,7 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP let prev: Rect | null = null let frame = 0 let stopped = false + let pendingReason = 'initial' let pauseController: ReturnType | null = null const rendererPaused = () => pauseController?.isPaused() ?? document.visibilityState === 'hidden' @@ -97,11 +101,12 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP } } - const measure = (): boolean => { + const measure = (reason: string): boolean => { if (rendererPaused()) { return false } + markRightPanePerf('terminal-measure', reason) const r = slot.getBoundingClientRect() // floor top/left + ceil right/bottom: overlay always covers the slot's // full pixel footprint, so half-pixel rects can't leak page bg through. @@ -123,16 +128,18 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP return false } - const scheduleMeasure = () => { + const scheduleMeasure = (reason = 'unknown') => { if (stopped || rendererPaused() || frame !== 0) { return } + pendingReason = reason frame = window.requestAnimationFrame(() => { frame = 0 + const reason = pendingReason - if (measure()) { - scheduleMeasure() + if (measure(reason)) { + scheduleMeasure('settle') } }) } @@ -144,50 +151,69 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP return } - scheduleMeasure() + scheduleMeasure('visibility') } const observer = typeof ResizeObserver === 'undefined' ? null : new ResizeObserver(() => { - scheduleMeasure() + scheduleMeasure('resize-observer') }) const positionObserver = typeof MutationObserver === 'undefined' ? null : new MutationObserver(() => { - scheduleMeasure() + scheduleMeasure('ancestor-mutation') }) pauseController = createRendererLoopPauseController(handleVisibilityChange) - if (measure()) { - scheduleMeasure() + if (measure('initial')) { + scheduleMeasure('settle') } observer?.observe(slot) + const handleScroll = () => scheduleMeasure('scroll') + const scrollTargets: Array = [window] + window.addEventListener('scroll', handleScroll) + for (let node: HTMLElement | null = slot; node; node = node.parentElement) { positionObserver?.observe(node, { attributeFilter: ['class', 'style', 'hidden', 'aria-hidden', 'data-state'], attributes: true, childList: true, - subtree: true + subtree: false }) + // Scroll does not bubble. Listen only on the slot's own ancestor chain, + // so a transcript/file-tree/xterm viewport scroll elsewhere cannot wake + // terminal positioning. + node.addEventListener('scroll', handleScroll) + scrollTargets.push(node) } - window.addEventListener('resize', scheduleMeasure) - window.addEventListener('scroll', scheduleMeasure, true) + // Nested layout-tree and pane-state commits can move the slot without + // changing its own size. Subscribe to the actual layout authorities instead + // of observing every descendant mutation under every ancestor (chat stream + // and file-tree updates are unrelated and used to wake this tracker). + const unsubscribeLayout = $layoutTree.listen(() => scheduleMeasure('layout-tree')) + const unsubscribePanes = $paneStates.listen(() => scheduleMeasure('pane-state')) + + const handleResize = () => scheduleMeasure('window-resize') + + window.addEventListener('resize', handleResize) return () => { stopped = true cancelFrame() observer?.disconnect() positionObserver?.disconnect() - window.removeEventListener('resize', scheduleMeasure) - window.removeEventListener('scroll', scheduleMeasure, true) + unsubscribeLayout() + unsubscribePanes() + window.removeEventListener('resize', handleResize) + scrollTargets.forEach(target => target.removeEventListener('scroll', handleScroll)) pauseController?.dispose() } }, [slot]) @@ -215,7 +241,7 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP // booting xterm/node-pty at 0×0 starts the shell at 80×24 and spawns a visible // conhost on Windows. After that `mounted` latches: shells persist while hidden. return ( -
+
{mounted && }
) diff --git a/apps/desktop/src/app/right-sidebar/terminal/use-agent-terminal.ts b/apps/desktop/src/app/right-sidebar/terminal/use-agent-terminal.ts index c0947eb0d9..339d70144b 100644 --- a/apps/desktop/src/app/right-sidebar/terminal/use-agent-terminal.ts +++ b/apps/desktop/src/app/right-sidebar/terminal/use-agent-terminal.ts @@ -5,9 +5,11 @@ import { Terminal } from '@xterm/xterm' import { useEffect, useRef } from 'react' import { writeClipboardText } from '@/components/ui/copy-button' +import { markRightPanePerf } from '@/debug/right-pane-events' import { triggerHaptic } from '@/lib/haptics' import { useTheme } from '@/themes/context' +import { observeActiveTerminalResize } from './active-resize' import { registerAgentTerminalWriter } from './agent-terminal-stream' import { makeTerminalReader, registerTerminalReader } from './buffer' import { mirrorSelection, terminalClipboardIntent } from './clipboard' @@ -24,7 +26,8 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id: const hostRef = useRef(null) const termRef = useRef(null) const webglRef = useRef(null) - const fitRef = useRef<(() => void) | null>(null) + const fitRef = useRef<((visible: boolean) => void) | null>(null) + const initialActiveFitRef = useRef(false) const { latestFontFamilyRef, mountedRef } = useTerminalFontController({ fitRef, termRef, webglRef }) const surfaceTheme = () => { @@ -47,7 +50,6 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id: } let disposed = false - let observer: ResizeObserver | null = null let unregister = () => {} @@ -101,10 +103,11 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id: return false }) - fitRef.current = () => { + fitRef.current = visible => { if (host.clientWidth > 0 && host.clientHeight > 0) { try { fit.fit() + markRightPanePerf(visible ? 'terminal-fit-active' : 'terminal-fit-hidden', id) } catch { // Mid-transition layout — the next observer tick refits. } @@ -132,9 +135,8 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id: // No WebGL — xterm falls back to the DOM renderer. } - fitRef.current?.() - observer = new ResizeObserver(() => fitRef.current?.()) - observer.observe(host) + fitRef.current?.(active) + initialActiveFitRef.current = active // Stream live output straight into the terminal (replays backlog on attach). unregister = registerAgentTerminalWriter(procId, chunk => term.write(chunk)) @@ -159,7 +161,6 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id: unregister() unregisterReader() selectionDisposable.dispose() - observer?.disconnect() fitRef.current = null term.dispose() termRef.current = null @@ -184,25 +185,39 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id: // eslint-disable-next-line react-hooks/exhaustive-deps }, [renderedMode, themeName]) - // A visibility:hidden xterm doesn't paint — refit + redraw on re-activation. + // Keep inactive agent terminals mounted for their backlog, but do not observe + // or fit them until they become the visible tab. + // eslint-disable-next-line no-restricted-syntax -- lifecycle flag prevents a duplicate first-mount fit useEffect(() => { if (!active) { + initialActiveFitRef.current = false + return } - const frame = requestAnimationFrame(() => { - const term = termRef.current + const host = hostRef.current - fitRef.current?.() - webglRef.current?.clearTextureAtlas() - term?.refresh(0, term.rows - 1) - // Take focus on activation (parity with the user terminal) so the active - // agent tab holds focus and ⌘W's isFocusWithin('[data-terminal]') routes - // the close to this tab rather than to a preview. - term?.focus() + if (!host) { + return + } + + const fitOnActivate = !initialActiveFitRef.current + initialActiveFitRef.current = false + + return observeActiveTerminalResize(host, { + fitOnActivate, + onFit: () => fitRef.current?.(true), + onActivate: () => { + const term = termRef.current + + webglRef.current?.clearTextureAtlas() + term?.refresh(0, term.rows - 1) + // Take focus on activation (parity with the user terminal) so the active + // agent tab holds focus and ⌘W's isFocusWithin('[data-terminal]') routes + // the close to this tab rather than to a preview. + term?.focus() + } }) - - return () => cancelAnimationFrame(frame) }, [active]) return { hostRef } diff --git a/apps/desktop/src/app/right-sidebar/terminal/use-terminal-font.ts b/apps/desktop/src/app/right-sidebar/terminal/use-terminal-font.ts index 656968071f..8bcf76426e 100644 --- a/apps/desktop/src/app/right-sidebar/terminal/use-terminal-font.ts +++ b/apps/desktop/src/app/right-sidebar/terminal/use-terminal-font.ts @@ -7,7 +7,7 @@ import type { RefObject } from 'react' import { $terminalFontFamily, applyTerminalFontFamily, resolveTerminalFontFamily } from './terminal-font' interface TerminalFontControllerOptions { - fitRef: RefObject<(() => void) | null> + fitRef: RefObject<((visible: boolean) => void) | null> termRef: RefObject webglRef: RefObject } @@ -38,7 +38,7 @@ export function useTerminalFontController({ fitRef, termRef, webglRef }: Termina void applyTerminalFontFamily({ clearTextureAtlas: () => webglRef.current?.clearTextureAtlas(), - fit: () => fitRef.current?.(), + fit: () => fitRef.current?.(true), fontFamily, isCurrent: () => !cancelled && generationRef.current === generation, term diff --git a/apps/desktop/src/app/right-sidebar/terminal/use-terminal-session.ts b/apps/desktop/src/app/right-sidebar/terminal/use-terminal-session.ts index e4e80fd3a4..24531cabe2 100644 --- a/apps/desktop/src/app/right-sidebar/terminal/use-terminal-session.ts +++ b/apps/desktop/src/app/right-sidebar/terminal/use-terminal-session.ts @@ -7,12 +7,14 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react' import type { CSSProperties } from 'react' import { writeClipboardText } from '@/components/ui/copy-button' +import { markRightPanePerf } from '@/debug/right-pane-events' import { triggerHaptic } from '@/lib/haptics' import { $previewTarget } from '@/store/preview' import { useTheme } from '@/themes/context' import { $terminalInjection } from '../store' +import { observeActiveTerminalResize } from './active-resize' import { makeTerminalReader, registerTerminalReader } from './buffer' import { mirrorSelection, terminalClipboardIntent } from './clipboard' import { terminalLinkHandler, terminalWebLinksAddon } from './links' @@ -413,6 +415,7 @@ export function useTerminalSession({ // drag-and-drop paths, or an injected command). Gates idle-buffer handling in // persistSnapshot so an untouched tab never re-saves an accumulating snapshot. const hasSessionActivityRef = useRef(false) + const initialActiveRef = useRef(active) const shellNameRef = useRef('shell') const selectionLabelRef = useRef('') const selectionRef = useRef('') @@ -420,7 +423,8 @@ export function useTerminalSession({ const onShellRef = useRef(onShell) // Re-fit on activation: a tab hidden via display:none has a 0×0 host, so its // last fit is stale by the time it's shown again. - const fitRef = useRef<(() => void) | null>(null) + const fitRef = useRef<((visible: boolean) => void) | null>(null) + const initialActiveFitRef = useRef(false) const { latestFontFamilyRef, mountedRef } = useTerminalFontController({ fitRef, termRef, webglRef }) const [status, setStatus] = useState('starting') const [selection, setSelection] = useState('') @@ -742,56 +746,28 @@ export function useTerminalSession({ term.write(next) } - const fitAndResize = () => { + const fitAndResize = (visible: boolean) => { if (disposed || !host.isConnected || host.clientWidth <= 0 || host.clientHeight <= 0) { return } try { fit.fit() + markRightPanePerf(visible ? 'terminal-fit-active' : 'terminal-fit-hidden', id) } catch { return } - const id = sessionIdRef.current + const sessionId = sessionIdRef.current - if (id && (lastSentSize?.cols !== term.cols || lastSentSize?.rows !== term.rows)) { + if (sessionId && (lastSentSize?.cols !== term.cols || lastSentSize?.rows !== term.rows)) { lastSentSize = { cols: term.cols, rows: term.rows } - void terminalApi.resize(id, { cols: term.cols, rows: term.rows }) + void terminalApi.resize(sessionId, { cols: term.cols, rows: term.rows }) } } fitRef.current = fitAndResize - // Coalesce ResizeObserver bursts through rAF — running fit.fit() - // synchronously while sibling panes are mid-transition (e.g. file browser - // collapsing to 0px) crashes the WebGL renderer mid texture-atlas rebuild. - let pendingFrame = 0 - - const scheduleResize = () => { - if (pendingFrame) { - return - } - - pendingFrame = window.requestAnimationFrame(() => { - pendingFrame = 0 - - if (!disposed) { - fitAndResize() - } - }) - } - - const resizeObserver = new ResizeObserver(scheduleResize) - resizeObserver.observe(host) - cleanup.push(() => { - resizeObserver.disconnect() - - if (pendingFrame) { - window.cancelAnimationFrame(pendingFrame) - } - }) - const dataDisposable = term.onData(data => { hasSessionActivityRef.current = true const id = sessionIdRef.current @@ -901,9 +877,7 @@ export function useTerminalSession({ ) window.requestAnimationFrame(() => { - fitAndResize() term.clearSelection() // drop any selection painted over transient boot rows - term.focus() }) }) .catch(error => { @@ -938,7 +912,8 @@ export function useTerminalSession({ console.warn('[hermes-terminal] WebGL unavailable; falling back to DOM', err) } - fitAndResize() + fitAndResize(initialActiveRef.current) + initialActiveFitRef.current = initialActiveRef.current startSession() } @@ -1013,24 +988,39 @@ export function useTerminalSession({ return term ? registerTerminalReader(id, makeTerminalReader(term)) : undefined }, [id, status]) - // On (re)activation: a WebGL terminal doesn't paint while visibility:hidden, so - // it reveals a stale/garbled frame. Refit, rebuild the glyph atlas, and force a - // full redraw against the live buffer, then focus. + // Only the active terminal observes its host. Every terminal stays mounted + // (PTY + scrollback preserved), but hidden tabs do no FitAddon/layout work. + // Re-activation owns one fit + atlas rebuild + redraw. + // eslint-disable-next-line no-restricted-syntax -- lifecycle flag prevents a duplicate first-mount fit useEffect(() => { if (!active || status !== 'open') { + if (!active) { + initialActiveFitRef.current = false + } + return } - const frame = requestAnimationFrame(() => { - const term = termRef.current + const host = hostRef.current - fitRef.current?.() - webglRef.current?.clearTextureAtlas() - term?.refresh(0, term.rows - 1) - term?.focus() + if (!host) { + return + } + + const fitOnActivate = !initialActiveFitRef.current + initialActiveFitRef.current = false + + return observeActiveTerminalResize(host, { + fitOnActivate, + onFit: () => fitRef.current?.(true), + onActivate: () => { + const term = termRef.current + + webglRef.current?.clearTextureAtlas() + term?.refresh(0, term.rows - 1) + term?.focus() + } }) - - return () => cancelAnimationFrame(frame) }, [active, status]) // Flush a queued command (e.g. a provider-disconnect) into the live session. diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/delta-flush.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/delta-flush.test.tsx index 1a572c329a..a630d9dcf3 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/delta-flush.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/delta-flush.test.tsx @@ -1,11 +1,14 @@ import { QueryClient } from '@tanstack/react-query' import { act, cleanup, render } from '@testing-library/react' -import { useEffect, useRef } from 'react' +import { type MutableRefObject, useEffect, useRef } from 'react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { ClientSessionState } from '@/app/types' +import type { ChatMessage } from '@/lib/chat-messages' import { createClientSessionState } from '@/lib/chat-runtime' +import { useSessionStateCache } from '../use-session-state-cache' + import { useMessageStream } from './index' const SID = 'session-1' @@ -90,6 +93,42 @@ describe('useMessageStream delta flush scheduling', () => { expect(assistantText()).toBe('still streaming') }) + it('flushes queued text immediately when a hidden window becomes visible', () => { + vi.mocked(performance.now).mockReturnValue(0) + mountStream() + + act(() => appendAssistantDelta!(SID, 'caught up on focus')) + expect(assistantText()).toBe('') + expect(vi.getTimerCount()).toBe(1) + + Object.defineProperty(globalThis.document, 'visibilityState', { + configurable: true, + value: 'visible' + }) + + act(() => globalThis.document.dispatchEvent(new Event('visibilitychange'))) + + expect(assistantText()).toBe('caught up on focus') + expect(vi.getTimerCount()).toBe(0) + }) + + it('flushes queued text on focus when visibility remains visible', () => { + vi.mocked(performance.now).mockReturnValue(0) + Object.defineProperty(globalThis.document, 'visibilityState', { + configurable: true, + value: 'visible' + }) + mountStream() + + act(() => appendAssistantDelta!(SID, 'focused without visibility change')) + expect(assistantText()).toBe('') + expect(vi.getTimerCount()).toBe(1) + + act(() => globalThis.window.dispatchEvent(new Event('focus'))) + + expect(assistantText()).toBe('focused without visibility change') + }) + it('cancels the pending timer on unmount and flushes exactly once', async () => { vi.mocked(performance.now).mockReturnValue(0) mountStream() @@ -108,4 +147,275 @@ describe('useMessageStream delta flush scheduling', () => { expect(updateSessionState).toHaveBeenCalledTimes(updatesAfterUnmount) expect(window.requestAnimationFrame).not.toHaveBeenCalled() }) + + it('stretches the flush gap when the deferred commit frame is expensive', async () => { + // The streaming-path $messages publish (React commit + Streamdown + // re-parse) is deferred to a view-sync rAF inside updateSessionState, so + // the flush cost must be measured through that frame. Simulate one + // expensive frame and expect the next gap to adapt to 3x the frame cost. + let now = 1000 + vi.mocked(performance.now).mockImplementation(() => now) + const rafCallbacks: FrameRequestCallback[] = [] + vi.mocked(window.requestAnimationFrame).mockImplementation(cb => { + rafCallbacks.push(cb) + + return rafCallbacks.length + }) + + mountStream() + + act(() => appendAssistantDelta!(SID, 'first')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + expect(assistantText()).toBe('first') + expect(rafCallbacks).toHaveLength(1) + + // Frame started at 1040, the measurement callback runs at 1100: 60ms of + // in-frame work (view sync + commit), so the next floor is 180ms. + now = 1100 + act(() => rafCallbacks[0](1040)) + + act(() => appendAssistantDelta!(SID, 'second')) + await act(async () => { + await vi.advanceTimersByTimeAsync(79) + }) + + expect(assistantText()).toBe('first') + + await act(async () => { + await vi.advanceTimersByTimeAsync(1) + }) + + expect(assistantText()).toBe('firstsecond') + }) + + it('keeps the write-cost floor when no frame fires (hidden renderer)', async () => { + // A parked renderer never runs rAF callbacks. The cost must stay at the + // synchronous store-write measurement so the gap falls back to the fixed + // 33ms floor instead of waiting on a frame that will never come. + let now = 1000 + vi.mocked(performance.now).mockImplementation(() => now) + vi.mocked(window.requestAnimationFrame).mockImplementation(() => 1) + + mountStream() + + act(() => appendAssistantDelta!(SID, 'first')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + expect(assistantText()).toBe('first') + + // 100ms later (well past the 33ms floor): the next flush is immediate. + now = 1100 + act(() => appendAssistantDelta!(SID, 'second')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + expect(assistantText()).toBe('firstsecond') + }) + + it('ignores a late frame measurement once a newer flush has started', async () => { + let now = 1000 + vi.mocked(performance.now).mockImplementation(() => now) + const rafCallbacks: FrameRequestCallback[] = [] + vi.mocked(window.requestAnimationFrame).mockImplementation(cb => { + rafCallbacks.push(cb) + + return rafCallbacks.length + }) + + mountStream() + + act(() => appendAssistantDelta!(SID, 'a')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + // A second flush starts before the first flush's frame lands. + now = 1010 + act(() => appendAssistantDelta!(SID, 'b')) + await act(async () => { + await vi.advanceTimersByTimeAsync(23) + }) + + expect(assistantText()).toBe('ab') + expect(rafCallbacks).toHaveLength(2) + + // The stale callback must not overwrite the newer flush's cost. If it + // did, cost would read 30ms and the next gap would stretch to 70ms. + now = 1030 + act(() => rafCallbacks[0](1000)) + + act(() => appendAssistantDelta!(SID, 'c')) + await act(async () => { + await vi.advanceTimersByTimeAsync(13) + }) + + expect(assistantText()).toBe('abc') + }) +}) + +describe('useMessageStream composed with the real useSessionStateCache', () => { + // The tests above mock updateSessionState, so they validate the adaptive + // arithmetic but not the production ordering contract: runFlush's + // measurement rAF must be registered AFTER the view-sync rAF that the real + // updateSessionState schedules inside syncSessionStateToView, so the + // measured frame cost includes the deferred $messages commit it adapts to. + let cache: ReturnType | null = null + let published: ChatMessage[] + + function ComposedHarness() { + const busyRef: MutableRefObject = { current: false } + const queryClientRef = useRef(new QueryClient()) + + const sessionCache = useSessionStateCache({ + activeSessionId: SID, + busyRef, + selectedStoredSessionId: null, + setAwaitingResponse: () => undefined, + setBusy: () => undefined, + setMessages: messages => { + published = messages + } + }) + + const stream = useMessageStream({ + activeSessionIdRef: sessionCache.activeSessionIdRef, + hydrateFromStoredSession: vi.fn(async () => undefined), + queryClient: queryClientRef.current, + refreshHermesConfig: vi.fn(async () => undefined), + refreshSessions: vi.fn(async () => undefined), + sessionStateByRuntimeIdRef: sessionCache.sessionStateByRuntimeIdRef, + updateSessionState: sessionCache.updateSessionState + }) + + useEffect(() => { + appendAssistantDelta = stream.appendAssistantDelta + cache = sessionCache + }, [stream.appendAssistantDelta, sessionCache]) + + return null + } + + function cachedText() { + const message = cache?.sessionStateByRuntimeIdRef.current.get(SID)?.messages.at(-1) + const part = message?.parts.at(-1) + + return part?.type === 'text' ? part.text : '' + } + + function publishedText() { + const part = published.at(-1)?.parts.at(-1) + + return part?.type === 'text' ? part.text : '' + } + + beforeEach(() => { + vi.useFakeTimers() + appendAssistantDelta = null + cache = null + published = [] + vi.spyOn(performance, 'now').mockReturnValue(100) + vi.spyOn(window, 'requestAnimationFrame').mockImplementation(() => 1) + vi.spyOn(window, 'cancelAnimationFrame').mockImplementation(() => undefined) + vi.spyOn(document, 'hasFocus').mockReturnValue(false) + }) + + afterEach(() => { + cleanup() + vi.useRealTimers() + vi.restoreAllMocks() + }) + + it('measures the frame cost through the real view-sync rAF and adapts the next gap', async () => { + let now = 1000 + vi.mocked(performance.now).mockImplementation(() => now) + const rafCallbacks: FrameRequestCallback[] = [] + vi.mocked(window.requestAnimationFrame).mockImplementation(cb => { + rafCallbacks.push(cb) + + return rafCallbacks.length + }) + + render() + expect(appendAssistantDelta).not.toBeNull() + + // Mid-turn state: busy keeps the view sync on the deferred rAF path + // (terminal/needing-input states flush synchronously instead). + act(() => { + cache!.updateSessionState(SID, state => ({ ...state, busy: true })) + }) + expect(rafCallbacks).toHaveLength(1) + // Drain the seed's own view-sync rAF so the flush below starts clean. + act(() => rafCallbacks.shift()!(now)) + + act(() => appendAssistantDelta!(SID, 'first')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + // The store write landed synchronously, but the $messages publish is + // deferred: exactly two rAF callbacks are pending — first the cache's + // view-sync, then runFlush's measurement. + expect(cachedText()).toBe('first') + expect(publishedText()).toBe('') + expect(rafCallbacks).toHaveLength(2) + + // Draining the FIRST registered callback must be what publishes the + // deferred commit; that identity is the ordering contract. It runs until + // 60ms into the frame (React commit + Streamdown re-parse). + now = 1100 + act(() => rafCallbacks[0](1040)) + expect(publishedText()).toBe('first') + + // The measurement callback closes the same frame: 60ms of in-frame work, + // so the next adaptive floor is 3x = 180ms. + act(() => rafCallbacks[1](1040)) + + act(() => appendAssistantDelta!(SID, 'second')) + await act(async () => { + await vi.advanceTimersByTimeAsync(79) + }) + + expect(cachedText()).toBe('first') + + await act(async () => { + await vi.advanceTimersByTimeAsync(1) + }) + + expect(cachedText()).toBe('firstsecond') + }) + + it('keeps the write-cost fallback when the parked renderer never fires rAF', async () => { + let now = 1000 + vi.mocked(performance.now).mockImplementation(() => now) + // Parked renderer: rAF callbacks are accepted but never run. + vi.mocked(window.requestAnimationFrame).mockImplementation(() => 1) + + render() + + act(() => { + cache!.updateSessionState(SID, state => ({ ...state, busy: true })) + }) + + act(() => appendAssistantDelta!(SID, 'first')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + expect(cachedText()).toBe('first') + + // 100ms later (well past the 33ms floor): the next flush is immediate. + now = 1100 + act(() => appendAssistantDelta!(SID, 'second')) + await act(async () => { + await vi.advanceTimersByTimeAsync(0) + }) + + expect(cachedText()).toBe('firstsecond') + }) }) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts index 29051cc365..a73852f6e9 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts @@ -188,6 +188,9 @@ export function useMessageStream({ // What the previous flush cost on the main thread — drives the adaptive // flush floor in scheduleDeltaFlush so multi-stream load yields to input. const lastFlushCostRef = useRef(0) + // The pending commit-cost measurement rAF, so a newer flush (or unmount) + // can cancel it instead of letting parked callbacks pile up while hidden. + const measureRafRef = useRef(null) const nativeSubagentSessionsRef = useRef>(new Set()) // Turns that auto-compacted: skip post-turn hydrate so live scrollback survives. const compactedTurnRef = useRef>(new Set()) @@ -257,6 +260,8 @@ export function useMessageStream({ // keeps the thread ~75% idle for input at any load: cheap flushes stay at // 30fps of text growth, expensive multi-stream flushes degrade text fps // instead of interactivity — capped so text never updates slower than 4/s. + // The cost has to include the deferred view-sync frame where the commit + // actually happens; see runFlush below. const sinceLast = performance.now() - lastFlushAtRef.current const adaptiveFloor = Math.min( @@ -269,7 +274,39 @@ export function useMessageStream({ const startedAt = performance.now() lastFlushAtRef.current = startedAt flushQueuedDeltas() - lastFlushCostRef.current = performance.now() - startedAt + // The store write above is only the cheap half of a flush. While a + // session streams, syncSessionStateToView defers the $messages publish + // (and with it the React commit + Streamdown re-parse the floor is meant + // to account for) to its own rAF inside updateSessionState, which runs + // after this timer task. Stopping the clock here pins lastFlushCostRef + // near zero and collapses the adaptive floor to 33ms no matter the load. + // Our rAF is registered after the view-sync one, so it runs in the same + // frame right after that commit; its timestamp marks frame start, so + // (now - frameStart) counts only work done inside the frame, not the + // vsync wait. A hidden renderer never fires rAF, so the write cost + // stays as the fallback. + const writeCost = performance.now() - startedAt + lastFlushCostRef.current = writeCost + + // At most one measurement rAF may be pending: only the newest flush's + // measurement matters (the guard below discards stale frames), and a + // hidden renderer parks rAF callbacks — without cancellation a long + // hidden stream at the floor would accumulate thousands of parked + // closures that all fire in the first frame on refocus. + if (measureRafRef.current !== null) { + window.cancelAnimationFrame(measureRafRef.current) + } + + measureRafRef.current = window.requestAnimationFrame(frameStart => { + measureRafRef.current = null + + // A newer flush already started; its own measurement wins. + if (lastFlushAtRef.current !== startedAt) { + return + } + + lastFlushCostRef.current = writeCost + Math.max(0, performance.now() - frameStart) + }) } // Always a timer, never requestAnimationFrame. Chromium pauses rAF for a @@ -312,11 +349,46 @@ export function useMessageStream({ } flushHandleRef.current = null + + if (measureRafRef.current !== null && typeof window !== 'undefined') { + window.cancelAnimationFrame(measureRafRef.current) + } + + measureRafRef.current = null flushQueuedDeltas() }, [flushQueuedDeltas] ) + // Page Visibility does not report every Windows/Linux focus transition. + // Flush queued deltas on both signals so returning to a chat cannot leave a + // completed chunk waiting for the next throttled timer. + // eslint-disable-next-line no-restricted-syntax -- timer-handle clear inside effect, not an atom mirror + useEffect(() => { + const flushPendingDeltas = () => { + if (flushHandleRef.current !== null) { + window.clearTimeout(flushHandleRef.current) + flushHandleRef.current = null + } + + flushQueuedDeltas() + } + + const flushWhenVisible = () => { + if (document.visibilityState === 'visible') { + flushPendingDeltas() + } + } + + document.addEventListener('visibilitychange', flushWhenVisible) + window.addEventListener('focus', flushPendingDeltas) + + return () => { + document.removeEventListener('visibilitychange', flushWhenVisible) + window.removeEventListener('focus', flushPendingDeltas) + } + }, [flushQueuedDeltas]) + const appendAssistantDelta = useCallback( (sessionId: string, delta: string) => { if (!delta) { diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/stream-flush.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/stream-flush.test.tsx index 427e62cf8a..840416a18e 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/stream-flush.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/stream-flush.test.tsx @@ -87,7 +87,10 @@ describe('stream delta delivery', () => { }) expect(states.get(SID)?.messages.at(-1)?.parts).toEqual([{ type: 'text', text: 'first and the rest' }]) - // The flush must not have depended on a frame at all. - expect(rafSpy).not.toHaveBeenCalled() + // The flush must not have depended on a frame: this mock parks every rAF + // callback, yet the text arrived. runFlush still registers its + // adaptive-floor measurement callback here; that one is allowed to wait + // for a frame that may never come. + expect(rafSpy).toHaveBeenCalled() }) }) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx index 57eea9ca06..9ac368fd44 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx @@ -1797,7 +1797,8 @@ describe('usePromptActions submit / queue drain semantics', () => { expect(accepted).toBe(true) expect(requestGateway).toHaveBeenCalledWith('session.resume', { session_id: 'stored-session-b', - source: 'desktop' + source: 'desktop', + omit_messages: true }) expect(requestGateway).toHaveBeenCalledWith( 'prompt.submit', @@ -1933,7 +1934,8 @@ describe('usePromptActions submit / queue drain semantics', () => { // Must resume the correct stored session to get the right runtime id. expect(requestGateway).toHaveBeenCalledWith('session.resume', { session_id: 'stored-session-a', - source: 'desktop' + source: 'desktop', + omit_messages: true }) // The prompt must land in the resumed session, NOT the foreground. expect(requestGateway).toHaveBeenCalledWith( @@ -2194,7 +2196,7 @@ describe('usePromptActions redirectPrompt', () => { expect(await handle!.redirectPrompt('reconnect nudge')).toBe(true) expect(calls.map(c => c.method)).toEqual(['session.redirect', 'session.resume', 'session.redirect']) expect(calls[0]?.params).toEqual({ session_id: RUNTIME_SESSION_ID, text: 'reconnect nudge' }) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' }) + expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', omit_messages: true }) expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID, text: 'reconnect nudge' }) expect(handle!.activeSessionIdRef.current).toBe(RECOVERED_SESSION_ID) }) @@ -2735,7 +2737,7 @@ describe('usePromptActions sleep/wake session recovery', () => { expect(ok).toBe(true) // First submit (stale id) → session.resume (stored id) → retry submit (fresh id). expect(calls.map(c => c.method)).toEqual(['prompt.submit', 'session.resume', 'prompt.submit']) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' }) + expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', omit_messages: true }) expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID, text: 'message after wake' }) }) @@ -2779,7 +2781,12 @@ describe('usePromptActions sleep/wake session recovery', () => { ) expect(await handle!.submitText('message after wake')).toBe(true) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', profile: 'work' }) + expect(calls[1]?.params).toEqual({ + session_id: STORED_SESSION_ID, + source: 'desktop', + omit_messages: true, + profile: 'work' + }) setSessions(() => []) }) @@ -2826,7 +2833,12 @@ describe('usePromptActions sleep/wake session recovery', () => { ) expect(await handle!.submitText('message after wake')).toBe(true) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', profile: 'work' }) + expect(calls[1]?.params).toEqual({ + session_id: STORED_SESSION_ID, + source: 'desktop', + omit_messages: true, + profile: 'work' + }) vi.mocked(getSession).mockReset() setSessions(() => []) @@ -2887,7 +2899,11 @@ describe('usePromptActions sleep/wake session recovery', () => { session_id: 'rt-background-stale', text: 'queued background message after wake' }) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' }) + expect(calls[1]?.params).toEqual({ + session_id: STORED_SESSION_ID, + source: 'desktop', + omit_messages: true + }) expect(calls[2]?.params).toEqual({ queued: true, session_id: RECOVERED_SESSION_ID, @@ -2935,7 +2951,11 @@ describe('usePromptActions sleep/wake session recovery', () => { expect(calls.map(c => c.method)).toEqual(['session.interrupt', 'session.resume', 'session.interrupt']) expect(calls[0]?.params).toEqual({ session_id: RUNTIME_SESSION_ID }) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' }) + expect(calls[1]?.params).toEqual({ + session_id: STORED_SESSION_ID, + source: 'desktop', + omit_messages: true + }) expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID }) }) @@ -3067,7 +3087,11 @@ describe('usePromptActions sleep/wake session recovery', () => { expect(ok).toBe(true) expect(calls.map(c => c.method)).toEqual(['prompt.submit', 'session.resume', 'prompt.submit']) - expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' }) + expect(calls[1]?.params).toEqual({ + session_id: STORED_SESSION_ID, + source: 'desktop', + omit_messages: true + }) expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID, text: 'message during starved loop' @@ -3110,7 +3134,11 @@ describe('usePromptActions sleep/wake session recovery', () => { expect(ok).toBe(true) expect(createBackendSessionForSend).not.toHaveBeenCalled() expect(calls.map(c => c.method)).toEqual(['session.resume', 'prompt.submit']) - expect(calls[0]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' }) + expect(calls[0]?.params).toEqual({ + session_id: STORED_SESSION_ID, + source: 'desktop', + omit_messages: true + }) expect(calls[1]?.params).toMatchObject({ session_id: RECOVERED_SESSION_ID }) }) @@ -3421,7 +3449,8 @@ describe('usePromptActions submit session-context isolation (#54527)', () => { expect(calls.some(c => c.method === 'prompt.submit')).toBe(false) expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ session_id: STORED_SESSION_A, - source: 'desktop' + source: 'desktop', + omit_messages: true }) }) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts index 6307969421..186773a874 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts @@ -639,6 +639,7 @@ export function usePromptActions({ const resumed = await requestGateway<{ session_id: string }>('session.resume', { session_id: selectedStoredSessionIdRef.current, source: 'desktop', + omit_messages: true, ...(resumeProfile ? { profile: resumeProfile } : {}) }) @@ -744,6 +745,7 @@ export function usePromptActions({ const resumed = await requestGateway<{ session_id: string }>('session.resume', { session_id: selectedStoredSessionIdRef.current, source: 'desktop', + omit_messages: true, ...(resumeProfile ? { profile: resumeProfile } : {}) }) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts index 20de6ea35c..36c8bfd82d 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts @@ -484,6 +484,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { const resumed = await requestGateway<{ session_id: string }>('session.resume', { session_id: targetStoredSessionId, source: 'desktop', + omit_messages: true, ...(resumeProfile ? { profile: resumeProfile } : {}) }) @@ -637,6 +638,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { const resumed = await requestGateway<{ session_id: string }>('session.resume', { session_id: recoverStoredSessionId, source: 'desktop', + omit_messages: true, ...(resumeProfile ? { profile: resumeProfile } : {}) }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx b/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx index 35110e481b..1bdb521efa 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx @@ -896,7 +896,7 @@ describe('resumeSession failure recovery', () => { expect(resumeParams).not.toHaveProperty('lazy') expect(resumeParams).not.toHaveProperty('eager_build') - expect(resumeParams).toMatchObject({ source: 'desktop' }) + expect(resumeParams).toMatchObject({ source: 'desktop', omit_messages: true }) }) it('arms the failure latch when resume succeeds with an empty transcript for a non-empty stored session', async () => { @@ -1431,6 +1431,10 @@ describe('resumeSession warm-cache mapping integrity', () => { expect(methods).toContain('session.activate') expect(methods).not.toContain('session.resume') expect(getSessionMessages).toHaveBeenCalledWith('stored-A', undefined) + expect(requestGateway).toHaveBeenCalledWith( + 'session.activate', + expect.objectContaining({ omit_messages: true, session_id: 'rt-A' }) + ) expect(runtimeIdByStoredSessionIdRef.current.get('stored-A')).toBe('rt-A') }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index 6a7e568cba..0f13a375c0 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -700,7 +700,8 @@ export function useSessionActions({ try { activated = await requestGateway('session.activate', { session_id: cachedRuntimeId, - cols: 96 + cols: 96, + omit_messages: true }) } catch (error) { // Compatibility for older backends. Modern backends require @@ -866,12 +867,14 @@ export function useSessionActions({ session_id: storedSessionId, cols: 96, source: 'desktop', + // REST is the transcript authority for Desktop. Avoid duplicating a + // potentially huge compression lineage in the WebSocket response. // Watch windows attach lazily (live mirror). Every other cold resume // gets the gateway's default deferred build: the RPC returns the // transcript immediately instead of blocking the switch on _make_agent // (MCP discovery / prompt build), and the agent pre-warms in the // background while the prefetch above paints the transcript. - ...(watchWindow ? { lazy: true } : {}), + ...(watchWindow ? { lazy: true } : { omit_messages: true }), ...(sessionProfile ? { profile: sessionProfile } : {}) }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/resume-structural-parts.test.ts b/apps/desktop/src/app/session/hooks/use-session-actions/resume-structural-parts.test.ts index 1ef5376d8d..2dff424a9f 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/resume-structural-parts.test.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/resume-structural-parts.test.ts @@ -73,4 +73,70 @@ describe('reconcileResumeMessages — structural parts on a mid-turn switch', () expect(assistant.parts.filter(p => p.type === 'tool-call')).toHaveLength(1) }) + + it('keeps live-tail structure when the flat dump is not a strict text extension', () => { + // Mid-turn sandwich path: cache holds reasoning/tools; resume returns a + // longer non-extending dump. Structure source must be live-tail. + const cached: ChatMessage[] = [ + { + id: 'assistant-stream-1', + pending: true, + parts: [ + { type: 'reasoning', text: 'thinking about tools' }, + { type: 'tool-call', toolCallId: 'c1', toolName: 'terminal', args: {} }, + { type: 'text', text: 'partial' } + ], + role: 'assistant' + } + ] + + const authoritative: ChatMessage[] = [ + { + id: 'assistant-stream-1', + pending: true, + parts: [{ type: 'text', text: 'thinking about tools\nRan terminal\npartial and more dump' }], + role: 'assistant' + } + ] + + const [assistant] = reconcileResumeMessages(authoritative, cached) + + expect(assistant.parts.some(part => part.type === 'reasoning')).toBe(true) + expect(assistant.parts.some(part => part.type === 'tool-call')).toBe(true) + expect(assistant.parts.filter(part => part.type === 'text').map(part => ('text' in part ? part.text : ''))).toEqual( + ['partial'] + ) + }) + + it('does not graft historical structure onto a live text-only row after compression rewrote ordinals', () => { + // Previous cache still has a completed structured assistant at ordinal 0. + // Resume after compression returns a new live text-only assistant at the + // same role ordinal for an unrelated turn — must not inherit foreign parts. + const cached: ChatMessage[] = [ + { + id: 'old-assistant', + parts: [ + { type: 'reasoning', text: 'old thinking' }, + { type: 'tool-call', toolCallId: 'old-call', toolName: 'terminal', args: {} }, + { type: 'text', text: 'old answer' } + ], + role: 'assistant' + } + ] + + const authoritative: ChatMessage[] = [ + { + id: 'assistant-stream-runtime-1', + pending: true, + parts: [{ type: 'text', text: 'brand new partial' }], + role: 'assistant' + } + ] + + const [assistant] = reconcileResumeMessages(authoritative, cached) + + expect(assistant.parts.some(part => part.type === 'reasoning')).toBe(false) + expect(assistant.parts.some(part => part.type === 'tool-call')).toBe(false) + expect(assistant.parts).toEqual([{ type: 'text', text: 'brand new partial' }]) + }) }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts index ea40e7705a..ed61eda70d 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest' -import type { ChatMessage } from '@/lib/chat-messages' +import { textWithoutReferenceLines, WIRE_REFERENCE_KINDS } from '@/components/assistant-ui/reference-kinds' +import { type ChatMessage, type ChatMessagePart, chatMessageText } from '@/lib/chat-messages' import { $approvalModes, approvalModeForProfile } from '@/store/approval-mode' import { $desktopOnboarding } from '@/store/onboarding' import { $activeGatewayProfile } from '@/store/profile' @@ -24,6 +25,21 @@ import { const msg = (id: string, role: ChatMessage['role'], text: string, extra: Partial = {}): ChatMessage => ({ id, role, parts: [{ type: 'text', text }], ...extra }) as ChatMessage +// A live assistant row carrying the structure the gateway's text-only inflight +// snapshot cannot: reasoning and tool calls, with or without any text yet. +const streamingMsg = (id: string, text: string, extra: Partial = {}): ChatMessage => + ({ + id, + role: 'assistant', + parts: [ + { type: 'reasoning', text: 'planning' }, + { type: 'tool-call', toolCallId: 'call-1', toolName: 'terminal', result: 'done' }, + ...(text ? [{ type: 'text', text } as ChatMessagePart] : []) + ], + pending: true, + ...extra + }) as ChatMessage + const session = (over: Partial): SessionInfo => over as SessionInfo describe('applyRuntimeInfo approval mode', () => { @@ -394,6 +410,126 @@ describe('reconcileResumeMessages', () => { expect(out.attachmentRefs).toBeUndefined() }) + + // #75825: switching sessions mid-stream can re-hydrate an empty inflight shell + // at the same ordinal as the live stream row that still holds the full reply. + it('prefers a richer local pending assistant over an empty projection shell', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'hello from stream', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: true })] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(reconciled[1]).toMatchObject({ id: 'assistant-stream-live', pending: true }) + expect(chatMessageText(reconciled[1])).toBe('hello from stream') + }) + + it('prefers a richer local pending assistant when the projection lags mid-stream', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'hello world', { pending: true }) + ] + + const next = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'hello', { pending: true }) + ] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(chatMessageText(reconciled[1])).toBe('hello world') + expect(reconciled[1].id).toBe('assistant-stream-live') + }) + + it('does not override when the authoritative assistant has advanced further', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'hello', { pending: true }) + ] + + const next = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'hello world', { pending: true }) + ] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(chatMessageText(reconciled[1])).toBe('hello world') + expect(reconciled[1].id).toBe('assistant-stream-sess') + }) + + // The reported "no inference traces or tool calls": mid tool-work, the local + // row holds reasoning + tool calls and NO text yet, so both bodies are empty + // text and a text-length comparison cannot tell them apart. + it('prefers a traces-only local pending row over an empty shell', () => { + const previous = [msg('1-user', 'user', 'run the tools'), streamingMsg('assistant-stream-live', '')] + + const next = [ + msg('1-user', 'user', 'run the tools'), + msg('assistant-stream-sess', 'assistant', '', { pending: true }) + ] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(reconciled[1].id).toBe('assistant-stream-live') + expect(reconciled[1].parts.map(part => part.type)).toEqual(['reasoning', 'tool-call']) + }) + + // A longer local body that is NOT an extension of the authoritative text is a + // different turn at the same ordinal (compression rewrites history) and must + // not hijack the slot. + it('leaves a shorter non-prefix authoritative assistant intact', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'a long local reply about something else entirely', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('9-assistant', 'assistant', 'short authoritative answer')] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(reconciled[1].id).toBe('9-assistant') + expect(chatMessageText(reconciled[1])).toBe('short authoritative answer') + }) + + // A retained failure snapshot (`inflight.error`) is projected with empty text. + // Preferring the local partial over it would erase the error and repaint the + // turn as healthy. + it('does not treat an errored authoritative row as an empty shell', () => { + const previous = [ + msg('1-user', 'user', 'do the thing'), + msg('assistant-stream-live', 'assistant', 'partial answer before the failure', { pending: true }) + ] + + const next = [ + msg('1-user', 'user', 'do the thing'), + msg('assistant-stream-sess', 'assistant', '', { error: 'model call failed: 500' }) + ] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(reconciled[1].error).toBe('model call failed: 500') + }) + + // Content comes from the renderer; liveness stays the backend's call. A + // settled shell (queued turn behind a finished inflight one) must not leave + // the preserved reply spinning forever. + it('takes the local body but the authoritative settled state', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'streamed body', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: false })] + + const reconciled = reconcileResumeMessages(next, previous) + + expect(reconciled[1]).toMatchObject({ id: 'assistant-stream-live', pending: false }) + expect(chatMessageText(reconciled[1])).toBe('streamed body') + }) }) describe('preserveLocalPendingTurnMessages', () => { @@ -556,7 +692,7 @@ describe('preserveLocalPendingTurnMessages', () => { // `attachmentRefs`. A naive text compare (chatMessageText a === b) therefore // always mismatched whenever an image was attached and re-appended the // optimistic row as a distinct, duplicate user bubble. Both sides must now - // reduce to the same visible text via textWithoutImageRefs. + // reduce to the same visible text via textWithoutReferenceLines. it('does not duplicate the optimistic image turn when the persisted turn carries @image refs', () => { const previous = [ msg('1-user', 'user', 'first'), @@ -575,6 +711,91 @@ describe('preserveLocalPendingTurnMessages', () => { expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) }) + it('does not duplicate the optimistic file turn when the persisted turn carries @file refs', () => { + const previous = [ + msg('1-user', 'user', 'first'), + msg('2-assistant', 'assistant', 'first answer'), + msg('user-optimistic', 'user', 'text', { + attachmentRefs: ['@file:X'] + }) + ] + + const next = [ + msg('1-user-stored', 'user', 'first'), + msg('2-assistant-stored', 'assistant', 'first answer'), + msg('3-user-stored', 'user', '@file:X\n\ntext') + ] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + }) + + it.each(WIRE_REFERENCE_KINDS.filter(kind => kind !== 'file' && kind !== 'image'))( + 'does not duplicate the optimistic %s turn when the persisted turn carries its directive', + kind => { + const ref = `@${kind}:X` + + const previous = [ + msg('1-user', 'user', 'first'), + msg('2-assistant', 'assistant', 'first answer'), + msg('user-optimistic', 'user', 'text', { + attachmentRefs: [ref] + }) + ] + + const next = [ + msg('1-user-stored', 'user', 'first'), + msg('2-assistant-stored', 'assistant', 'first answer'), + msg('3-user-stored', 'user', `${ref}\n\ntext`) + ] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + } + ) + + it('does not duplicate a directive-only file turn', () => { + const previous = [ + msg('1-user', 'user', 'first'), + msg('2-assistant', 'assistant', 'first answer'), + msg('user-optimistic', 'user', '', { + attachmentRefs: ['@file:X'] + }) + ] + + const next = [ + msg('1-user-stored', 'user', 'first'), + msg('2-assistant-stored', 'assistant', 'first answer'), + msg('3-user-stored', 'user', '@file:X') + ] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + }) + + it('does not duplicate a turn with multiple CRLF directives and Unicode payloads', () => { + const refs = ['@file:`資料/über notes.md`', '@url:`https://example.com/café?q=✓`'] + + const previous = [ + msg('1-user', 'user', 'first'), + msg('2-assistant', 'assistant', 'first answer'), + msg('user-optimistic', 'user', 'text', { + attachmentRefs: refs + }) + ] + + const next = [ + msg('1-user-stored', 'user', 'first'), + msg('2-assistant-stored', 'assistant', 'first answer'), + msg('3-user-stored', 'user', `${refs.join('\r\n')}\r\n\r\ntext`) + ] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + }) + + it('strips only complete reference lines from visible text', () => { + expect(textWithoutReferenceLines('see @file:X here')).toBe('see @file:X here') + expect(textWithoutReferenceLines('@file:X trailing prose')).toBe('@file:X trailing prose') + expect(textWithoutReferenceLines(' @file:X')).toBe('@file:X') + }) + it('still keeps a genuinely uncommitted optimistic image turn when the persisted text differs', () => { const previous = [ msg('1-user', 'user', 'first'), @@ -599,6 +820,177 @@ describe('preserveLocalPendingTurnMessages', () => { 'user-optimistic' ]) }) + + // #75825: an empty inflight projection shell at the same ordinal must not + // discard the local pending assistant that still holds the streamed content. + // Replace the shell (do not append) so the transcript shows one reply. + it('replaces an empty inflight shell with a fuller local pending assistant', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'partial answer so far', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: true })] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved.map(message => message.id)).toEqual(['1-user', 'assistant-stream-live']) + expect(chatMessageText(preserved[1])).toBe('partial answer so far') + expect(preserved[1].pending).toBe(true) + }) + + it('replaces a lagging same-id shell with the fuller local pending body', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'full streamed content', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: true })] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved.map(message => message.id)).toEqual(['1-user', 'assistant-stream-sess']) + expect(chatMessageText(preserved[1])).toBe('full streamed content') + }) + + it('still drops local pending when authoritative text is at least as complete', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'partial', { pending: true }) + ] + + const next = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'partial and more', { pending: true }) + ] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + }) + + // Mid tool-work both bodies are empty text, so only the parts distinguish the + // live row from the shell — the reported "no inference traces or tool calls". + it('replaces an empty shell with a traces-only local pending row', () => { + const previous = [msg('1-user', 'user', 'run the tools'), streamingMsg('assistant-stream-live', '')] + + const next = [ + msg('1-user', 'user', 'run the tools'), + msg('assistant-stream-sess', 'assistant', '', { pending: true }) + ] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved).toHaveLength(2) + expect(preserved[1].parts.map(part => part.type)).toEqual(['reasoning', 'tool-call']) + }) + + // Length alone is not identity: a longer local row that does not extend the + // authoritative text belongs to another turn and must not take its slot — by + // ordinal or by reusing the stream id. + it('leaves a shorter non-prefix authoritative assistant intact', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'a long local reply about something else entirely', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('9-assistant', 'assistant', 'short authoritative answer')] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved.map(message => message.id)).toEqual(['1-user', '9-assistant']) + expect(chatMessageText(preserved[1])).toBe('short authoritative answer') + }) + + it('leaves a shorter non-prefix authoritative assistant intact on the same stream id', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'a long local reply about something else entirely', { pending: true }) + ] + + const next = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'short authoritative answer') + ] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(chatMessageText(preserved[1])).toBe('short authoritative answer') + }) + + it('does not erase a retained failure with the local partial', () => { + const previous = [ + msg('1-user', 'user', 'do the thing'), + msg('assistant-stream-live', 'assistant', 'partial answer before the failure', { pending: true }) + ] + + const next = [ + msg('1-user', 'user', 'do the thing'), + msg('assistant-stream-sess', 'assistant', '', { error: 'model call failed: 500' }) + ] + + const assistant = preserveLocalPendingTurnMessages(next, previous).find(message => message.role === 'assistant') + + expect(assistant?.error).toBe('model call failed: 500') + expect(assistant?.pending).not.toBe(true) + }) + + it('takes the local body but the authoritative settled state', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-live', 'assistant', 'streamed body', { pending: true }) + ] + + const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: false })] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved[1]).toMatchObject({ id: 'assistant-stream-live', pending: false }) + expect(chatMessageText(preserved[1])).toBe('streamed body') + }) + + // #70209: history committed the reply under its own id, so the settled local + // stream row sits at a later ordinal, pairs with nothing, and gets appended — + // the same answer twice. + it('does not re-append a settled stream row the authoritative history already carries', () => { + const next = [msg('1-user-stored', 'user', 'question'), msg('2-assistant-stored', 'assistant', 'answer')] + const settledLocalStream = msg('assistant-stream-runtime-1', 'assistant', 'answer', { pending: false }) + + expect(preserveLocalPendingTurnMessages(next, [...next, settledLocalStream])).toBe(next) + }) + + // The reply finished locally but the gateway had not committed it when the + // session was reopened — the local row is the only copy and must survive. + it('keeps a settled stream row the authoritative history has not committed', () => { + const previous = [ + msg('1-user', 'user', 'question'), + msg('assistant-stream-sess', 'assistant', 'the finished reply', { pending: false }) + ] + + const next = [msg('1-user', 'user', 'question')] + + expect(preserveLocalPendingTurnMessages(next, previous).map(message => message.id)).toEqual([ + '1-user', + 'assistant-stream-sess' + ]) + }) + + // The whole point of replacing rather than appending: one reply on screen, + // and the committed history around the live turn untouched. + it('does not duplicate or rewrite committed history around the live turn', () => { + const history = [ + msg('1-user', 'user', 'first question'), + msg('2-assistant', 'assistant', 'first answer'), + msg('3-user', 'user', 'run the tools') + ] + + const previous = [...history, streamingMsg('assistant-stream-live', 'here is the full reply')] + const next = [...history, msg('assistant-stream-sess', 'assistant', '', { pending: true })] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved).toHaveLength(4) + expect(chatMessageText(preserved[1])).toBe('first answer') + expect(preserved.filter(message => message.role === 'assistant')).toHaveLength(2) + }) }) describe('appendLiveSessionProjection', () => { @@ -730,4 +1122,73 @@ describe('appendLiveSessionProjection', () => { expect(appendLiveSessionProjection(stored, { session_id: 'runtime-1' })).toBe(stored) }) + + it('does not sandwich a structured mid-turn row with the inflight flat dump (#76444)', () => { + const stored: ChatMessage[] = [ + msg('stored-user', 'user', 'do the work'), + { + id: 'live-assistant', + role: 'assistant', + pending: true, + parts: [ + { type: 'reasoning', text: 'thinking about tools' }, + { type: 'tool-call', toolCallId: 'c1', toolName: 'terminal', args: {} }, + { type: 'text', text: 'partial' } + ] + } + ] + + const restored = appendLiveSessionProjection(stored, { + session_id: 'runtime-1', + inflight: { + user: 'do the work', + // Flat dump includes thinking chatter + tool narration — longer than + // the answer text alone, which is how the sandwich used to grow. + assistant: 'thinking about tools\nRan terminal\npartial and more dump', + streaming: true + } + }) + + const assistants = restored.filter(message => message.role === 'assistant') + expect(assistants).toHaveLength(1) + expect(assistants[0].id).toBe('live-assistant') + expect(assistants[0].parts.some(part => part.type === 'reasoning')).toBe(true) + expect(assistants[0].parts.some(part => part.type === 'tool-call')).toBe(true) + // Answer text stays the structured row's text, not the dump. + expect( + assistants[0].parts.filter(part => part.type === 'text').map(part => ('text' in part ? part.text : '')) + ).toEqual(['partial']) + }) + + it('still projects inflight when only a completed historical tool reply has structure', () => { + // Older completed assistants keep reasoning/tool parts in the full + // transcript; they must not suppress a new turn's text projection. + const stored: ChatMessage[] = [ + msg('old-user', 'user', 'previous task'), + { + id: 'old-assistant', + role: 'assistant', + parts: [ + { type: 'tool-call', toolCallId: 'old', toolName: 'terminal', args: {} }, + { type: 'text', text: 'done earlier' } + ] + }, + msg('new-user', 'user', 'new task') + ] + + const restored = appendLiveSessionProjection(stored, { + session_id: 'runtime-1', + inflight: { + user: 'new task', + assistant: 'working on it', + streaming: true + } + }) + + expect(restored.map(message => message.id)).toContain('assistant-stream-runtime-1') + expect(restored.at(-1)).toMatchObject({ + id: 'assistant-stream-runtime-1', + pending: true + }) + }) }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 1f86fd7c40..b6bc4ae6d5 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -1,7 +1,8 @@ +import { textWithoutReferenceLines } from '@/components/assistant-ui/reference-kinds' import { getSession } from '@/hermes' import { assistantTextPart, type ChatMessage, chatMessageText, textPart } from '@/lib/chat-messages' import { normalizePersonalityValue } from '@/lib/chat-runtime' -import { embeddedImageUrls, textWithoutEmbeddedImages, textWithoutImageRefs } from '@/lib/embedded-images' +import { embeddedImageUrls, textWithoutEmbeddedImages } from '@/lib/embedded-images' import { reconcileApprovalModeForProfile } from '@/store/approval-mode' import { requestDesktopOnboardingForCredentialWarning } from '@/store/onboarding' import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/profile' @@ -46,6 +47,41 @@ function withAppendedText(message: ChatMessage, suffix: string): ChatMessage { return appended ? { ...message, parts } : message } +/** Reasoning / tool-call parts that the gateway inflight dump cannot express. */ +function hasStructuralParts(message: ChatMessage): boolean { + return message.parts.some(part => part.type === 'reasoning' || part.type === 'tool-call') +} + +/** + * A live-turn row — the gateway's text-only `inflight` projection, a + * still-streaming local bubble, or an interim row sealed inside the running + * turn — as opposed to a committed transcript row. + */ +function isLiveTailRow(message: ChatMessage): boolean { + return ( + message.pending === true || + message.id.startsWith('assistant-stream-') || + message.id.startsWith('inflight-assistant-') || + message.interim === true + ) +} + +/** + * True when `next` is a pure forward extension of the previous *answer* text. + * Empty previous answer never accepts a dump as an extension — that is how the + * mid-turn inflight flat dump used to sandwich structured rows (#76444). + */ +export function isStrictAnswerTextExtension(next: string, previous: string): boolean { + const n = next.trim() + const p = previous.trim() + + if (!p || !n) { + return false + } + + return n.startsWith(p) +} + /** * Carry structural parts an authoritative row cannot express. * @@ -250,8 +286,20 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes const nextText = chatMessageText(message).trim() const previousText = chatMessageText(previous) const previousVisibleText = textWithoutEmbeddedImages(previousText) + const previousTrimmed = previousVisibleText.trim() let preserved = message + // #75825: resume can project an empty (or lagging) inflight assistant shell + // at the same role-ordinal as the live stream row that still holds the + // streamed text, reasoning and tool calls. Prefer that richer pending row + // instead of painting the shell — otherwise the reply vanishes until + // restart. Guarded to the same reply further along (see + // localPendingSupersedes) so a different turn at the same ordinal cannot + // hijack the slot. + if (localPendingSupersedes(previous, message)) { + return withAuthoritativeTurnState(previous, message) + } + const sameText = nextText === previousVisibleText || nextText === previousText.trim() // Mid-turn, the authoritative text has advanced past the cached copy by one @@ -260,12 +308,36 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes // for structural carry-over. Attachment refs and image re-appending stay on // the strict equality path — they reconcile a SETTLED row, and a growing // row is by definition not settled. + // + // Live-tail identity: structure-only same-turn carry is allowed only when + // the *structure-bearing cached row* is still the in-flight stream + // (pending / stream id / interim). Marking only the text-only next row + // live is not enough — after compression a new live assistant can share a + // role ordinal with an unrelated historical structured row and must not + // inherit its reasoning/tool parts (#76444 review / salvage). const sameTurn = sameText || - (nextText.length > 0 && previousVisibleText.length > 0 && nextText.startsWith(previousVisibleText.trim())) + (nextText.length > 0 && previousTrimmed.length > 0 && isStrictAnswerTextExtension(nextText, previousTrimmed)) || + (message.role === 'assistant' && + previous.role === 'assistant' && + hasStructuralParts(previous) && + !hasStructuralParts(message) && + isLiveTailRow(previous)) if (sameTurn) { preserved = preserveStructuralParts(preserved, previous) + + // Never replace structured answer text with a non-extending flat dump. + if ( + message.role === 'assistant' && + hasStructuralParts(previous) && + !hasStructuralParts(message) && + !isStrictAnswerTextExtension(nextText, previousVisibleText) + ) { + const nonText = preserved.parts.filter(part => part.type !== 'text') + const priorAnswer = previous.parts.filter(part => part.type === 'text') + preserved = { ...preserved, parts: [...nonText, ...priorAnswer] } + } } if ( @@ -315,8 +387,10 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes * history window. Preserve only the newest optimistic user row: compression * rewrites past context, so older `user-*` rows in a warm cache are stale * history, not in-flight work. The latest authoritative user confirms whether - * that tail has persisted; any authoritative assistant at the same ordinal - * supersedes the local stream. + * that tail has persisted. An authoritative assistant at the same ordinal + * supersedes the local stream only when it is at least as complete; an empty + * or lagging inflight shell must not discard a fuller local pending reply + * (#75825). * * Gateway bookkeeping markers (the model-switch / personality notices written * by tui_gateway/server.py) are persisted as role=user but are not user turns. @@ -328,6 +402,64 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes const isGatewaySystemMarker = (message: ChatMessage): boolean => message.role === 'user' && chatMessageText(message).trimStart().startsWith('[System:') +/** + * Does the row carry anything a viewer would miss — streamed answer text, or + * the reasoning / tool-call structure the gateway's flat dump cannot express? + * An empty inflight shell carries none of it. + */ +const hasStreamedContent = (message: ChatMessage): boolean => + chatMessageText(message).trim().length > 0 || hasStructuralParts(message) + +/** + * May the cached local row stand in for this authoritative assistant? + * + * Only for a live projection of the SAME reply that the local copy is further + * along on: an empty shell, or text the local row strictly extends. Comparing + * lengths alone lets an unrelated (merely longer) local row hijack the ordinal + * — or the stream id — of a genuine stored reply. A retained failure snapshot + * (`inflight.error`, projected with empty text) is never a shell: repainting it + * from the local partial would hide the error and mark the turn healthy again. + */ +const localPendingSupersedes = (local: ChatMessage, authoritative: ChatMessage): boolean => { + if (local.role !== 'assistant' || !isLiveTailRow(local)) { + return false + } + + if (!isLiveTailRow(authoritative) || authoritative.error) { + return false + } + + const authoritativeText = chatMessageText(authoritative).trim() + + if (!authoritativeText.length) { + return hasStreamedContent(local) + } + + const localText = chatMessageText(local).trim() + + return localText.length > authoritativeText.length && isStrictAnswerTextExtension(localText, authoritativeText) +} + +/** + * Take the cached row's content, but never its liveness. The renderer holds the + * only copy of the streamed parts; the gateway remains the authority on whether + * the turn is still running and on durable row identity — so a settled shell + * must not repaint the reply as perpetually streaming. + */ +const withAuthoritativeTurnState = (local: ChatMessage, authoritative: ChatMessage): ChatMessage => { + const merged: ChatMessage = { ...local, pending: authoritative.pending === true } + + if (local.rowId === undefined && authoritative.rowId !== undefined) { + merged.rowId = authoritative.rowId + } + + if (local.reactions === undefined && authoritative.reactions?.length) { + merged.reactions = [...authoritative.reactions] + } + + return merged +} + export function preserveLocalPendingTurnMessages( nextMessages: ChatMessage[], previousMessages: ChatMessage[] @@ -378,6 +510,9 @@ export function preserveLocalPendingTurnMessages( const latestAuthoritativeUser = [...nextMessages].reverse().find(message => message.role === 'user') const preserved: ChatMessage[] = [] + // Authoritative id → richer local pending row. Replacing (not appending) + // avoids painting both the empty inflight shell and the full stream bubble. + const replacements = new Map() for (const message of previousMessages) { if (isGatewaySystemMarker(message)) { @@ -392,7 +527,21 @@ export function preserveLocalPendingTurnMessages( const isPendingAssistant = message.role === 'assistant' && (message.pending === true || message.id.startsWith('assistant-stream-')) - if ((!isOptimisticUser && !isPendingAssistant) || nextIds.has(message.id)) { + if (!isOptimisticUser && !isPendingAssistant) { + continue + } + + // Same id already present: still prefer a strictly more complete local + // pending body over an empty/stale shell that reused the stream id. + if (nextIds.has(message.id)) { + if (isPendingAssistant) { + const existing = nextMessages.find(candidate => candidate.id === message.id) + + if (existing && localPendingSupersedes(message, existing)) { + replacements.set(message.id, withAuthoritativeTurnState(message, existing)) + } + } + continue } @@ -403,19 +552,50 @@ export function preserveLocalPendingTurnMessages( if ( isOptimisticUser && latestAuthoritativeUser && - textWithoutImageRefs(chatMessageText(latestAuthoritativeUser)) === textWithoutImageRefs(chatMessageText(message)) + textWithoutReferenceLines(chatMessageText(latestAuthoritativeUser)) === + textWithoutReferenceLines(chatMessageText(message)) ) { continue } const authoritative = nextByRoleOrdinal.get(`${message.role}:${ordinal}`) + // A settled stream row (`pending: false` after message.complete) whose reply + // the authoritative transcript already carries under its committed id is + // stale: ordinal pairing can't see it, because the commit shifted the row + // one ordinal earlier, and re-appending it renders the same answer twice + // (#70209). Only text-identical rows are dropped — a settled row the backend + // has NOT committed yet is the only copy of that reply and must survive. + if ( + isPendingAssistant && + message.pending !== true && + nextMessages.some( + candidate => + candidate.role === 'assistant' && + textWithoutReferenceLines(chatMessageText(candidate)) === textWithoutReferenceLines(chatMessageText(message)) + ) + ) { + continue + } + if (authoritative) { if (isPendingAssistant) { + // Keep the local pending row when it is the same reply further along + // and the authoritative row is an empty projection shell or a prefix. + // #75825 + if (!localPendingSupersedes(message, authoritative)) { + continue + } + + replacements.set(authoritative.id, withAuthoritativeTurnState(message, authoritative)) + continue } - if (textWithoutImageRefs(chatMessageText(authoritative)) === textWithoutImageRefs(chatMessageText(message))) { + if ( + textWithoutReferenceLines(chatMessageText(authoritative)) === + textWithoutReferenceLines(chatMessageText(message)) + ) { continue } } @@ -423,7 +603,10 @@ export function preserveLocalPendingTurnMessages( preserved.push(message) } - return preserved.length ? [...nextMessages, ...preserved] : nextMessages + const withReplacements = + replacements.size > 0 ? nextMessages.map(message => replacements.get(message.id) ?? message) : nextMessages + + return preserved.length ? [...withReplacements, ...preserved] : withReplacements } /** @@ -484,7 +667,9 @@ export function appendLiveSessionProjection( } const persistedInLatestRun = (text: string): boolean => - latestUserRun.some(message => textWithoutImageRefs(chatMessageText(message)) === textWithoutImageRefs(text)) + latestUserRun.some( + message => textWithoutReferenceLines(chatMessageText(message)) === textWithoutReferenceLines(text) + ) const inflightUserAlreadyPersisted = Boolean(inflightUser) && persistedInLatestRun(inflightUser) @@ -514,14 +699,54 @@ export function appendLiveSessionProjection( // Keep a pending assistant boundary even before the first delta when a // queued user turn follows it. This preserves the two distinct turns. + // + // When the *current live turn* already holds a structured mid-turn assistant + // row (reasoning / tool-call from the live stream or journal), do NOT append + // a pure-text projection of `inflight.assistant` — that flat dump re-renders + // thinking as answer text and sandwiches the structured parts (#76444). + // Only inspect the live tail after the latest user run — never a completed + // historical tool-bearing reply earlier in the transcript (review feedback). + const liveStreamId = `assistant-stream-${sessionId}` + + const liveAssistantOfCurrentTurn = ((): ChatMessage | null => { + const byStreamId = messages.find(message => message.id === liveStreamId) + + if (byStreamId) { + return byStreamId + } + + // Assistants after the latest user row belong to this turn's tail. + if (latestUserIndex < 0) { + return null + } + + for (let index = messages.length - 1; index > latestUserIndex; index -= 1) { + if (messages[index].role === 'assistant') { + return messages[index] + } + } + + return null + })() + + const turnAlreadyStructured = Boolean( + liveAssistantOfCurrentTurn && + hasStructuralParts(liveAssistantOfCurrentTurn) && + isLiveTailRow(liveAssistantOfCurrentTurn) + ) + if (inflightAssistant || inflightStreaming || inflightError || (inflightUser && queuedUser)) { - projected.push({ - id: `assistant-stream-${sessionId}`, - role: 'assistant', - parts: inflightAssistant ? [assistantTextPart(inflightAssistant)] : [], - pending: inflightStreaming, - ...(inflightError ? { error: inflightError } : {}) - }) + if (turnAlreadyStructured && !inflightError) { + // Structure is authoritative; skip the text-only dump row. + } else { + projected.push({ + id: liveStreamId, + role: 'assistant', + parts: inflightAssistant ? [assistantTextPart(inflightAssistant)] : [], + pending: inflightStreaming, + ...(inflightError ? { error: inflightError } : {}) + }) + } } if (queuedUser) { diff --git a/apps/desktop/src/components/assistant-ui/reference-kinds.ts b/apps/desktop/src/components/assistant-ui/reference-kinds.ts index 453d05411d..7d6c397584 100644 --- a/apps/desktop/src/components/assistant-ui/reference-kinds.ts +++ b/apps/desktop/src/components/assistant-ui/reference-kinds.ts @@ -157,3 +157,17 @@ const REFERENCE_PATTERN = /@(file|folder|url|image|tool|line|terminal|session):( export function referenceRe(): RegExp { return new RegExp(REFERENCE_PATTERN.source, 'g') } + +/** Remove reference-only lines when comparing visible message text. */ +// Anchored + non-global: no shared `lastIndex` state (the hazard referenceRe() +// exists to avoid), and hoisting skips a RegExp construction per call — this +// runs on both sides of every message comparison in the reconcile loops. +const REFERENCE_LINE_RE = new RegExp(`^(?:${REFERENCE_PATTERN.source})$`) + +export function textWithoutReferenceLines(text: string): string { + return text + .split('\n') + .filter(line => !REFERENCE_LINE_RE.test(line.trimEnd())) + .join('\n') + .trim() +} diff --git a/apps/desktop/src/components/assistant-ui/thread/list.test.ts b/apps/desktop/src/components/assistant-ui/thread/list.test.ts index 413aab8865..a44314d154 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.test.ts +++ b/apps/desktop/src/components/assistant-ui/thread/list.test.ts @@ -1,5 +1,7 @@ import { describe, expect, it } from 'vitest' +import { messageRenderWeight, RENDER_WEIGHT_CHARS } from '@/lib/render-weight' + import { buildGroups, firstVisibleGroupIndex, @@ -7,8 +9,7 @@ import { LIVE_TAIL_PARTS, liveTailStart, type MessageGroup, - messageRenderWeight, - RENDER_WEIGHT_CHARS + resolveThreadScrollTarget } from './list' // Signature rows are `${index}:${id}:${role}:${weight}` (see the useAuiState @@ -62,6 +63,53 @@ describe('buildGroups', () => { }) }) +describe('resolveThreadScrollTarget', () => { + const context = (scrollElement: Pick) => ({ + contentElement: document.createElement('div'), + scrollElement: scrollElement as HTMLElement + }) + + it('settles when the browser clamps the requested bottom within half a CSS pixel', () => { + let actualScrollTop = 0 + let writes = 0 + + const scrollElement = { + get scrollTop() { + return actualScrollTop + }, + set scrollTop(value: number) { + writes += 1 + actualScrollTop = value - 0.125 + } + } + + const target = 899 + + const requested = resolveThreadScrollTarget(target, context(scrollElement)) + scrollElement.scrollTop = requested + const settled = resolveThreadScrollTarget(target, context(scrollElement)) + + expect(requested).toBe(target) + expect(actualScrollTop).toBe(898.875) + expect(settled).toBe(actualScrollTop) + expect(actualScrollTop < settled).toBe(false) + expect(writes).toBe(1) + }) + + it('keeps following while more than half a CSS pixel remains', () => { + const scrollElement = { scrollTop: 898.25 } + + expect(resolveThreadScrollTarget(899, context(scrollElement))).toBe(899) + }) + + it('re-arms after streaming content increases the target', () => { + const scrollElement = { scrollTop: 898.875 } + + expect(resolveThreadScrollTarget(899, context(scrollElement))).toBe(898.875) + expect(resolveThreadScrollTarget(999, context(scrollElement))).toBe(999) + }) +}) + describe('firstVisibleGroupIndex', () => { const group = (id: string, weight: number): MessageGroup => ({ id, index: 0, kind: 'standalone', weight }) diff --git a/apps/desktop/src/components/assistant-ui/thread/list.tsx b/apps/desktop/src/components/assistant-ui/thread/list.tsx index faa0724883..304b3eb3ec 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/list.tsx @@ -13,9 +13,10 @@ import { useRef, useState } from 'react' -import { useStickToBottom } from 'use-stick-to-bottom' +import { type GetTargetScrollTop, useStickToBottom } from 'use-stick-to-bottom' import { useI18n } from '@/i18n' +import { messageRenderWeight } from '@/lib/render-weight' import { cn } from '@/lib/utils' import { onScrollToBottomRequest, @@ -28,6 +29,8 @@ import { isSecondaryWindow } from '@/store/windows' import { MessageRenderBoundary } from '../message-render-boundary' +import { resolveShowEarlierAction, useTranscriptWindow } from './transcript-window' + type ThreadMessageComponents = ComponentProps['components'] export type MessageGroup = { id: string; weight: number } & ( @@ -47,8 +50,6 @@ export type MessageGroup = { id: string; weight: number } & ( // a virtualizer — pure rendering, never touches scrollTop, so it can't fight // use-stick-to-bottom (the single scroll owner). const RENDER_BUDGET = 300 -export const RENDER_WEIGHT_CHARS = 512 -const MAX_MEASURED_MESSAGE_CHARS = RENDER_BUDGET * RENDER_WEIGHT_CHARS // On session switch, paint a small budget first (enough for the bottom turn(s) // the user actually sees after scroll-to-bottom), then bump to the full budget // in a requestAnimationFrame — defers the heavy markdown+syntax-highlight render @@ -62,68 +63,18 @@ const MAX_MEASURED_MESSAGE_CHARS = RENDER_BUDGET * RENDER_WEIGHT_CHARS // blocks the click-to-paint path. const FIRST_PAINT_BUDGET = 20 -const contentWeightCache = new WeakMap() -const NON_RENDERED_CONTENT_FIELDS = new Set(['id', 'role', 'toolCallId', 'toolName', 'type']) +// Browsers may quantize a requested scrollTop to a nearby device-pixel +// boundary. use-stick-to-bottom otherwise compares the lower actual value to +// the integer target forever, re-requesting the same instant scroll every +// frame. Treat a subpixel remainder as achieved; larger gaps still follow new +// streamed content normally. +const SCROLL_TARGET_EPSILON_PX = 0.5 -/** - * Estimate the synchronous renderer cost of one assistant-ui message. - * - * The traversal is capped once a single message has enough text to consume a - * complete render page. Going further cannot affect which whole turn crosses - * the budget, and avoiding an unbounded walk matters for deeply nested tool - * payloads. A WeakMap keeps settled history O(message count) on later store - * updates; assistant-ui publishes a new content array when a streaming message - * changes, so the live tail still receives a fresh weight. - */ -export function messageRenderWeight(content: unknown): number { - if (!Array.isArray(content)) { - return 1 - } +export const resolveThreadScrollTarget: GetTargetScrollTop = (targetScrollTop, { scrollElement }) => { + const currentScrollTop = scrollElement.scrollTop + const remaining = targetScrollTop - currentScrollTop - const cached = contentWeightCache.get(content) - - if (cached !== undefined) { - return cached - } - - const seen = new WeakSet() - const pending: unknown[] = [...content] - let characters = 0 - - while (pending.length > 0 && characters < MAX_MEASURED_MESSAGE_CHARS) { - const value = pending.pop() - - if (typeof value === 'string') { - characters += Math.min(value.length, MAX_MEASURED_MESSAGE_CHARS - characters) - - continue - } - - if (!value || typeof value !== 'object' || seen.has(value)) { - continue - } - - seen.add(value) - - if (Array.isArray(value)) { - for (const nested of value) { - pending.push(nested) - } - - continue - } - - for (const [key, nested] of Object.entries(value)) { - if (!NON_RENDERED_CONTENT_FIELDS.has(key)) { - pending.push(nested) - } - } - } - - const weight = Math.max(1, content.length) + Math.ceil(characters / RENDER_WEIGHT_CHARS) - contentWeightCache.set(content, weight) - - return weight + return remaining >= 0 && remaining <= SCROLL_TARGET_EPSILON_PX ? currentScrollTop : targetScrollTop } interface ThreadMessageListProps { @@ -297,9 +248,12 @@ const ThreadMessageListInner: FC = ({ // settling. Its refs hang off our own DOM so the sticky human bubbles survive. const { scrollRef, contentRef, isAtBottom, scrollToBottom, stopScroll } = useStickToBottom({ initial: 'instant', - resize: 'instant' + resize: 'instant', + targetScrollTop: resolveThreadScrollTarget }) + const { olderAvailable, expandWindow } = useTranscriptWindow() + const [renderBudget, setRenderBudget] = useState(FIRST_PAINT_BUDGET) // Cut the budget during RENDER, not in the post-commit layout effect. An @@ -525,11 +479,26 @@ const ThreadMessageListInner: FC = ({ // Prepend an older page while preserving the on-screen position. The user is // scrolled up (reading history) so the stick-to-bottom lock is escaped and - // won't fight this manual restore. + // won't fight this manual restore. Spend the already-materialized DOM page + // first; only when that is exhausted pull more messages out of the session + // store (#55191). const showEarlier = useCallback(() => { + const action = resolveShowEarlierAction(hiddenCount, olderAvailable) + + if (!action) { + return + } + anchorBeforePrepend() - setRenderBudget(budget => budget + RENDER_BUDGET) - }, [anchorBeforePrepend]) + + if (action === 'dom') { + setRenderBudget(budget => budget + RENDER_BUDGET) + + return + } + + expandWindow() + }, [anchorBeforePrepend, expandWindow, hiddenCount, olderAvailable]) useLayoutEffect(() => { const el = scrollRef.current @@ -538,7 +507,8 @@ const ThreadMessageListInner: FC = ({ el.scrollTop = el.scrollHeight - restoreFromBottomRef.current restoreFromBottomRef.current = null } - }, [scrollRef, renderBudget]) + // renderBudget covers DOM pages; groups.length covers store-window expands. + }, [scrollRef, renderBudget, groups.length]) // The row array is memoized on the inputs the rows actually read. This // component re-renders on every isAtBottom flip — and use-stick-to-bottom @@ -632,7 +602,7 @@ const ThreadMessageListInner: FC = ({ data-slot="aui_thread-content" ref={contentRef as React.RefCallback} > - {hiddenCount > 0 && ( + {(hiddenCount > 0 || olderAvailable) && (