Merge updated tool metrics into skill metrics
Signed-off-by: Alex Fournier <afournier@nvidia.com>
This commit is contained in:
@@ -107,7 +107,9 @@ jobs:
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Set up Python 3.11 (for docker tests)
|
||||
run: uv python install 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install Python dependencies (for docker tests)
|
||||
# ``dev`` extra pulls in pytest, pytest-asyncio —
|
||||
|
||||
@@ -66,8 +66,12 @@ jobs:
|
||||
cache-dependency-glob: |
|
||||
pyproject.toml
|
||||
uv.lock
|
||||
|
||||
- name: Set up Python 3.11
|
||||
run: uv python install 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install Python dependencies
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
|
||||
@@ -163,10 +163,18 @@ jobs:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v5
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
python-version: "3.11"
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Set up Python 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Run footgun checker
|
||||
run: python scripts/check-windows-footguns.py --all
|
||||
|
||||
@@ -90,7 +90,9 @@ jobs:
|
||||
uv.lock
|
||||
|
||||
- name: Set up Python 3.11
|
||||
run: uv python install 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install dependencies
|
||||
# `uv sync --locked` installs the exact pinned set from uv.lock (and
|
||||
|
||||
@@ -158,9 +158,16 @@ def make_approval_callback(
|
||||
|
||||
try:
|
||||
response = future.result(timeout=timeout)
|
||||
except (FutureTimeout, Exception) as exc:
|
||||
except FutureTimeout:
|
||||
future.cancel()
|
||||
logger.warning("Permission request timed out or failed: %s", exc)
|
||||
logger.warning("Permission request timed out after %ss", timeout)
|
||||
# Distinct from an explicit deny: the client never answered.
|
||||
# tools.approval callers report this as "timed out without user
|
||||
# response" instead of a user denial.
|
||||
return "timeout"
|
||||
except Exception as exc:
|
||||
future.cancel()
|
||||
logger.warning("Permission request failed: %s", exc)
|
||||
return "deny"
|
||||
|
||||
if response is None:
|
||||
|
||||
+23
-1
@@ -209,7 +209,18 @@ def _context_route_mismatch(
|
||||
|
||||
if active_route:
|
||||
configured_routes = _provider_default_routes(configured_provider)
|
||||
return not configured_routes or active_route not in configured_routes
|
||||
if configured_routes:
|
||||
return active_route not in configured_routes
|
||||
# Named/custom providers have no catalog default routes. An empty
|
||||
# configured URL with a matching provider identity is still the same
|
||||
# route — agent_init fills base_url from custom_providers before this
|
||||
# check, but gateway display/hygiene paths historically compared the
|
||||
# raw empty model.base_url and falsely dropped model.context_length,
|
||||
# falling through to family defaults (e.g. qwen → 131072) on Discord
|
||||
# session-reset banners while /status still showed the config pin.
|
||||
if active_provider and configured_provider == active_provider:
|
||||
return False
|
||||
return True
|
||||
return bool(
|
||||
configured_provider
|
||||
and active_provider
|
||||
@@ -1580,6 +1591,17 @@ def init_agent(
|
||||
"reasoning_config": reasoning_config,
|
||||
"max_tokens": max_tokens,
|
||||
}
|
||||
# Persist a process-scoped --yolo launch into the session row so a later
|
||||
# `hermes --resume <id>` can restore the bypass (CLI resume paths read
|
||||
# model_config.yolo_mode back via SessionDB.session_yolo_enabled).
|
||||
# Session-scoped /yolo toggles persist separately through
|
||||
# SessionDB.set_session_yolo at toggle time.
|
||||
try:
|
||||
from tools.approval import _YOLO_MODE_FROZEN
|
||||
if _YOLO_MODE_FROZEN:
|
||||
agent._session_init_model_config["yolo_mode"] = True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# In-memory todo list for task planning (one per agent/session)
|
||||
from tools.todo_tool import TodoStore
|
||||
|
||||
+131
-60
@@ -36,7 +36,7 @@ from hermes_cli.timeouts import get_provider_request_timeout
|
||||
from agent.prompt_builder import format_steer_marker
|
||||
from agent.tool_dispatch_helpers import _trajectory_normalize_msg, make_tool_result_message
|
||||
from agent.trajectory import convert_scratchpad_to_think
|
||||
from agent.credential_pool import STATUS_EXHAUSTED
|
||||
from agent.credential_pool import STATUS_EXHAUSTED, credential_pool_matches_provider
|
||||
from agent.error_classifier import FailoverReason
|
||||
from agent.turn_context import drop_stale_api_content
|
||||
from utils import base_url_host_matches, base_url_hostname, env_var_enabled, atomic_json_write
|
||||
@@ -52,6 +52,45 @@ logger = logging.getLogger(__name__)
|
||||
_MAX_AUTH_REFRESH_ATTEMPTS = 2
|
||||
|
||||
|
||||
_REASONING_TAG_NAMES = ("think", "thinking", "reasoning", "REASONING_SCRATCHPAD", "thought")
|
||||
_TOOL_CALL_TAG_NAMES = ("tool_call", "tool_calls", "tool_result", "function_call", "function_calls")
|
||||
|
||||
_REASONING_BLOCK_PATTERNS = tuple(
|
||||
re.compile(rf"<{name}>.*?</{name}>", re.DOTALL | re.IGNORECASE)
|
||||
for name in _REASONING_TAG_NAMES
|
||||
)
|
||||
|
||||
_TOOL_CALL_BLOCK_PATTERNS = tuple(
|
||||
re.compile(rf"<{name}\b[^>]*>.*?</{name}>", re.DOTALL | re.IGNORECASE)
|
||||
for name in _TOOL_CALL_TAG_NAMES
|
||||
)
|
||||
|
||||
# Named <function name=...> blocks — see strip_think_blocks step 1c for the
|
||||
# full rationale (sentence-boundary lookbehind + tempered-dot body so a plain
|
||||
# prose mention of "function" is never eaten).
|
||||
_NAMED_FUNCTION_BLOCK_PATTERN = re.compile(
|
||||
r'(?:(?<=^)|(?<=[\n\r.!?:]))[ \t]*'
|
||||
r'<function\b[^>]*\bname\s*=[^>]*>'
|
||||
r'(?:(?:(?!</function>).)*)</function>',
|
||||
re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
|
||||
_UNTERMINATED_REASONING_BLOCK_PATTERN = re.compile(
|
||||
rf'(?:^|\n)[ \t]*<(?:{"|".join(_REASONING_TAG_NAMES)})\b[^>]*>.*$',
|
||||
re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
|
||||
_ORPHAN_REASONING_TAG_PATTERN = re.compile(
|
||||
rf'</?(?:{"|".join(_REASONING_TAG_NAMES)})>\s*',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
_STRAY_TOOL_CALL_CLOSER_PATTERN = re.compile(
|
||||
rf'</(?:{"|".join(_TOOL_CALL_TAG_NAMES)}|function)>\s*',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _ra():
|
||||
"""Lazy ``run_agent`` reference for test-patch routing."""
|
||||
import run_agent
|
||||
@@ -826,62 +865,31 @@ def strip_think_blocks(agent, content: str) -> str:
|
||||
# 1. Closed tag pairs — case-insensitive for all variants so
|
||||
# mixed-case tags (<THINK>, <Thinking>) don't slip through to
|
||||
# the unterminated-tag pass and take trailing content with them.
|
||||
content = re.sub(r'<think>.*?</think>', '', content, flags=re.DOTALL | re.IGNORECASE)
|
||||
content = re.sub(r'<thinking>.*?</thinking>', '', content, flags=re.DOTALL | re.IGNORECASE)
|
||||
content = re.sub(r'<reasoning>.*?</reasoning>', '', content, flags=re.DOTALL | re.IGNORECASE)
|
||||
content = re.sub(r'<REASONING_SCRATCHPAD>.*?</REASONING_SCRATCHPAD>', '', content, flags=re.DOTALL | re.IGNORECASE)
|
||||
content = re.sub(r'<thought>.*?</thought>', '', content, flags=re.DOTALL | re.IGNORECASE)
|
||||
for _pattern in _REASONING_BLOCK_PATTERNS:
|
||||
content = _pattern.sub('', content)
|
||||
# 1b. Tool-call XML blocks (openclaw/openclaw#67318). Handle the
|
||||
# generic tag names first — they have no attribute gating since
|
||||
# a literal <tool_call> in prose is already vanishingly rare.
|
||||
for _tc_name in ("tool_call", "tool_calls", "tool_result",
|
||||
"function_call", "function_calls"):
|
||||
content = re.sub(
|
||||
rf'<{_tc_name}\b[^>]*>.*?</{_tc_name}>',
|
||||
'',
|
||||
content,
|
||||
flags=re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
for _pattern in _TOOL_CALL_BLOCK_PATTERNS:
|
||||
content = _pattern.sub('', content)
|
||||
# 1c. <function name="...">...</function> — Gemma-style standalone
|
||||
# tool call. Only strip when the tag sits at a block boundary
|
||||
# (start of text, after a newline, or after sentence-ending
|
||||
# punctuation) AND carries a name="..." attribute. This keeps
|
||||
# prose mentions like "Use <function> to declare" safe.
|
||||
content = re.sub(
|
||||
r'(?:(?<=^)|(?<=[\n\r.!?:]))[ \t]*'
|
||||
r'<function\b[^>]*\bname\s*=[^>]*>'
|
||||
r'(?:(?:(?!</function>).)*)</function>',
|
||||
'',
|
||||
content,
|
||||
flags=re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
content = _NAMED_FUNCTION_BLOCK_PATTERN.sub('', content)
|
||||
# 2. Unterminated reasoning block — open tag at a block boundary
|
||||
# (start of text, or after a newline) with no matching close.
|
||||
# Strip from the tag to end of string. Fixes #8878 / #9568
|
||||
# (MiniMax M2.7 leaking raw reasoning into assistant content).
|
||||
content = re.sub(
|
||||
r'(?:^|\n)[ \t]*<(?:think|thinking|reasoning|thought|REASONING_SCRATCHPAD)\b[^>]*>.*$',
|
||||
'',
|
||||
content,
|
||||
flags=re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
content = _UNTERMINATED_REASONING_BLOCK_PATTERN.sub('', content)
|
||||
# 3. Stray orphan open/close tags that slipped through.
|
||||
content = re.sub(
|
||||
r'</?(?:think|thinking|reasoning|thought|REASONING_SCRATCHPAD)>\s*',
|
||||
'',
|
||||
content,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
content = _ORPHAN_REASONING_TAG_PATTERN.sub('', content)
|
||||
# 3b. Stray tool-call closers. (We do NOT strip bare <function> or
|
||||
# unterminated <function name="..."> because a truncated tail
|
||||
# during streaming may still be valuable to the user; matches
|
||||
# OpenClaw's intentional asymmetry.)
|
||||
content = re.sub(
|
||||
r'</(?:tool_call|tool_calls|tool_result|function_call|function_calls|function)>\s*',
|
||||
'',
|
||||
content,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
content = _STRAY_TOOL_CALL_CLOSER_PATTERN.sub('', content)
|
||||
return content
|
||||
|
||||
|
||||
@@ -1456,6 +1464,64 @@ def restore_primary_runtime(agent) -> bool:
|
||||
if getattr(agent, "_rate_limited_until", 0) > time.monotonic():
|
||||
return False # primary still in rate-limit cooldown, stay on fallback
|
||||
|
||||
# ── Reset-aware gate ──
|
||||
# The 60s ``_rate_limited_until`` cooldown covers transient rate limits,
|
||||
# but subscription-style providers (Claude Pro/Max 5-hour windows, ChatGPT
|
||||
# weekly limits) report reset times hours or days away. The credential
|
||||
# pool already stores those timestamps (``last_error_reset_at``); until
|
||||
# the earliest one elapses, every restore attempt is a *guaranteed*
|
||||
# failure that costs two prompt-cache invalidations per turn (switch to
|
||||
# primary, fail, switch back to fallback) and re-marshals the full
|
||||
# context each way. Skip the restore while the pool says nobody can
|
||||
# serve, and come back the moment the reset time passes.
|
||||
#
|
||||
# Fail-open by design: any error (unreadable auth store, legacy pool
|
||||
# adapter without ``next_available_at``) falls through to the existing
|
||||
# every-turn retry. A pool with no reset info returns ``None`` and also
|
||||
# falls through — this gate only ever *adds* skips for provably
|
||||
# limited windows, so recovery can never be later than it is today.
|
||||
#
|
||||
# When the attached pool belongs to the fallback provider (cross-provider
|
||||
# fallback rebinds it), the primary pool is loaded here and handed to the
|
||||
# pool-rebind block below via ``prefetched_primary_pool`` so the load
|
||||
# happens at most once per restore.
|
||||
prefetched_primary_pool = None
|
||||
try:
|
||||
primary_provider = str(
|
||||
(agent._primary_runtime or {}).get("provider") or ""
|
||||
).strip().lower()
|
||||
pool = getattr(agent, "_credential_pool", None)
|
||||
if not credential_pool_matches_provider(
|
||||
pool,
|
||||
primary_provider,
|
||||
base_url=str((agent._primary_runtime or {}).get("base_url") or ""),
|
||||
):
|
||||
from agent.credential_pool import load_pool
|
||||
|
||||
prefetched_primary_pool = (
|
||||
load_pool(primary_provider) if primary_provider else None
|
||||
)
|
||||
pool = prefetched_primary_pool
|
||||
next_at = getattr(pool, "next_available_at", lambda: None)()
|
||||
if next_at is not None and next_at > time.time():
|
||||
if not getattr(agent, "_restore_wait_logged", False):
|
||||
agent._restore_wait_logged = True
|
||||
logger.info(
|
||||
"Primary %s rate-limited until %s; staying on fallback "
|
||||
"%s/%s until the reset elapses",
|
||||
primary_provider or "?",
|
||||
datetime.fromtimestamp(next_at).isoformat(timespec="seconds"),
|
||||
agent.provider,
|
||||
agent.model,
|
||||
)
|
||||
return False
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Reset-aware restore gate failed; falling back to per-turn retry",
|
||||
exc_info=True,
|
||||
)
|
||||
agent._restore_wait_logged = False
|
||||
|
||||
rt = agent._primary_runtime
|
||||
try:
|
||||
# ── Core runtime state ──
|
||||
@@ -1549,9 +1615,14 @@ def restore_primary_runtime(agent) -> bool:
|
||||
agent._credential_pool = None
|
||||
agent._credential_pool_entry_id = None
|
||||
try:
|
||||
from agent.credential_pool import load_pool
|
||||
if prefetched_primary_pool is not None:
|
||||
# Reuse the pool the reset-aware gate already loaded for
|
||||
# this restore — avoids a second disk read of auth.json.
|
||||
agent._credential_pool = prefetched_primary_pool
|
||||
else:
|
||||
from agent.credential_pool import load_pool
|
||||
|
||||
agent._credential_pool = load_pool(primary_provider)
|
||||
agent._credential_pool = load_pool(primary_provider)
|
||||
except Exception as exc:
|
||||
logger.warning(
|
||||
"Restore could not reload primary credential pool for %s: %s",
|
||||
@@ -1631,6 +1702,7 @@ def restore_primary_runtime(agent) -> bool:
|
||||
# ── Reset fallback chain for the new turn ──
|
||||
agent._fallback_activated = False
|
||||
agent._fallback_index = 0
|
||||
agent._rate_limit_backoff_count = 0 # reset exponential backoff counter
|
||||
|
||||
# Reset the stale-call circuit breaker (#58962): the streak measured
|
||||
# the FALLBACK provider we're leaving; the restored primary deserves
|
||||
@@ -2003,12 +2075,12 @@ def anthropic_prompt_cache_policy(
|
||||
gateway implements the Anthropic cache_control contract
|
||||
(MiniMax, Zhipu GLM, LiteLLM's Anthropic proxy mode all do).
|
||||
|
||||
Qwen models on OpenCode and direct Alibaba (DashScope), plus DeepSeek
|
||||
models on OpenCode, also honour Anthropic-style ``cache_control`` markers
|
||||
on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi #3393
|
||||
documented this for opencode-go Qwen; #24617 reports the same gateway
|
||||
contract for DeepSeek. Without markers these providers serve zero cache
|
||||
hits, re-billing the full prompt on every turn.
|
||||
Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct
|
||||
Alibaba (DashScope) also honour Anthropic-style ``cache_control``
|
||||
markers on OpenAI-wire chat completions. Upstream pi-mono #3392 /
|
||||
pi #3393 documented this for opencode-go Qwen. Without markers
|
||||
these providers serve zero cache hits, re-billing the full prompt
|
||||
on every turn.
|
||||
|
||||
If the operator has set ``prompt_caching.cache_ttl`` to a falsy value
|
||||
(``false``, ``null``, ``"off"``, etc.) in config.yaml, prompt caching
|
||||
@@ -2135,22 +2207,21 @@ def anthropic_prompt_cache_policy(
|
||||
if is_minimax_provider or is_minimax_host:
|
||||
return True, True
|
||||
|
||||
# Qwen on OpenCode (Zen/Go) and native DashScope, plus DeepSeek on
|
||||
# OpenCode only: OpenAI-wire transports that accept Anthropic-style
|
||||
# cache_control markers and reward them with real cache hits. Keep direct
|
||||
# Alibaba specific to Qwen; its catalog does not establish the same
|
||||
# contract for DeepSeek.
|
||||
# Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire
|
||||
# transport that accepts Anthropic-style cache_control markers and
|
||||
# rewards them with real cache hits. Without this branch
|
||||
# qwen3.6-plus on opencode-go reports 0% cached tokens and burns
|
||||
# through the subscription on every turn.
|
||||
#
|
||||
# NOTE: DeepSeek models on OpenCode are intentionally excluded.
|
||||
# OpenCode Zen's relay rejects the Anthropic-style content block
|
||||
# format that cache markers produce (content becomes a block array
|
||||
# instead of a plain string), causing HTTP 400 (#77217).
|
||||
model_is_qwen = "qwen" in model_lower
|
||||
model_is_deepseek = "deepseek" in model_lower
|
||||
provider_is_opencode = provider_lower in {
|
||||
"opencode", "opencode-zen", "opencode-go",
|
||||
}
|
||||
provider_is_alibaba_family = provider_lower in {
|
||||
"opencode", "opencode-zen", "opencode-go", "alibaba",
|
||||
}
|
||||
if (provider_is_alibaba_family and model_is_qwen) or (
|
||||
provider_is_opencode and model_is_deepseek
|
||||
):
|
||||
if provider_is_alibaba_family and model_is_qwen:
|
||||
# Envelope layout (native_anthropic=False): markers on inner
|
||||
# content parts, not top-level tool messages. Matches
|
||||
# pi-mono's "alibaba" cacheControlFormat.
|
||||
|
||||
+142
-27
@@ -1334,7 +1334,7 @@ def _resolve_anthropic_pool_token() -> Optional[str]:
|
||||
# to auth.json or trigger a network refresh from a bare resolve. select()
|
||||
# is deliberately NOT used — it runs clear_expired=True, refresh=True,
|
||||
# which would violate this read-only contract.
|
||||
entries = pool._available_entries(clear_expired=False, refresh=False)
|
||||
entries, _pending = pool._available_entries(clear_expired=False, refresh=False)
|
||||
except Exception:
|
||||
logger.debug("Failed to read Anthropic credential_pool", exc_info=True)
|
||||
return None
|
||||
@@ -1360,19 +1360,27 @@ def resolve_anthropic_token() -> Optional[str]:
|
||||
Priority:
|
||||
1. ANTHROPIC_TOKEN env var (OAuth/setup token saved by Hermes)
|
||||
2. CLAUDE_CODE_OAUTH_TOKEN env var
|
||||
3. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json)
|
||||
3. ANTHROPIC_API_KEY env var (explicit regular API key)
|
||||
4. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json)
|
||||
— with automatic refresh if expired and a refresh token is available
|
||||
4. Anthropic credential_pool OAuth entry (~/.hermes/auth.json)
|
||||
5. ANTHROPIC_API_KEY env var (regular API key, or legacy fallback)
|
||||
5. Anthropic credential_pool OAuth entry (~/.hermes/auth.json)
|
||||
|
||||
Returns the token string or None.
|
||||
"""
|
||||
creds = read_claude_code_credentials()
|
||||
creds: Optional[Dict[str, Any]] = None
|
||||
creds_loaded = False
|
||||
|
||||
def _read_creds() -> Optional[Dict[str, Any]]:
|
||||
nonlocal creds, creds_loaded
|
||||
if not creds_loaded:
|
||||
creds = read_claude_code_credentials()
|
||||
creds_loaded = True
|
||||
return creds
|
||||
|
||||
# 1. Hermes-managed OAuth/setup token env var
|
||||
token = _getenv("ANTHROPIC_TOKEN").strip()
|
||||
if token:
|
||||
preferred = _prefer_refreshable_claude_code_token(token, creds)
|
||||
preferred = _prefer_refreshable_claude_code_token(token, _read_creds())
|
||||
if preferred:
|
||||
return preferred
|
||||
return token
|
||||
@@ -1380,27 +1388,27 @@ def resolve_anthropic_token() -> Optional[str]:
|
||||
# 2. CLAUDE_CODE_OAUTH_TOKEN (used by Claude Code for setup-tokens)
|
||||
cc_token = _getenv("CLAUDE_CODE_OAUTH_TOKEN").strip()
|
||||
if cc_token:
|
||||
preferred = _prefer_refreshable_claude_code_token(cc_token, creds)
|
||||
preferred = _prefer_refreshable_claude_code_token(cc_token, _read_creds())
|
||||
if preferred:
|
||||
return preferred
|
||||
return cc_token
|
||||
|
||||
# 3. Claude Code credential file
|
||||
resolved_claude_token = _resolve_claude_code_token_from_credentials(creds)
|
||||
if resolved_claude_token:
|
||||
return resolved_claude_token
|
||||
|
||||
# 4. Hermes credential_pool OAuth entry.
|
||||
resolved_pool_token = _resolve_anthropic_pool_token()
|
||||
if resolved_pool_token:
|
||||
return resolved_pool_token
|
||||
|
||||
# 5. Regular API key, or a legacy OAuth token saved in ANTHROPIC_API_KEY.
|
||||
# This remains as a compatibility fallback for pre-migration Hermes configs.
|
||||
# 3. Regular API key. An explicit user-configured key must not be shadowed
|
||||
# by auto-discovered Claude Code or credential-pool OAuth credentials.
|
||||
api_key = _getenv("ANTHROPIC_API_KEY").strip()
|
||||
if api_key:
|
||||
return api_key
|
||||
|
||||
# 4. Claude Code credential file
|
||||
resolved_claude_token = _resolve_claude_code_token_from_credentials(_read_creds())
|
||||
if resolved_claude_token:
|
||||
return resolved_claude_token
|
||||
|
||||
# 5. Hermes credential_pool OAuth entry.
|
||||
resolved_pool_token = _resolve_anthropic_pool_token()
|
||||
if resolved_pool_token:
|
||||
return resolved_pool_token
|
||||
|
||||
return None
|
||||
|
||||
|
||||
@@ -2311,13 +2319,14 @@ def _convert_user_message(content: Any) -> Dict[str, Any]:
|
||||
"""Validate and convert a user message to anthropic format."""
|
||||
if isinstance(content, list):
|
||||
converted_blocks = _convert_content_to_anthropic(content)
|
||||
if not converted_blocks or all(
|
||||
(b.get("text") or "").strip() == ""
|
||||
for b in converted_blocks
|
||||
if isinstance(b, dict) and b.get("type") == "text"
|
||||
):
|
||||
converted_blocks = [{"type": "text", "text": "(empty message)"}]
|
||||
return {"role": "user", "content": converted_blocks}
|
||||
kept_blocks = _fix_blank_text_blocks_in_list(
|
||||
converted_blocks,
|
||||
placeholder_text="(empty message)",
|
||||
msg_index=-1,
|
||||
role="user",
|
||||
location="_convert_user_message",
|
||||
)
|
||||
return {"role": "user", "content": kept_blocks}
|
||||
else:
|
||||
if not content or (isinstance(content, str) and not content.strip()):
|
||||
content = "(empty message)"
|
||||
@@ -2620,9 +2629,114 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
|
||||
Mirror the Bedrock Converse adapter, which unconditionally prepends a
|
||||
minimal user turn when the first message is not user
|
||||
(convert_messages_to_converse).
|
||||
|
||||
The inserted text block must be non-whitespace: Anthropic separately
|
||||
rejects any text content block whose text is empty or whitespace-only
|
||||
("text content blocks must contain non-whitespace text"), so a single
|
||||
space here traded the "leading assistant turn" 400 for that one (#69512
|
||||
class). Uses the same placeholder as every other synthesized filler
|
||||
block in this module for consistency.
|
||||
"""
|
||||
if result and result[0].get("role") != "user":
|
||||
result.insert(0, {"role": "user", "content": [{"type": "text", "text": " "}]})
|
||||
result.insert(
|
||||
0, {"role": "user", "content": [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]}
|
||||
)
|
||||
|
||||
|
||||
def _fix_blank_text_blocks_in_list(
|
||||
blocks: List[Any],
|
||||
*,
|
||||
placeholder_text: str,
|
||||
msg_index: int,
|
||||
role: Any,
|
||||
location: str,
|
||||
) -> List[Any]:
|
||||
"""Drop blank/whitespace-only text blocks from ``blocks``, in place logic.
|
||||
|
||||
Non-text blocks (tool_use, tool_result, image, document, thinking, …)
|
||||
and the relative order of everything else are left untouched. A
|
||||
cache_control marker riding on a dropped block is relocated onto the
|
||||
last surviving text/tool_use block so a breakpoint is never silently
|
||||
lost. If nothing survives, a single non-blank placeholder text block
|
||||
takes the dropped blocks' place (carrying the relocated cache_control,
|
||||
if any) so the message never has empty content.
|
||||
|
||||
Returns a new list; does not mutate ``blocks``.
|
||||
"""
|
||||
kept: List[Any] = []
|
||||
relocated_cache_control = None
|
||||
for block_index, blk in enumerate(blocks):
|
||||
if (
|
||||
isinstance(blk, dict)
|
||||
and blk.get("type") == "text"
|
||||
and not (isinstance(blk.get("text"), str) and blk["text"].strip())
|
||||
):
|
||||
if isinstance(blk.get("cache_control"), dict):
|
||||
relocated_cache_control = blk["cache_control"]
|
||||
logger.warning(
|
||||
"Pre-call sanitizer: dropped blank text content block "
|
||||
"(message_index=%d role=%s location=%s block_index=%d "
|
||||
"block_type=text)",
|
||||
msg_index,
|
||||
role,
|
||||
location,
|
||||
block_index,
|
||||
)
|
||||
continue
|
||||
kept.append(blk)
|
||||
if not kept:
|
||||
placeholder: Dict[str, Any] = {"type": "text", "text": placeholder_text}
|
||||
if relocated_cache_control is not None:
|
||||
placeholder["cache_control"] = relocated_cache_control
|
||||
kept.append(placeholder)
|
||||
elif relocated_cache_control is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(kept, relocated_cache_control)
|
||||
return kept
|
||||
|
||||
|
||||
def _scrub_blank_text_blocks(result: List[Dict[str, Any]]) -> None:
|
||||
"""Final provider-boundary guard against blank Anthropic text blocks.
|
||||
|
||||
Anthropic rejects any text content block whose ``text`` is empty or
|
||||
whitespace-only with HTTP 400 ("text content blocks must contain
|
||||
non-whitespace text"). ``_convert_assistant_message``,
|
||||
``_convert_user_message`` and ``_ensure_leading_user_turn`` already
|
||||
avoid emitting these for the paths that build them, but this pass runs
|
||||
last — after every other transform in ``convert_messages_to_anthropic``
|
||||
— so a blank block from any current or future producer (including one
|
||||
nested inside a ``tool_result``'s own content list) never reaches the
|
||||
wire. Diagnostics are structural only: message index, role, content
|
||||
location, block index/type. Never logs message text, tool arguments,
|
||||
tokens, or credentials. Mutates ``result`` in place.
|
||||
"""
|
||||
for msg_index, msg in enumerate(result):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
role = msg.get("role")
|
||||
content = msg.get("content")
|
||||
if not isinstance(content, list) or not content:
|
||||
continue
|
||||
placeholder_text = _EMPTY_TEXT_PLACEHOLDER if role == "assistant" else "(empty message)"
|
||||
new_content = _fix_blank_text_blocks_in_list(
|
||||
content,
|
||||
placeholder_text=placeholder_text,
|
||||
msg_index=msg_index,
|
||||
role=role,
|
||||
location="content",
|
||||
)
|
||||
for blk in new_content:
|
||||
if not isinstance(blk, dict) or blk.get("type") != "tool_result":
|
||||
continue
|
||||
inner = blk.get("content")
|
||||
if isinstance(inner, list) and inner:
|
||||
blk["content"] = _fix_blank_text_blocks_in_list(
|
||||
inner,
|
||||
placeholder_text="(no output)",
|
||||
msg_index=msg_index,
|
||||
role=role,
|
||||
location="tool_result",
|
||||
)
|
||||
msg["content"] = new_content
|
||||
|
||||
|
||||
def convert_messages_to_anthropic(
|
||||
@@ -2686,6 +2800,7 @@ def convert_messages_to_anthropic(
|
||||
_ensure_leading_user_turn(result)
|
||||
_manage_thinking_signatures(result, base_url, model)
|
||||
_evict_old_screenshots(result)
|
||||
_scrub_blank_text_blocks(result)
|
||||
|
||||
return system, result
|
||||
|
||||
|
||||
@@ -7604,6 +7604,77 @@ def _get_task_extra_body(task: str) -> Dict[str, Any]:
|
||||
return result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Per-task concurrency limiting (#23324)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Background auxiliary work (title generation, context compression, etc.) can
|
||||
# spawn unbounded concurrent LLM calls when many sessions are active. During
|
||||
# provider incidents each call also retries / fans out across the fallback
|
||||
# chain, multiplying request volume on already-degraded endpoints. A per-task
|
||||
# semaphore caps in-flight calls so retry amplification stays bounded.
|
||||
|
||||
_aux_sync_semaphores: Dict[str, Tuple[int, threading.BoundedSemaphore]] = {}
|
||||
_aux_async_semaphores: Dict[Tuple[str, int], Tuple[int, Any]] = {}
|
||||
_aux_sem_lock = threading.Lock()
|
||||
|
||||
|
||||
def _get_task_max_concurrency(task: Optional[str]) -> Optional[int]:
|
||||
"""Return ``auxiliary.<task>.max_concurrency`` as a positive int, or None."""
|
||||
if not task or task == "vision":
|
||||
# Vision already uses this key for its encode/resize CPU worker pool;
|
||||
# its LLM calls deliberately remain concurrent.
|
||||
return None
|
||||
raw = _get_auxiliary_task_config(task).get("max_concurrency")
|
||||
if raw is None:
|
||||
return None
|
||||
try:
|
||||
value = int(raw)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return value if value > 0 else None
|
||||
|
||||
|
||||
def _acquire_sync_aux_semaphore(task: Optional[str]) -> Optional[threading.BoundedSemaphore]:
|
||||
"""Get a per-task sync semaphore, rebuilding it after a config change."""
|
||||
limit = _get_task_max_concurrency(task)
|
||||
if limit is None:
|
||||
return None
|
||||
with _aux_sem_lock:
|
||||
entry = _aux_sync_semaphores.get(task)
|
||||
if entry is None or entry[0] != limit:
|
||||
semaphore = threading.BoundedSemaphore(limit)
|
||||
_aux_sync_semaphores[task] = (limit, semaphore)
|
||||
return semaphore
|
||||
return entry[1]
|
||||
|
||||
|
||||
def _acquire_async_aux_semaphore(task: Optional[str]):
|
||||
"""Get a per-task, per-event-loop async semaphore after config lookup."""
|
||||
limit = _get_task_max_concurrency(task)
|
||||
if limit is None:
|
||||
return None
|
||||
import asyncio
|
||||
try:
|
||||
loop = asyncio.get_running_loop()
|
||||
except RuntimeError:
|
||||
return None
|
||||
key = (task, id(loop))
|
||||
with _aux_sem_lock:
|
||||
entry = _aux_async_semaphores.get(key)
|
||||
if entry is None or entry[0] != limit:
|
||||
semaphore = asyncio.Semaphore(limit)
|
||||
_aux_async_semaphores[key] = (limit, semaphore)
|
||||
return semaphore
|
||||
return entry[1]
|
||||
|
||||
|
||||
def _reset_aux_semaphores() -> None:
|
||||
"""Drop cached semaphores (test helper)."""
|
||||
with _aux_sem_lock:
|
||||
_aux_sync_semaphores.clear()
|
||||
_aux_async_semaphores.clear()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Anthropic-compatible endpoint detection + image block conversion
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -8476,6 +8547,75 @@ def call_llm(
|
||||
api_mode: str = None,
|
||||
stream: bool = False,
|
||||
stream_options: dict = None,
|
||||
) -> Any:
|
||||
"""Run an auxiliary LLM request, applying the configured task limit."""
|
||||
semaphore = _acquire_sync_aux_semaphore(task)
|
||||
if semaphore is not None:
|
||||
semaphore.acquire()
|
||||
try:
|
||||
response = _call_llm_impl(
|
||||
task=task,
|
||||
provider=provider,
|
||||
model=model,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
main_runtime=main_runtime,
|
||||
messages=messages,
|
||||
temperature=temperature,
|
||||
max_tokens=max_tokens,
|
||||
tools=tools,
|
||||
timeout=timeout,
|
||||
extra_body=extra_body,
|
||||
reasoning_config=reasoning_config,
|
||||
extra_headers=extra_headers,
|
||||
api_mode=api_mode,
|
||||
stream=stream,
|
||||
stream_options=stream_options,
|
||||
)
|
||||
if stream and semaphore is not None:
|
||||
stream_semaphore = semaphore
|
||||
semaphore = None
|
||||
return _release_sync_semaphore_after_stream(response, stream_semaphore)
|
||||
return response
|
||||
finally:
|
||||
if semaphore is not None:
|
||||
semaphore.release()
|
||||
|
||||
|
||||
def _release_sync_semaphore_after_stream(
|
||||
stream: Any, semaphore: threading.BoundedSemaphore,
|
||||
):
|
||||
"""Release a permit only after a streaming response is consumed or closed."""
|
||||
try:
|
||||
yield from stream
|
||||
finally:
|
||||
try:
|
||||
close = getattr(stream, "close", None)
|
||||
if callable(close):
|
||||
close()
|
||||
finally:
|
||||
semaphore.release()
|
||||
|
||||
|
||||
def _call_llm_impl(
|
||||
task: str = None,
|
||||
*,
|
||||
provider: str = None,
|
||||
model: str = None,
|
||||
base_url: str = None,
|
||||
api_key: str = None,
|
||||
main_runtime: Optional[Dict[str, Any]] = None,
|
||||
messages: list,
|
||||
temperature: Optional[float] = None,
|
||||
max_tokens: int = None,
|
||||
tools: list = None,
|
||||
timeout: float = None,
|
||||
extra_body: dict = None,
|
||||
reasoning_config: Optional[dict] = None,
|
||||
extra_headers: Optional[Dict[str, str]] = None,
|
||||
api_mode: str = None,
|
||||
stream: bool = False,
|
||||
stream_options: dict = None,
|
||||
) -> Any:
|
||||
"""Centralized synchronous LLM call.
|
||||
|
||||
@@ -9239,6 +9379,47 @@ async def async_call_llm(
|
||||
timeout: float = None,
|
||||
extra_body: dict = None,
|
||||
reasoning_config: Optional[dict] = None,
|
||||
) -> Any:
|
||||
"""Run an asynchronous auxiliary LLM request under the configured limit."""
|
||||
semaphore = _acquire_async_aux_semaphore(task)
|
||||
if semaphore is not None:
|
||||
await semaphore.acquire()
|
||||
try:
|
||||
return await _async_call_llm_impl(
|
||||
task=task,
|
||||
provider=provider,
|
||||
model=model,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
main_runtime=main_runtime,
|
||||
messages=messages,
|
||||
temperature=temperature,
|
||||
max_tokens=max_tokens,
|
||||
tools=tools,
|
||||
timeout=timeout,
|
||||
extra_body=extra_body,
|
||||
reasoning_config=reasoning_config,
|
||||
)
|
||||
finally:
|
||||
if semaphore is not None:
|
||||
semaphore.release()
|
||||
|
||||
|
||||
async def _async_call_llm_impl(
|
||||
task: str = None,
|
||||
*,
|
||||
provider: str = None,
|
||||
model: str = None,
|
||||
base_url: str = None,
|
||||
api_key: str = None,
|
||||
main_runtime: Optional[Dict[str, Any]] = None,
|
||||
messages: list,
|
||||
temperature: Optional[float] = None,
|
||||
max_tokens: int = None,
|
||||
tools: list = None,
|
||||
timeout: float = None,
|
||||
extra_body: dict = None,
|
||||
reasoning_config: Optional[dict] = None,
|
||||
) -> Any:
|
||||
"""Centralized asynchronous LLM call.
|
||||
|
||||
|
||||
@@ -1712,7 +1712,20 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool
|
||||
current_provider = (getattr(agent, "provider", "") or "").strip().lower()
|
||||
primary_provider = ((agent._primary_runtime or {}).get("provider") or "").strip().lower()
|
||||
if (not fallback_already_active) or (primary_provider and current_provider == primary_provider):
|
||||
agent._rate_limited_until = time.monotonic() + 60
|
||||
# Exponential backoff: keep upstream's 60s first-hit cooldown and
|
||||
# escalate on CONSECUTIVE rate-limits: 60s → 2m → 4m → 8m → ... →
|
||||
# 4h cap. The first 429 must NOT bench the primary for half an
|
||||
# hour — fast primary restore is the common case; escalation only
|
||||
# punishes providers that keep 429ing.
|
||||
# Counter is reset by restore_primary_runtime on successful restore.
|
||||
backoff_count = getattr(agent, "_rate_limit_backoff_count", 0)
|
||||
agent._rate_limit_backoff_count = backoff_count + 1
|
||||
backoff_seconds = min(60 * (2 ** backoff_count), 14400)
|
||||
agent._rate_limited_until = time.monotonic() + backoff_seconds
|
||||
logging.info(
|
||||
"Rate-limit backoff level %d: cooldown %d s (%.1f min, backoff#%d)",
|
||||
backoff_count, backoff_seconds, backoff_seconds / 60, backoff_count + 1,
|
||||
)
|
||||
if agent._fallback_index >= len(agent._fallback_chain):
|
||||
# Chain exhausted. If we actually walked a non-empty chain and the
|
||||
# failure was NOT a rate-limit/billing event (those already armed
|
||||
|
||||
@@ -6730,6 +6730,26 @@ This compaction should PRIORITISE preserving all information related to the focu
|
||||
_strip_persistence_markers(compressed)
|
||||
self._last_compression_made_progress = True
|
||||
|
||||
# A successful compaction just freed the largest allocation a long
|
||||
# session ever drops (the compressed-away message dicts), which makes
|
||||
# this the natural point to hand allocator pages back to the OS.
|
||||
# #76905's trim lifecycle covers the gateway/TUI housekeeping loops but
|
||||
# not the CLI compression path, so RSS keeps the pre-compaction
|
||||
# high-water mark until exit. The helper is glibc-gated, config-gated
|
||||
# and rate-limited, so this is a safe no-op elsewhere. (#70782)
|
||||
try:
|
||||
from hermes_cli.mem_trim import trim_memory
|
||||
|
||||
trim_memory(reason="post-compression")
|
||||
except Exception as exc:
|
||||
# debug, not warning: sibling trim sites all log failures at
|
||||
# debug, and compression must never fail because of a trim.
|
||||
logger.debug(
|
||||
"post-compression memory trim failed: %s: %s",
|
||||
type(exc).__name__,
|
||||
exc,
|
||||
)
|
||||
|
||||
# Batch compaction invalidates micro-compaction state: the batch
|
||||
# marker now holds MORE history than the in-memory rolling summary
|
||||
# (it summarized everything in the window, including exchanges micro
|
||||
|
||||
+109
-8
@@ -58,6 +58,11 @@ from agent.message_sanitization import (
|
||||
_strip_images_from_messages,
|
||||
_strip_non_ascii,
|
||||
)
|
||||
# Must mirror _STALE_TOOL_CALL_MARKER_RE in hermes_state.py — kept local
|
||||
# to avoid importing hermes_state at module load time (its module-level
|
||||
# DEFAULT_DB_PATH = get_hermes_home() / "state.db" breaks tests that
|
||||
# monkeypatch get_hermes_home to return a str).
|
||||
_STALE_MARKER_RE = re.compile(r"^\[[A-Za-z_][A-Za-z0-9_.-]*\]$")
|
||||
from agent.model_metadata import (
|
||||
MINIMUM_CONTEXT_LENGTH,
|
||||
_estimate_tools_tokens_rough,
|
||||
@@ -4360,9 +4365,11 @@ def run_conversation(
|
||||
compression_attempts += 1
|
||||
if compression_attempts <= max_compression_attempts:
|
||||
original_len = len(messages)
|
||||
# Option A (LCM issue 441): overhead-aware request size so recovery arms on
|
||||
# the true request (msgs + tools + system), not the tool-blind message count.
|
||||
messages, active_system_prompt = agent._compress_context(
|
||||
messages, system_message,
|
||||
approx_tokens=approx_tokens,
|
||||
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
|
||||
task_id=effective_task_id,
|
||||
)
|
||||
conversation_history = conversation_history_after_compression(
|
||||
@@ -4617,8 +4624,11 @@ def run_conversation(
|
||||
original_len = len(messages)
|
||||
original_tokens = estimate_messages_tokens_rough(messages)
|
||||
_overflow_input = messages
|
||||
# Option A (LCM issue 441): overhead-aware request size so recovery arms on the
|
||||
# true request (msgs + tools + system), not the tool-blind message count.
|
||||
messages, active_system_prompt = agent._compress_context(
|
||||
messages, system_message, approx_tokens=approx_tokens,
|
||||
messages, system_message,
|
||||
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
|
||||
task_id=effective_task_id,
|
||||
)
|
||||
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
|
||||
@@ -4754,6 +4764,41 @@ def run_conversation(
|
||||
"failed": True,
|
||||
"compression_exhausted": True,
|
||||
}
|
||||
# Also compress the message history so the output-cap
|
||||
# retry does not just spin on max_tokens alone. The
|
||||
# compressor drops the middle window, freeing enough
|
||||
# tokens for the total to fit inside context_length.
|
||||
# (#55546)
|
||||
try:
|
||||
original_len = len(messages)
|
||||
original_tokens = estimate_messages_tokens_rough(messages)
|
||||
_overflow_input = messages
|
||||
messages, active_system_prompt = agent._compress_context(
|
||||
messages, system_message,
|
||||
approx_tokens=request_input_estimate,
|
||||
task_id=effective_task_id,
|
||||
)
|
||||
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
|
||||
compression_attempts -= 1
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return _compression_deferred_result(
|
||||
agent, messages, api_call_count
|
||||
)
|
||||
conversation_history = conversation_history_after_compression(
|
||||
agent, messages, conversation_history
|
||||
)
|
||||
new_tokens = estimate_messages_tokens_rough(messages)
|
||||
if len(messages) < original_len:
|
||||
agent._buffer_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages)))
|
||||
elif new_tokens > 0 and new_tokens < original_tokens * 0.95:
|
||||
agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens))
|
||||
except Exception:
|
||||
# Compression must never turn an output-cap error
|
||||
# fatal — fall through and retry on max_tokens alone.
|
||||
logger.warning(
|
||||
"%sOutput-cap compression hit an error; retrying on max_tokens only.",
|
||||
agent.log_prefix,
|
||||
)
|
||||
_retry.restart_with_compressed_messages = True
|
||||
break
|
||||
|
||||
@@ -4878,8 +4923,13 @@ def run_conversation(
|
||||
original_len = len(messages)
|
||||
original_tokens = estimate_messages_tokens_rough(messages)
|
||||
_overflow_input = messages
|
||||
# Option A (LCM issue 441): pass the OVERHEAD-AWARE request size (msgs + tool
|
||||
# schemas + system), not the tool-blind message count, so LCM forced-overflow
|
||||
# recovery arms on the TRUE request that overflowed. See hermes-lcm engine
|
||||
# _should_force_overflow_recovery. (approx_tokens stays for the status display.)
|
||||
messages, active_system_prompt = agent._compress_context(
|
||||
messages, system_message, approx_tokens=approx_tokens,
|
||||
messages, system_message,
|
||||
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
|
||||
task_id=effective_task_id,
|
||||
)
|
||||
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
|
||||
@@ -6107,9 +6157,25 @@ def run_conversation(
|
||||
]
|
||||
|
||||
assistant_msg = agent._build_assistant_message(assistant_message, finish_reason)
|
||||
|
||||
|
||||
turn_content = assistant_message.content or ""
|
||||
|
||||
# Some local tool-call templates emit a bare bracketed token
|
||||
# (for example ``[memory]``) as assistant content alongside a
|
||||
# function call. It is protocol scaffolding, not an answer.
|
||||
# Persisting or caching it as visible content lets the empty
|
||||
# post-tool fallback replay that token forever after compaction (#78148).
|
||||
if (
|
||||
assistant_message.tool_calls
|
||||
and _STALE_MARKER_RE.fullmatch(turn_content.strip())
|
||||
):
|
||||
logger.warning(
|
||||
"Discarding bare tool-call marker from assistant content: %s",
|
||||
turn_content,
|
||||
)
|
||||
turn_content = ""
|
||||
assistant_msg["content"] = ""
|
||||
|
||||
# Classify tools in this turn to determine if they are all housekeeping.
|
||||
# This classification is needed regardless of whether the turn has visible content,
|
||||
# because a substantive tool-only turn must invalidate any older housekeeping fallback.
|
||||
@@ -6373,9 +6439,12 @@ def run_conversation(
|
||||
_clear_warn()
|
||||
agent._safe_print(" ⟳ compacting context…")
|
||||
_post_tool_input = messages
|
||||
# Route the overhead-aware _real_tokens (computed above) into compression, not
|
||||
# the bare last_prompt_tokens — which is 0 in the no-usage fallback, hiding the
|
||||
# true request size from the engine's overflow guard (upstream PR #77169 review).
|
||||
messages, active_system_prompt = agent._compress_context(
|
||||
messages, system_message,
|
||||
approx_tokens=agent.context_compressor.last_prompt_tokens,
|
||||
approx_tokens=_real_tokens,
|
||||
task_id=effective_task_id,
|
||||
)
|
||||
if (
|
||||
@@ -6662,15 +6731,47 @@ def run_conversation(
|
||||
)
|
||||
if _truly_empty and (not _has_structured or _prefill_exhausted) and agent._empty_content_retries < 3:
|
||||
agent._empty_content_retries += 1
|
||||
wait_time = jittered_backoff(
|
||||
agent._empty_content_retries,
|
||||
base_delay=5.0,
|
||||
max_delay=60.0,
|
||||
)
|
||||
logger.warning(
|
||||
"Empty response (no content or reasoning) — "
|
||||
"retry %d/3 (model=%s)",
|
||||
agent._empty_content_retries, agent.model,
|
||||
"retry %d/3 in %.1fs (model=%s)",
|
||||
agent._empty_content_retries, wait_time, agent.model,
|
||||
)
|
||||
agent._buffer_status(
|
||||
f"⚠️ Empty response from model — retrying "
|
||||
f"({agent._empty_content_retries}/3)"
|
||||
f"({agent._empty_content_retries}/3) in {wait_time:.0f}s"
|
||||
)
|
||||
# Sleep in small increments to stay responsive to interrupts
|
||||
sleep_end = time.time() + wait_time
|
||||
_backoff_touch_counter = 0
|
||||
while time.time() < sleep_end:
|
||||
if agent._interrupt_requested:
|
||||
agent._vprint(f"{agent.log_prefix}⚡ Interrupt detected during empty-response retry wait, aborting.", force=True)
|
||||
_interrupt_text = (
|
||||
f"Operation interrupted: retrying empty response from model "
|
||||
f"(retry {agent._empty_content_retries}/3)."
|
||||
)
|
||||
close_interrupted_tool_sequence(messages, _interrupt_text)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
agent.clear_interrupt()
|
||||
return {
|
||||
"final_response": _interrupt_text,
|
||||
"messages": messages,
|
||||
"api_calls": api_call_count,
|
||||
"completed": False,
|
||||
"interrupted": True,
|
||||
}
|
||||
time.sleep(0.2)
|
||||
_backoff_touch_counter += 1
|
||||
if _backoff_touch_counter % 150 == 0: # 150 × 0.2s = 30s
|
||||
agent._touch_activity(
|
||||
f"empty response retry backoff ({agent._empty_content_retries}/3), "
|
||||
f"{int(sleep_end - time.time())}s remaining"
|
||||
)
|
||||
continue
|
||||
|
||||
# ── Exhausted retries — try fallback provider ──
|
||||
|
||||
+252
-95
@@ -588,7 +588,12 @@ class CredentialPool:
|
||||
self._entries = sorted(entries, key=lambda entry: entry.priority)
|
||||
self._current_id: Optional[str] = None
|
||||
self._strategy = get_pool_strategy(provider)
|
||||
self._lock = threading.Lock()
|
||||
# RLock: the mutation primitives below (_replace_entry/_persist)
|
||||
# self-acquire this lock so the DEFERRED single-use-token refresh
|
||||
# path (which runs network I/O outside the lock by design) still
|
||||
# serializes its pool mutations. In-lock callers re-acquire
|
||||
# reentrantly at negligible cost.
|
||||
self._lock = threading.RLock()
|
||||
self._active_leases: Dict[str, int] = {}
|
||||
self._max_concurrent = DEFAULT_MAX_CONCURRENT_PER_CREDENTIAL
|
||||
# Monotonic timestamp of the last "no available entries" log, used to
|
||||
@@ -618,7 +623,36 @@ class CredentialPool:
|
||||
# otherwise a status probe here can race a concurrent ``select`` /
|
||||
# rotation and tear ``self._entries`` or double-write auth.json.
|
||||
with self._lock:
|
||||
return bool(self._available_entries())
|
||||
available, _pending = self._available_entries()
|
||||
return bool(available)
|
||||
|
||||
def next_available_at(self) -> Optional[float]:
|
||||
"""Earliest epoch time (seconds) any entry re-enters rotation.
|
||||
|
||||
Returns ``None`` when at least one entry is available right now, or
|
||||
when no exhausted entry carries a usable recovery time (empty pool,
|
||||
or only ``STATUS_DEAD`` entries, which never re-enter via TTL).
|
||||
Callers must treat ``None`` as "no wait information", not
|
||||
"unavailable".
|
||||
|
||||
Like :meth:`has_available`, expired cooldowns are left uncleared
|
||||
(``clear_expired=False``); the only writes are the same
|
||||
re-auth/token sync paths ``has_available`` already performs — which
|
||||
is exactly why this must run under ``self._lock`` like every other
|
||||
``_available_entries`` caller (see the comment on ``has_available``).
|
||||
"""
|
||||
with self._lock:
|
||||
available, _pending = self._available_entries()
|
||||
if available:
|
||||
return None
|
||||
candidates: List[float] = []
|
||||
for entry in self._entries:
|
||||
if entry.last_status != STATUS_EXHAUSTED:
|
||||
continue
|
||||
until = _exhausted_until(entry)
|
||||
if until is not None:
|
||||
candidates.append(until)
|
||||
return min(candidates) if candidates else None
|
||||
|
||||
def entries(self) -> List[PooledCredential]:
|
||||
with self._lock:
|
||||
@@ -656,18 +690,27 @@ class CredentialPool:
|
||||
return matches[0].id if len(matches) == 1 else None
|
||||
|
||||
def _replace_entry(self, old: PooledCredential, new: PooledCredential) -> None:
|
||||
"""Swap an entry in-place by id, preserving sort order."""
|
||||
for idx, entry in enumerate(self._entries):
|
||||
if entry.id == old.id:
|
||||
self._entries[idx] = new
|
||||
return
|
||||
"""Swap an entry in-place by id, preserving sort order.
|
||||
|
||||
Self-locking (RLock) so the deferred refresh path — which
|
||||
deliberately runs outside the pool lock — cannot tear
|
||||
``self._entries`` against a concurrent select()/rotation.
|
||||
"""
|
||||
with self._lock:
|
||||
for idx, entry in enumerate(self._entries):
|
||||
if entry.id == old.id:
|
||||
self._entries[idx] = new
|
||||
return
|
||||
|
||||
def _persist(self, *, removed_ids: Optional[List[str]] = None) -> None:
|
||||
write_credential_pool(
|
||||
self.provider,
|
||||
[entry.to_dict() for entry in self._entries],
|
||||
removed_ids=removed_ids,
|
||||
)
|
||||
# Self-locking (RLock): snapshotting self._entries must not race a
|
||||
# concurrent rotation when called from the deferred refresh path.
|
||||
with self._lock:
|
||||
write_credential_pool(
|
||||
self.provider,
|
||||
[entry.to_dict() for entry in self._entries],
|
||||
removed_ids=removed_ids,
|
||||
)
|
||||
|
||||
def _is_terminal_auth_failure(
|
||||
self,
|
||||
@@ -1415,17 +1458,22 @@ class CredentialPool:
|
||||
logger.debug(
|
||||
"Failed to clear terminal xAI OAuth state: %s", clear_exc
|
||||
)
|
||||
removed_ids = [
|
||||
item.id for item in self._entries
|
||||
if item.source == "device_code"
|
||||
]
|
||||
self._entries = [
|
||||
item for item in self._entries
|
||||
if item.source != "device_code"
|
||||
]
|
||||
if self._current_id == entry.id:
|
||||
self._current_id = None
|
||||
self._persist(removed_ids=removed_ids)
|
||||
# Read-modify-write of self._entries: must be atomic.
|
||||
# This runs on the DEFERRED refresh path (outside the
|
||||
# pool lock), so take it here. self._lock is an RLock,
|
||||
# so the still-locked callers re-enter safely.
|
||||
with self._lock:
|
||||
removed_ids = [
|
||||
item.id for item in self._entries
|
||||
if item.source == "device_code"
|
||||
]
|
||||
self._entries = [
|
||||
item for item in self._entries
|
||||
if item.source != "device_code"
|
||||
]
|
||||
if self._current_id == entry.id:
|
||||
self._current_id = None
|
||||
self._persist(removed_ids=removed_ids)
|
||||
return None
|
||||
# For openai-codex: same race as xAI/nous — another Hermes process
|
||||
# may have consumed the refresh token between our proactive sync
|
||||
@@ -1485,17 +1533,22 @@ class CredentialPool:
|
||||
logger.debug(
|
||||
"Failed to clear terminal Codex OAuth state: %s", clear_exc
|
||||
)
|
||||
removed_ids = [
|
||||
item.id for item in self._entries
|
||||
if item.source == "device_code"
|
||||
]
|
||||
self._entries = [
|
||||
item for item in self._entries
|
||||
if item.source != "device_code"
|
||||
]
|
||||
if self._current_id == entry.id:
|
||||
self._current_id = None
|
||||
self._persist(removed_ids=removed_ids)
|
||||
# Read-modify-write of self._entries: must be atomic.
|
||||
# This runs on the DEFERRED refresh path (outside the
|
||||
# pool lock), so take it here. self._lock is an RLock,
|
||||
# so the still-locked callers re-enter safely.
|
||||
with self._lock:
|
||||
removed_ids = [
|
||||
item.id for item in self._entries
|
||||
if item.source == "device_code"
|
||||
]
|
||||
self._entries = [
|
||||
item for item in self._entries
|
||||
if item.source != "device_code"
|
||||
]
|
||||
if self._current_id == entry.id:
|
||||
self._current_id = None
|
||||
self._persist(removed_ids=removed_ids)
|
||||
return None
|
||||
# For nous: another process may have consumed the refresh token
|
||||
# between our proactive sync and the HTTP call. Re-sync from
|
||||
@@ -1552,17 +1605,19 @@ class CredentialPool:
|
||||
auth_mod.NOUS_DEVICE_CODE_SOURCE,
|
||||
f"manual:{auth_mod.NOUS_DEVICE_CODE_SOURCE}",
|
||||
}
|
||||
removed_ids = [
|
||||
item.id for item in self._entries
|
||||
if item.source in singleton_sources
|
||||
]
|
||||
self._entries = [
|
||||
item for item in self._entries
|
||||
if item.source not in singleton_sources
|
||||
]
|
||||
if self._current_id == entry.id:
|
||||
self._current_id = None
|
||||
self._persist(removed_ids=removed_ids)
|
||||
# Atomic read-modify-write; see the note above.
|
||||
with self._lock:
|
||||
removed_ids = [
|
||||
item.id for item in self._entries
|
||||
if item.source in singleton_sources
|
||||
]
|
||||
self._entries = [
|
||||
item for item in self._entries
|
||||
if item.source not in singleton_sources
|
||||
]
|
||||
if self._current_id == entry.id:
|
||||
self._current_id = None
|
||||
self._persist(removed_ids=removed_ids)
|
||||
return None
|
||||
self._mark_exhausted(entry, None)
|
||||
return None
|
||||
@@ -1646,26 +1701,64 @@ class CredentialPool:
|
||||
return False
|
||||
|
||||
def select(self) -> Optional[PooledCredential]:
|
||||
with self._lock:
|
||||
entry = self._select_unlocked()
|
||||
if entry is not None:
|
||||
# A normal (non-recovery) selection starts a fresh episode —
|
||||
# don't let a leftover unmatched-rotation streak from an old
|
||||
# failure trip the #70401 bound early next time.
|
||||
self._unmatched_rotation_streak = 0
|
||||
entry, pending_refresh = self._select_under_lock()
|
||||
if pending_refresh:
|
||||
self._refresh_pending_entries(pending_refresh)
|
||||
if entry is not None:
|
||||
self._unmatched_rotation_streak = 0
|
||||
return entry
|
||||
# If no entry was available but we just refreshed some, re-select
|
||||
# now that the refreshed entries are back in the pool.
|
||||
if pending_refresh:
|
||||
entry, _ = self._select_under_lock()
|
||||
if entry is not None:
|
||||
self._unmatched_rotation_streak = 0
|
||||
return entry
|
||||
|
||||
def _available_entries(self, *, clear_expired: bool = False, refresh: bool = False) -> List[PooledCredential]:
|
||||
"""Return entries not currently in exhaustion cooldown.
|
||||
def _select_under_lock(self) -> Tuple[Optional[PooledCredential], List[tuple]]:
|
||||
"""Run selection under the lock, returning entry + pending refreshes."""
|
||||
with self._lock:
|
||||
return self._select_unlocked()
|
||||
|
||||
def _refresh_pending_entries(self, pending: List[tuple]) -> None:
|
||||
"""Refresh deferred single-use-token entries outside the lock.
|
||||
|
||||
Each entry is refreshed under the cross-process ``_auth_store_lock``
|
||||
(which can block for 20+ seconds) and then merged into the pool.
|
||||
On failure the entry is silently skipped.
|
||||
"""
|
||||
for entry, sync_fn in pending:
|
||||
# _refresh_entry merges the refreshed entry into the pool
|
||||
# internally. Its mutation primitives (_replace_entry, _persist)
|
||||
# are self-locking, and the quarantine paths inside
|
||||
# _refresh_entry_impl take self._lock explicitly around their
|
||||
# read-modify-write of self._entries — required because this
|
||||
# call site runs OUTSIDE the pool lock.
|
||||
self._refresh_entry(entry, force=False)
|
||||
|
||||
def _available_entries(
|
||||
self, *, clear_expired: bool = False, refresh: bool = False,
|
||||
) -> Tuple[List[PooledCredential], List[tuple]]:
|
||||
"""Return (available, pending_refresh) for entries not in cooldown.
|
||||
|
||||
When *clear_expired* is True, entries whose cooldown has elapsed are
|
||||
reset to STATUS_OK and persisted. When *refresh* is True, entries
|
||||
that need a token refresh are refreshed (skipped on failure).
|
||||
|
||||
Single-use-token refreshes (openai-codex, xai-oauth) are returned as
|
||||
*pending_refresh* tuples so the caller can execute them outside the
|
||||
lock, avoiding stalling all pool consumers during cross-process flock
|
||||
acquisition + OAuth network I/O.
|
||||
"""
|
||||
now = time.time()
|
||||
cleared_any = False
|
||||
entries_to_prune: List[str] = []
|
||||
available: List[PooledCredential] = []
|
||||
# Entries that need an OAuth refresh via a single-use token provider
|
||||
# (openai-codex, xai-oauth). These require a cross-process file lock
|
||||
# that can block for 20+ seconds. We collect them under self._lock
|
||||
# and refresh outside the lock to avoid stalling all pool consumers.
|
||||
pending_refresh: List[tuple] = [] # (entry, sync_entry_fn)
|
||||
for entry in self._entries:
|
||||
# Borrowed credentials persist as metadata-only references and are
|
||||
# hydrated from their live source on load. A stale duplicate row
|
||||
@@ -1774,6 +1867,16 @@ class CredentialPool:
|
||||
entry = cleared
|
||||
cleared_any = True
|
||||
if refresh and self._entry_needs_refresh(entry):
|
||||
if self.provider in ("openai-codex", "xai-oauth"):
|
||||
# Defer single-use-token refresh to avoid holding the
|
||||
# threading lock during cross-process flock + network I/O.
|
||||
sync_fn = (
|
||||
self._sync_codex_entry_from_auth_store
|
||||
if self.provider == "openai-codex"
|
||||
else self._sync_xai_oauth_entry_from_pool_store
|
||||
)
|
||||
pending_refresh.append((entry, sync_fn))
|
||||
continue
|
||||
refreshed = self._refresh_entry(entry, force=False)
|
||||
if refreshed is None:
|
||||
continue
|
||||
@@ -1784,7 +1887,7 @@ class CredentialPool:
|
||||
self._entries = [e for e in self._entries if e.id not in pruned_ids]
|
||||
if cleared_any:
|
||||
self._persist(removed_ids=entries_to_prune)
|
||||
return available
|
||||
return available, pending_refresh
|
||||
|
||||
def _log_no_available_entries(self) -> None:
|
||||
"""Emit the empty-pool INFO line at most once per throttle window.
|
||||
@@ -1800,12 +1903,17 @@ class CredentialPool:
|
||||
self._last_no_entries_log_at = now
|
||||
logger.info("credential pool: no available entries (all exhausted or empty)")
|
||||
|
||||
def _select_unlocked(self, *, refresh: bool = True) -> Optional[PooledCredential]:
|
||||
available = self._available_entries(clear_expired=True, refresh=refresh)
|
||||
def _select_unlocked(self, *, refresh: bool = True) -> Tuple[Optional[PooledCredential], List[tuple]]:
|
||||
"""Select the best available credential entry.
|
||||
|
||||
Returns ``(entry, pending_refresh)`` where *pending_refresh* contains
|
||||
single-use-token entries that must be refreshed outside the lock.
|
||||
"""
|
||||
available, pending_refresh = self._available_entries(clear_expired=True, refresh=refresh)
|
||||
if not available:
|
||||
self._current_id = None
|
||||
self._log_no_available_entries()
|
||||
return None
|
||||
return None, pending_refresh
|
||||
|
||||
# A successful selection means the pool recovered; re-arm the throttle
|
||||
# so a later re-exhaustion logs immediately rather than being silenced
|
||||
@@ -1815,7 +1923,7 @@ class CredentialPool:
|
||||
if self._strategy == STRATEGY_RANDOM:
|
||||
entry = random.choice(available)
|
||||
self._current_id = entry.id
|
||||
return entry
|
||||
return entry, pending_refresh
|
||||
|
||||
if self._strategy == STRATEGY_LEAST_USED and len(available) > 1:
|
||||
entry = min(available, key=lambda e: e.request_count)
|
||||
@@ -1823,7 +1931,7 @@ class CredentialPool:
|
||||
updated = replace(entry, request_count=entry.request_count + 1)
|
||||
self._replace_entry(entry, updated)
|
||||
self._current_id = entry.id
|
||||
return updated
|
||||
return updated, pending_refresh
|
||||
|
||||
if self._strategy == STRATEGY_ROUND_ROBIN and len(available) > 1:
|
||||
entry = available[0]
|
||||
@@ -1832,11 +1940,11 @@ class CredentialPool:
|
||||
self._entries = [replace(candidate, priority=idx) for idx, candidate in enumerate(rotated)]
|
||||
self._persist()
|
||||
self._current_id = entry.id
|
||||
return self._current_unlocked() or entry
|
||||
return self._current_unlocked() or entry, pending_refresh
|
||||
|
||||
entry = available[0]
|
||||
self._current_id = entry.id
|
||||
return entry
|
||||
return entry, pending_refresh
|
||||
|
||||
def peek(self) -> Optional[PooledCredential]:
|
||||
# Single lock acquisition for the whole read; call the unlocked
|
||||
@@ -1845,7 +1953,7 @@ class CredentialPool:
|
||||
current = self._current_unlocked()
|
||||
if current is not None:
|
||||
return current
|
||||
available = self._available_entries()
|
||||
available, _pending = self._available_entries()
|
||||
return available[0] if available else None
|
||||
|
||||
def mark_exhausted_and_rotate(
|
||||
@@ -1894,7 +2002,8 @@ class CredentialPool:
|
||||
# guessing and surface the error (no cooldown is written for
|
||||
# anybody — healthy keys stay available for the next turn).
|
||||
self._unmatched_rotation_streak += 1
|
||||
available_count = len(self._available_entries())
|
||||
available_count, _ = self._available_entries()
|
||||
available_count = len(available_count)
|
||||
if self._unmatched_rotation_streak > max(available_count, 1):
|
||||
logger.warning(
|
||||
"credential pool: failed credential identity matched no "
|
||||
@@ -1913,8 +2022,9 @@ class CredentialPool:
|
||||
self.provider,
|
||||
)
|
||||
self._current_id = None
|
||||
next_entry = self._select_unlocked()
|
||||
if next_entry is not None and len(self._available_entries()) == 1:
|
||||
next_entry, _pending = self._select_unlocked(refresh=False)
|
||||
avail, _ = self._available_entries()
|
||||
if next_entry is not None and len(avail) == 1:
|
||||
# A single-entry pool cannot rotate. Returning its only
|
||||
# entry reports a successful recovery without changing
|
||||
# the credential, so the caller retries the same 401
|
||||
@@ -1927,7 +2037,7 @@ class CredentialPool:
|
||||
# streak is stale (this mark WILL advance pool state).
|
||||
self._unmatched_rotation_streak = 0
|
||||
if entry is None:
|
||||
entry = self._current_unlocked() or self._select_unlocked()
|
||||
entry = self._current_unlocked() or self._select_unlocked(refresh=False)[0]
|
||||
if entry is None:
|
||||
return None
|
||||
_label = entry.label or entry.id[:8]
|
||||
@@ -1972,7 +2082,7 @@ class CredentialPool:
|
||||
_label, status_code,
|
||||
)
|
||||
self._current_id = None
|
||||
next_entry = self._select_unlocked()
|
||||
next_entry, _pending = self._select_unlocked(refresh=False)
|
||||
if next_entry:
|
||||
_next_label = next_entry.label or next_entry.id[:8]
|
||||
logger.info("credential pool: rotated to %s", _next_label)
|
||||
@@ -1986,15 +2096,32 @@ class CredentialPool:
|
||||
a stable tie-breaker. When every credential is already at the soft cap,
|
||||
still return the least-leased one instead of blocking.
|
||||
"""
|
||||
chosen_id, pending_refresh = self._acquire_lease_under_lock(credential_id)
|
||||
if pending_refresh:
|
||||
self._refresh_pending_entries(pending_refresh)
|
||||
# Mirror select(): if nothing was leasable but we just refreshed
|
||||
# deferred single-use-token entries, retry now that they are back
|
||||
# in rotation. Without this, a pool whose only entries all needed
|
||||
# a refresh returns None even though the refresh succeeded — the
|
||||
# caller sees "no credentials available" and fails a request that
|
||||
# should have gone through.
|
||||
if chosen_id is None:
|
||||
chosen_id, _ = self._acquire_lease_under_lock(credential_id)
|
||||
return chosen_id
|
||||
|
||||
def _acquire_lease_under_lock(
|
||||
self, credential_id: Optional[str],
|
||||
) -> Tuple[Optional[str], List[tuple]]:
|
||||
"""Run lease acquisition under the lock, returning id + pending refreshes."""
|
||||
with self._lock:
|
||||
if credential_id:
|
||||
self._active_leases[credential_id] = self._active_leases.get(credential_id, 0) + 1
|
||||
self._current_id = credential_id
|
||||
return credential_id
|
||||
return credential_id, []
|
||||
|
||||
available = self._available_entries(clear_expired=True, refresh=True)
|
||||
available, pending_refresh = self._available_entries(clear_expired=True, refresh=True)
|
||||
if not available:
|
||||
return None
|
||||
return None, pending_refresh
|
||||
|
||||
below_cap = [
|
||||
entry for entry in available
|
||||
@@ -2007,7 +2134,7 @@ class CredentialPool:
|
||||
)
|
||||
self._active_leases[chosen.id] = self._active_leases.get(chosen.id, 0) + 1
|
||||
self._current_id = chosen.id
|
||||
return chosen.id
|
||||
return chosen.id, pending_refresh
|
||||
|
||||
def release_lease(self, credential_id: str) -> None:
|
||||
"""Release a previously acquired credential lease."""
|
||||
@@ -2059,7 +2186,7 @@ class CredentialPool:
|
||||
else:
|
||||
entry = self._current_unlocked() or self._select_unlocked(
|
||||
refresh=False
|
||||
)
|
||||
)[0]
|
||||
if entry is None:
|
||||
return None
|
||||
self._current_id = entry.id
|
||||
@@ -2387,9 +2514,41 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup
|
||||
# env vars (COPILOT_GITHUB_TOKEN / GH_TOKEN). They don't live in
|
||||
# the auth store or credential pool, so we resolve them here.
|
||||
try:
|
||||
from hermes_cli.copilot_auth import resolve_copilot_token, get_copilot_api_token
|
||||
from hermes_cli.copilot_auth import (
|
||||
COPILOT_ENV_VARS,
|
||||
resolve_copilot_token,
|
||||
get_copilot_api_token,
|
||||
)
|
||||
# All-sources suppression gate BEFORE any work — including the
|
||||
# `gh auth token` subprocess spawn. resolve_copilot_token()
|
||||
# shells out (~30ms), and the exchange retries 3x with backoff
|
||||
# (~35s worst case); a user who suppressed every copilot source
|
||||
# (hermes auth remove copilot gh_cli) must not pay either on
|
||||
# every pool load (model picker open, /model, agent startup).
|
||||
# Enumerating the full source space here matches what
|
||||
# credential_sources._remove_copilot_gh suppresses, so an
|
||||
# all-suppressed check is stable.
|
||||
copilot_sources = ["gh_cli"] + [f"env:{v}" for v in COPILOT_ENV_VARS]
|
||||
if all(_is_suppressed(provider, s) for s in copilot_sources):
|
||||
return changed, active_sources
|
||||
token, source = resolve_copilot_token()
|
||||
if token:
|
||||
# ``resolve_copilot_token`` returns exactly "gh auth token"
|
||||
# for the CLI path; env-sourced tokens return the var name.
|
||||
# Match exactly — a substring test classifies GH_TOKEN and
|
||||
# GITHUB_TOKEN as gh_cli, silently bypassing a user's
|
||||
# per-env-var suppression.
|
||||
source_name = "gh_cli" if source == "gh auth token" else f"env:{source}"
|
||||
# Per-source suppression gate (a user may suppress only the
|
||||
# gh CLI path and keep an env var, or vice versa) BEFORE the
|
||||
# network exchange. The exchange retries 3x with 10s
|
||||
# timeouts and 4.5s total backoff (~35s worst case), so a
|
||||
# source the user already suppressed
|
||||
# must not burn that dead time just to have the entry
|
||||
# discarded afterwards. Same early-gate pattern every other
|
||||
# singleton branch uses.
|
||||
if _is_suppressed(provider, source_name):
|
||||
return changed, active_sources
|
||||
api_token, enterprise_base_url = get_copilot_api_token(token)
|
||||
# Observability: get_copilot_api_token falls back to returning
|
||||
# the RAW token when the exchange fails. A raw ~40-char token
|
||||
@@ -2405,27 +2564,25 @@ def _seed_from_singletons(provider: str, entries: List[PooledCredential]) -> Tup
|
||||
"unavailable); enterprise-only models may 400 with "
|
||||
"model_not_available_for_integrator until exchange recovers."
|
||||
)
|
||||
source_name = "gh_cli" if "gh" in source.lower() else f"env:{source}"
|
||||
if not _is_suppressed(provider, source_name):
|
||||
active_sources.add(source_name)
|
||||
pconfig = PROVIDER_REGISTRY.get(provider)
|
||||
# Use enterprise base URL from token exchange if available,
|
||||
# otherwise fall back to the provider's default.
|
||||
effective_base_url = enterprise_base_url or (
|
||||
pconfig.inference_base_url if pconfig else ""
|
||||
)
|
||||
changed |= _upsert_entry(
|
||||
entries,
|
||||
provider,
|
||||
source_name,
|
||||
{
|
||||
"source": source_name,
|
||||
"auth_type": AUTH_TYPE_API_KEY,
|
||||
"access_token": api_token,
|
||||
"base_url": effective_base_url,
|
||||
"label": source,
|
||||
},
|
||||
)
|
||||
active_sources.add(source_name)
|
||||
pconfig = PROVIDER_REGISTRY.get(provider)
|
||||
# Use enterprise base URL from token exchange if available,
|
||||
# otherwise fall back to the provider's default.
|
||||
effective_base_url = enterprise_base_url or (
|
||||
pconfig.inference_base_url if pconfig else ""
|
||||
)
|
||||
changed |= _upsert_entry(
|
||||
entries,
|
||||
provider,
|
||||
source_name,
|
||||
{
|
||||
"source": source_name,
|
||||
"auth_type": AUTH_TYPE_API_KEY,
|
||||
"access_token": api_token,
|
||||
"base_url": effective_base_url,
|
||||
"label": source,
|
||||
},
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("Copilot token seed failed: %s", exc)
|
||||
|
||||
|
||||
@@ -1923,6 +1923,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
|
||||
credential_pool=_credential_pool,
|
||||
request_overrides=_request_overrides,
|
||||
**_agent_kwargs,
|
||||
enabled_toolsets=["skills", "terminal"],
|
||||
# Umbrella-building over a large skill collection is worth a
|
||||
# high iteration ceiling — the pass typically takes 50-100
|
||||
# API calls against hundreds of candidate skills. The
|
||||
|
||||
+39
-2
@@ -14,6 +14,7 @@ from dataclasses import dataclass, field
|
||||
from difflib import unified_diff
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from utils import safe_json_loads
|
||||
from agent.redact import redact_sensitive_text
|
||||
@@ -187,6 +188,15 @@ def _truncate_preview(text: str, max_len: int | None) -> str:
|
||||
return text
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolPreview:
|
||||
"""A compact tool preview plus presentation facts lost to truncation."""
|
||||
|
||||
text: str
|
||||
truncated: bool = False
|
||||
url: str | None = None
|
||||
|
||||
|
||||
_SHELL_SILENT_HEADS = {"cd", "pushd", "popd", "export", "set", "unset", "source", ".", "true", "false", ":"}
|
||||
_SHELL_PIPE_TAIL_HEADS = {"head", "tail", "wc", "sort", "uniq"}
|
||||
|
||||
@@ -556,6 +566,35 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -
|
||||
return preview
|
||||
|
||||
|
||||
def prepare_tool_preview(
|
||||
tool_name: str,
|
||||
args: dict | None,
|
||||
*,
|
||||
fallback: str,
|
||||
max_len: int,
|
||||
) -> ToolPreview:
|
||||
"""Build one canonical compact preview before platform formatting.
|
||||
|
||||
The uncapped preview is rebuilt from the tool arguments when possible so
|
||||
an upstream display cap cannot discard its link target. Platforms then
|
||||
receive explicit truncation and URL metadata instead of inferring either
|
||||
fact from the rendered text.
|
||||
"""
|
||||
full_text = build_tool_preview(tool_name, args, max_len=0) or fallback
|
||||
text = _truncate_preview(full_text, max_len)
|
||||
truncated = text != full_text
|
||||
url = None
|
||||
if truncated:
|
||||
candidate = _display_url(full_text)
|
||||
try:
|
||||
parsed = urlsplit(candidate)
|
||||
except ValueError:
|
||||
parsed = None
|
||||
if parsed and parsed.scheme.lower() in {"http", "https"} and parsed.netloc:
|
||||
url = candidate
|
||||
return ToolPreview(text=text, truncated=truncated, url=url)
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Friendly tool labels (human-phrased verbs for built-in tools)
|
||||
#
|
||||
@@ -1506,5 +1545,3 @@ def get_cute_tool_message(
|
||||
# =========================================================================
|
||||
# Honcho session line (one-liner with clickable OSC 8 hyperlink)
|
||||
# =========================================================================
|
||||
|
||||
|
||||
|
||||
+91
-28
@@ -99,6 +99,31 @@ class InsightsEngine:
|
||||
"""
|
||||
self.db = db
|
||||
self._conn = db._conn
|
||||
# INDEXED BY is a hard dependency (SQLite errors on a missing index).
|
||||
# A read-only open of a state.db written by an older version skips
|
||||
# schema init and lacks the partial index — probe once and fall back
|
||||
# to the unpinned variants (identical rows, optimizer-chosen plan).
|
||||
try:
|
||||
self._has_assistant_calls_index = bool(
|
||||
self._conn.execute(
|
||||
"SELECT 1 FROM sqlite_master WHERE type='index' AND name=?",
|
||||
(self._MESSAGES_ASSISTANT_CALLS_INDEX,),
|
||||
).fetchone()
|
||||
)
|
||||
except sqlite3.Error:
|
||||
self._has_assistant_calls_index = False
|
||||
if not self._has_assistant_calls_index:
|
||||
_strip = f" INDEXED BY {self._MESSAGES_ASSISTANT_CALLS_INDEX}"
|
||||
# Loop over every pinned statement so adding a new one can't
|
||||
# forget its strip line (which would be a hard `no such index`
|
||||
# crash on read-only DBs — the exact bug this fallback prevents).
|
||||
for _attr in (
|
||||
"_GET_TOOL_CALLS_WITH_SOURCE",
|
||||
"_GET_TOOL_CALLS_ALL",
|
||||
"_GET_SKILL_CALLS_WITH_SOURCE",
|
||||
"_GET_SKILL_CALLS_ALL",
|
||||
):
|
||||
setattr(self, _attr, getattr(self, _attr).replace(_strip, ""))
|
||||
|
||||
def generate(self, days: int = 30, source: str = None) -> Dict[str, Any]:
|
||||
"""
|
||||
@@ -171,6 +196,21 @@ class InsightsEngine:
|
||||
"top_sessions": top_sessions,
|
||||
}
|
||||
|
||||
def get_usage_breakdown(self, days: int = 30, source: str = None) -> Dict[str, Any]:
|
||||
"""Return the analytics-usage payload without running a full generate().
|
||||
|
||||
Uses the instr()-prefiltered _get_skill_usage query so only messages
|
||||
that reference skill_view or skill_manage are loaded from SQLite, while
|
||||
still preserving the per-tool breakdown used by the dashboard route.
|
||||
"""
|
||||
cutoff = time.time() - (days * 86400)
|
||||
tool_usage = self._get_tool_usage(cutoff, source)
|
||||
skill_usage = self._get_skill_usage(cutoff, source)
|
||||
return {
|
||||
"tools": self._compute_tool_breakdown(tool_usage),
|
||||
"skills": self._compute_skill_breakdown(skill_usage),
|
||||
}
|
||||
|
||||
# =========================================================================
|
||||
# Data gathering (SQL queries)
|
||||
# =========================================================================
|
||||
@@ -195,6 +235,53 @@ class InsightsEngine:
|
||||
" ORDER BY started_at DESC"
|
||||
)
|
||||
|
||||
# Assistant ``tool_calls`` scan for tool/skill usage. ``INDEXED BY`` pins
|
||||
# the partial index ``idx_messages_assistant_calls_by_session`` so the plan
|
||||
# is deterministic on a freshly initialized state.db (before ANALYZE has
|
||||
# run) for BOTH the unfiltered and source-filtered branches — without the
|
||||
# hint the optimizer falls back to ``idx_messages_session_active`` for the
|
||||
# source-filtered probe and scans each session's non-tool-call rows.
|
||||
#
|
||||
# The pin is a HARD dependency: SQLite raises ``no such index`` when the
|
||||
# named index is absent. That happens in practice — the web dashboard's
|
||||
# usage analytics open the DB ``read_only=True`` (skipping
|
||||
# ``_init_schema``), so a state.db created by an older writer has no
|
||||
# partial index yet. ``__init__`` probes for the index once and falls
|
||||
# back to the unpinned (still-correct, just optimizer-chosen) variants.
|
||||
_MESSAGES_ASSISTANT_CALLS_INDEX = "idx_messages_assistant_calls_by_session"
|
||||
_GET_TOOL_CALLS_WITH_SOURCE = (
|
||||
"SELECT m.tool_calls"
|
||||
f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}"
|
||||
" JOIN sessions s ON s.id = m.session_id"
|
||||
" WHERE s.started_at >= ? AND s.source = ?"
|
||||
" AND m.role = 'assistant' AND m.tool_calls IS NOT NULL"
|
||||
)
|
||||
_GET_TOOL_CALLS_ALL = (
|
||||
"SELECT m.tool_calls"
|
||||
f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}"
|
||||
" JOIN sessions s ON s.id = m.session_id"
|
||||
" WHERE s.started_at >= ?"
|
||||
" AND m.role = 'assistant' AND m.tool_calls IS NOT NULL"
|
||||
)
|
||||
_GET_SKILL_CALLS_WITH_SOURCE = (
|
||||
"SELECT m.tool_calls, m.timestamp"
|
||||
f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}"
|
||||
" JOIN sessions s ON s.id = m.session_id"
|
||||
" WHERE s.started_at >= ? AND s.source = ?"
|
||||
" AND m.role = 'assistant' AND m.tool_calls IS NOT NULL"
|
||||
" AND (instr(m.tool_calls, 'skill_view') > 0"
|
||||
" OR instr(m.tool_calls, 'skill_manage') > 0)"
|
||||
)
|
||||
_GET_SKILL_CALLS_ALL = (
|
||||
"SELECT m.tool_calls, m.timestamp"
|
||||
f" FROM messages m INDEXED BY {_MESSAGES_ASSISTANT_CALLS_INDEX}"
|
||||
" JOIN sessions s ON s.id = m.session_id"
|
||||
" WHERE s.started_at >= ?"
|
||||
" AND m.role = 'assistant' AND m.tool_calls IS NOT NULL"
|
||||
" AND (instr(m.tool_calls, 'skill_view') > 0"
|
||||
" OR instr(m.tool_calls, 'skill_manage') > 0)"
|
||||
)
|
||||
|
||||
def _get_sessions(self, cutoff: float, source: str = None) -> List[Dict]:
|
||||
"""Fetch sessions within the time window."""
|
||||
if source:
|
||||
@@ -243,22 +330,10 @@ class InsightsEngine:
|
||||
# (covers CLI sessions where tool_name is NULL on tool responses)
|
||||
if source:
|
||||
cursor2 = self._conn.execute(
|
||||
"""SELECT m.tool_calls
|
||||
FROM messages m
|
||||
JOIN sessions s ON s.id = m.session_id
|
||||
WHERE s.started_at >= ? AND s.source = ?
|
||||
AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""",
|
||||
(cutoff, source),
|
||||
self._GET_TOOL_CALLS_WITH_SOURCE, (cutoff, source)
|
||||
)
|
||||
else:
|
||||
cursor2 = self._conn.execute(
|
||||
"""SELECT m.tool_calls
|
||||
FROM messages m
|
||||
JOIN sessions s ON s.id = m.session_id
|
||||
WHERE s.started_at >= ?
|
||||
AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""",
|
||||
(cutoff,),
|
||||
)
|
||||
cursor2 = self._conn.execute(self._GET_TOOL_CALLS_ALL, (cutoff,))
|
||||
|
||||
tool_calls_counts = Counter()
|
||||
for row in cursor2.fetchall():
|
||||
@@ -301,22 +376,10 @@ class InsightsEngine:
|
||||
|
||||
if source:
|
||||
cursor = self._conn.execute(
|
||||
"""SELECT m.tool_calls, m.timestamp
|
||||
FROM messages m
|
||||
JOIN sessions s ON s.id = m.session_id
|
||||
WHERE s.started_at >= ? AND s.source = ?
|
||||
AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""",
|
||||
(cutoff, source),
|
||||
self._GET_SKILL_CALLS_WITH_SOURCE, (cutoff, source)
|
||||
)
|
||||
else:
|
||||
cursor = self._conn.execute(
|
||||
"""SELECT m.tool_calls, m.timestamp
|
||||
FROM messages m
|
||||
JOIN sessions s ON s.id = m.session_id
|
||||
WHERE s.started_at >= ?
|
||||
AND m.role = 'assistant' AND m.tool_calls IS NOT NULL""",
|
||||
(cutoff,),
|
||||
)
|
||||
cursor = self._conn.execute(self._GET_SKILL_CALLS_ALL, (cutoff,))
|
||||
|
||||
for row in cursor.fetchall():
|
||||
try:
|
||||
|
||||
@@ -34,12 +34,50 @@ Optional hooks (override to opt in):
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Prompts that carry no semantic signal — trivial acknowledgements, greetings,
|
||||
# slash commands, empty input. Single source of truth shared by the core
|
||||
# per-turn prefetch gate (agent/turn_context.py, run_agent.py) and provider-
|
||||
# side classifiers (plugins/memory/honcho) so the two can never drift apart.
|
||||
# The alternation is anchored and may only be followed by whitespace or
|
||||
# punctuation, so words that merely START with a trivial word ("k8s", "yolo",
|
||||
# "note", "hindsight") do NOT match, while trailing-punctuation variants
|
||||
# ("hi!", "hey.", "thanks :)", "done???") do.
|
||||
TRIVIAL_PROMPT_RE = re.compile(
|
||||
r'^(yes|no|ok|okay|sure|thanks|thank you|y|n|yep|nope|yeah|nah|'
|
||||
r'hi|hey|hello|yo|sup|'
|
||||
r'continue|go ahead|do it|proceed|got it|cool|nice|great|done|next|lgtm|k)'
|
||||
r'[\s!?.:;,"' + "'" + r'~\u2018\u2019\u201c\u201d\u2014\u2013\u2026()\[\]{}<>*&^%$#@!+=`\u00a0]*$',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def is_trivial_prompt(text: Optional[str]) -> bool:
|
||||
"""Return True if a user prompt is too trivial to warrant memory recall.
|
||||
|
||||
Empty/whitespace-only input, slash commands, and bare greetings or
|
||||
acknowledgements (with optional trailing punctuation) all count as
|
||||
trivial. Callers use this to skip memory-provider prefetch/injection
|
||||
on turns that carry no semantic signal — saving a blocking network
|
||||
round-trip and preventing stale user-model context from derailing
|
||||
one-word replies.
|
||||
"""
|
||||
if not text:
|
||||
return True
|
||||
stripped = text.strip()
|
||||
if not stripped:
|
||||
return True
|
||||
if stripped.startswith("/"):
|
||||
return True
|
||||
return bool(TRIVIAL_PROMPT_RE.match(stripped))
|
||||
|
||||
|
||||
class MemoryProvider(ABC):
|
||||
"""Abstract base class for memory providers."""
|
||||
|
||||
@@ -253,6 +291,10 @@ class MemoryProvider(ABC):
|
||||
required: True if required (default: False)
|
||||
default: default value (optional)
|
||||
choices: list of valid values (optional)
|
||||
type: text, integer, number, or boolean (optional)
|
||||
minimum: numeric lower bound for integer/number fields (optional)
|
||||
maximum: numeric upper bound for integer/number fields (optional)
|
||||
step: numeric input step for Dashboard rendering (optional)
|
||||
url: URL where user can get this credential (optional)
|
||||
env_var: explicit env var name for secrets (default: auto-generated)
|
||||
|
||||
|
||||
+68
-19
@@ -12,6 +12,7 @@ import hashlib
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor, wait as _futures_wait
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
@@ -151,6 +152,24 @@ def _redact_trace_accounting(acct: Any) -> Any:
|
||||
)
|
||||
|
||||
|
||||
# Cold-start caches. A MoA preset switch used to re-resolve the full
|
||||
# config + preset + every slot's provider runtime on EACH create() call
|
||||
# (once per tool-loop iteration), serially before the parallel fan-out could
|
||||
# start — adding 5-30s of "frozen" latency on complex presets
|
||||
# (#66793). The preset structure is immutable for the life of a turn, so
|
||||
# cache both the resolved preset and each (provider, model) runtime.
|
||||
_preset_cache_lock = threading.Lock()
|
||||
_preset_cache: dict[tuple, Any] = {}
|
||||
|
||||
_runtime_cache_lock = threading.Lock()
|
||||
_runtime_cache: dict[tuple[str, str], tuple[float, dict[str, Any]]] = {}
|
||||
|
||||
# Runtime entries go stale when providers/credentials change (key rotation,
|
||||
# base_url edits). Deliberately short-lived: 300s collapses the per-iteration
|
||||
# re-resolution inside a turn while bounding credential staleness between
|
||||
# turns — the non-MoA path picks up rotated keys immediately, this path
|
||||
# within 5 minutes.
|
||||
_RUNTIME_CACHE_TTL_SECONDS = 300.0
|
||||
|
||||
# Upper bound on concurrent reference-model calls. References are independent
|
||||
# advisory calls (no tools, no inter-dependence), so we fan them out the same
|
||||
@@ -322,34 +341,33 @@ def _slot_runtime(slot: dict[str, Any]) -> dict[str, Any]:
|
||||
api_key resolver the CLI, gateway, and delegate_task all use), so the slot
|
||||
gets its provider's real API surface — e.g. MiniMax → anthropic_messages,
|
||||
GPT-5/o-series → max_completion_tokens, custom endpoints → their base_url.
|
||||
|
||||
Returns the kwargs to pass through to ``call_llm`` (provider/model plus the
|
||||
resolved base_url/api_key when available). Falls back to the bare
|
||||
provider/model on any resolution error so a misconfigured slot still
|
||||
attempts the call rather than aborting the whole MoA turn.
|
||||
|
||||
The resolved runtime is cached per (provider, model) with a short TTL
|
||||
(``_RUNTIME_CACHE_TTL_SECONDS``): the resolution does real I/O (catalog
|
||||
query + config read) that used to run serially per create() call before
|
||||
the parallel fan-out could start — the dominant source of MoA cold-start
|
||||
latency (#66793). The TTL bounds credential staleness (key rotation,
|
||||
base_url edits) instead of caching for the process lifetime.
|
||||
"""
|
||||
provider = str(slot.get("provider") or "").strip()
|
||||
model = str(slot.get("model") or "").strip()
|
||||
cache_key = (provider, model)
|
||||
now = time.monotonic()
|
||||
with _runtime_cache_lock:
|
||||
entry = _runtime_cache.get(cache_key)
|
||||
if entry is not None:
|
||||
stamped_at, cached = entry
|
||||
if now - stamped_at < _RUNTIME_CACHE_TTL_SECONDS:
|
||||
return cached
|
||||
out: dict[str, Any] = {"provider": provider, "model": model}
|
||||
try:
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
|
||||
rt = resolve_runtime_provider(requested=provider, target_model=model)
|
||||
# Forward the resolved endpoint through to call_llm unconditionally.
|
||||
# call_llm's _resolve_task_provider_model() is the single chokepoint that
|
||||
# decides whether an explicit base_url collapses a call to the generic
|
||||
# ``custom`` route or keeps the provider's real identity: it preserves
|
||||
# identity for any first-class provider (via
|
||||
# _preserve_provider_with_base_url, a provider-catalog capability check),
|
||||
# so provider branches that add auth refresh / request metadata /
|
||||
# request-shape adapters — anthropic OAuth (Bearer + anthropic-beta),
|
||||
# openai-codex Responses wrapping + Cloudflare headers, xai-oauth,
|
||||
# bedrock SigV4 signing, nous Portal tags — still fire. Those branches
|
||||
# re-resolve their own credentials by name and ignore a forwarded
|
||||
# base_url/api_key, so forwarding is safe even for a placeholder key
|
||||
# (bedrock's "aws-sdk"). We used to maintain a name-preservation set here
|
||||
# too; that duplicated the chokepoint and drifted out of sync, so the
|
||||
# single source of truth now lives in call_llm.
|
||||
if rt.get("base_url"):
|
||||
out["base_url"] = rt["base_url"]
|
||||
if rt.get("api_key"):
|
||||
@@ -362,7 +380,14 @@ def _slot_runtime(slot: dict[str, Any]) -> dict[str, Any]:
|
||||
if isinstance(extra_body, dict) and extra_body:
|
||||
out["extra_body"] = dict(extra_body)
|
||||
except Exception as exc: # pragma: no cover - defensive
|
||||
logger.debug("MoA slot runtime resolution failed for %s: %s", _slot_label(slot), exc)
|
||||
logger.debug("MoA slot runtime resolution failed for %s: %s",
|
||||
_slot_label(slot), exc)
|
||||
# Never cache a fallback-shaped result: a transient resolution error
|
||||
# (config mid-write, catalog hiccup) would otherwise pin the bare
|
||||
# provider/model kwargs for a full TTL.
|
||||
return out
|
||||
with _runtime_cache_lock:
|
||||
_runtime_cache[cache_key] = (now, out)
|
||||
return out
|
||||
|
||||
|
||||
@@ -1830,11 +1855,35 @@ class MoAChatCompletions:
|
||||
raise TypeError("_moa_prepared_request must be a dict")
|
||||
return self._call_prepared_aggregator(prepared_request, api_kwargs)
|
||||
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import get_config_path, load_config
|
||||
from hermes_cli.moa_config import resolve_moa_preset
|
||||
|
||||
# Resolve the preset once per (config st_mtime_ns, preset_name).
|
||||
# resolve_moa_preset re-normalizes + re-validates the whole moa
|
||||
# config block on every call, and create() runs once per tool-loop
|
||||
# iteration — a serial cold-start cost before the parallel fan-out
|
||||
# can begin (#66793). Keyed on the config FILE's mtime_ns (not a
|
||||
# config-object attribute, which load_config()'s dicts don't carry),
|
||||
# so a config edit invalidates on the next call.
|
||||
try:
|
||||
_cfg_stamp = get_config_path().stat().st_mtime_ns
|
||||
except OSError:
|
||||
_cfg_stamp = None
|
||||
# load_config() is itself (mtime_ns, size)-cached upstream, so this
|
||||
# read is cheap; the expensive part this cache skips is
|
||||
# resolve_moa_preset's re-normalization + re-validation.
|
||||
_moa_raw = load_config().get("moa") or {}
|
||||
preset = resolve_moa_preset(_moa_raw, self.preset_name)
|
||||
preset_cache_key = (_cfg_stamp, self.preset_name)
|
||||
preset = None
|
||||
if _cfg_stamp is not None:
|
||||
with _preset_cache_lock:
|
||||
preset = _preset_cache.get(preset_cache_key)
|
||||
if preset is None:
|
||||
preset = resolve_moa_preset(_moa_raw, self.preset_name)
|
||||
if _cfg_stamp is not None:
|
||||
with _preset_cache_lock:
|
||||
_preset_cache.clear() # one live config stamp at a time
|
||||
_preset_cache[preset_cache_key] = preset
|
||||
# Privacy filter mode: '' (off, default) | 'display' | 'full'. See
|
||||
# coerce_privacy_filter / the pattern block at the top of this module.
|
||||
# Remembered on self so _call_prepared_aggregator (which may run on a
|
||||
|
||||
+181
-18
@@ -145,6 +145,96 @@ _ENDPOINT_MODEL_CACHE_TTL = 300
|
||||
_ENDPOINT_PROBE_TTL_SECONDS = 3600.0
|
||||
_endpoint_probe_path_cache: Dict[str, tuple] = {}
|
||||
|
||||
# A configured endpoint that is routable-but-dead — e.g. a corp LAN address
|
||||
# while off-VPN — blackholes TCP: the SYN draws no SYN-ACK, no RST and no ICMP
|
||||
# error, so a probe waits out its full timeout instead of failing fast. Startup
|
||||
# runs a whole waterfall of such probes across several functions here, and the
|
||||
# stalls stack into a minute-long hang before the banner renders.
|
||||
#
|
||||
# Once ANY probe has actually observed a connect timeout for an endpoint, the
|
||||
# others have nothing to gain by repeating it. Recording that observation and
|
||||
# short-circuiting on it performs no network I/O of its own — it adds no probe
|
||||
# for callers or tests to mock, and it can only ever fire after a real timeout
|
||||
# has already been paid, so it cannot suppress a probe that would have worked.
|
||||
_ENDPOINT_BLACKHOLE_TTL_SECONDS = 30.0
|
||||
# Values are monotonic timestamps of the last observed connect timeout.
|
||||
_endpoint_blackhole_cache: Dict[str, float] = {}
|
||||
|
||||
|
||||
def _endpoint_host_key(base_url: str) -> Optional[str]:
|
||||
"""Return a ``host:port`` key for ``base_url``, or None if it has no host.
|
||||
|
||||
Keyed on host:port rather than the full URL so every probe path for one
|
||||
server — ``/v1``-suffixed or not, LM Studio root or API root — shares a
|
||||
single entry.
|
||||
"""
|
||||
normalized = _normalize_base_url(base_url)
|
||||
if not normalized:
|
||||
return None
|
||||
url = normalized if "://" in normalized else f"http://{normalized}"
|
||||
try:
|
||||
parsed = urlparse(url)
|
||||
host = parsed.hostname
|
||||
port = parsed.port or (443 if parsed.scheme == "https" else 80)
|
||||
except Exception:
|
||||
return None
|
||||
return f"{host}:{port}" if host else None
|
||||
|
||||
|
||||
def _note_endpoint_blackholed(base_url: str) -> None:
|
||||
"""Record that a probe to ``base_url`` timed out during TCP connect."""
|
||||
key = _endpoint_host_key(base_url)
|
||||
if key is None:
|
||||
return
|
||||
_endpoint_blackhole_cache[key] = time.monotonic()
|
||||
logger.debug(
|
||||
"Endpoint %s timed out connecting — skipping further probes for %.0fs",
|
||||
key, _ENDPOINT_BLACKHOLE_TTL_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def _endpoint_blackholed(base_url: str) -> bool:
|
||||
"""True if a recent probe to ``base_url`` timed out during TCP connect.
|
||||
|
||||
Pure cache lookup; never touches the network. The entry expires after
|
||||
_ENDPOINT_BLACKHOLE_TTL_SECONDS — long enough to collapse one startup's
|
||||
burst of probes, short enough that bringing the VPN up mid-session is
|
||||
picked up without a restart.
|
||||
"""
|
||||
if _ENDPOINT_BLACKHOLE_TTL_SECONDS <= 0:
|
||||
return False
|
||||
key = _endpoint_host_key(base_url)
|
||||
if key is None:
|
||||
return False
|
||||
seen = _endpoint_blackhole_cache.get(key)
|
||||
if seen is None:
|
||||
return False
|
||||
if (time.monotonic() - seen) >= _ENDPOINT_BLACKHOLE_TTL_SECONDS:
|
||||
del _endpoint_blackhole_cache[key]
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _is_connect_timeout(exc: BaseException) -> bool:
|
||||
"""True for connect-phase timeouts raised by httpx or requests.
|
||||
|
||||
Read timeouts are deliberately excluded: those mean the server accepted
|
||||
the connection, which is the opposite of the blackhole this guards.
|
||||
"""
|
||||
try:
|
||||
import httpx
|
||||
if isinstance(exc, httpx.ConnectTimeout):
|
||||
return True
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from requests.exceptions import ConnectTimeout
|
||||
if isinstance(exc, ConnectTimeout):
|
||||
return True
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
# ── Disk L2 for local-endpoint probe results ────────────────────────────────
|
||||
# The in-process caches above die with the process, so every CLI cold start
|
||||
# with a local model re-paid the probe waterfall in AIAgent.__init__:
|
||||
@@ -382,6 +472,7 @@ DEFAULT_CONTEXT_LENGTHS = {
|
||||
"llama": 131072,
|
||||
# Qwen — specific model families before the catch-all.
|
||||
# Official docs: https://help.aliyun.com/zh/model-studio/developer-reference/
|
||||
"qwen3.8-max": 1_000_000, # 1M context (OpenRouter & Nous portal, verified 2026-08-03)
|
||||
"qwen3.6-plus": 1048576, # 1M context (DashScope/Alibaba & OpenRouter)
|
||||
"qwen3.7-plus": 1048576, # 1M context (DashScope/Alibaba)
|
||||
"qwen3-coder-plus": 1000000, # 1M context
|
||||
@@ -836,7 +927,10 @@ def _localhost_to_ipv4(url: str) -> str:
|
||||
``http://localhost...`` (e.g. ``?upstream=http://localhost:11434``)
|
||||
passes through untouched.
|
||||
"""
|
||||
if not url:
|
||||
if not url or not isinstance(url, str):
|
||||
# Non-string values (test doubles, lazily-resolved config objects)
|
||||
# previously flowed through these call sites untouched — keep that
|
||||
# contract; re.sub would raise TypeError.
|
||||
return url
|
||||
return re.sub(
|
||||
r"^(https?://)localhost(?=[:/]|$)",
|
||||
@@ -873,6 +967,13 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
if cached is not None and (time.monotonic() - cached[1]) < _ENDPOINT_PROBE_TTL_SECONDS:
|
||||
return cached[0]
|
||||
|
||||
# The host already blackholed a connect: skip the waterfall below, each leg
|
||||
# of which would otherwise burn its full 2s timeout. Deliberately NOT
|
||||
# written to _endpoint_probe_path_cache — that entry lives for an hour,
|
||||
# which would pin the endpoint to "undetected" long after it comes back.
|
||||
if _endpoint_blackholed(server_url):
|
||||
return None
|
||||
|
||||
# Disk L2: a fresh cross-process verdict skips the HTTP waterfall
|
||||
# entirely (back-to-back CLI invocations, cron ticks).
|
||||
disk_hit = _local_probe_disk_get("server_type", server_url)
|
||||
@@ -882,6 +983,16 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
|
||||
headers = _auth_headers(api_key)
|
||||
|
||||
def _probe_failed(exc: Exception) -> None:
|
||||
"""Swallow a probe error — or abort the waterfall if we were blackholed.
|
||||
|
||||
Re-raising propagates out of the ``with`` block to the outer handler,
|
||||
so the remaining legs are skipped instead of each stalling in turn.
|
||||
"""
|
||||
if _is_connect_timeout(exc):
|
||||
_note_endpoint_blackholed(server_url)
|
||||
raise exc
|
||||
|
||||
result: Optional[str] = None
|
||||
try:
|
||||
with httpx.Client(timeout=2.0, headers=headers) as client:
|
||||
@@ -890,8 +1001,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
r = client.get(f"{lmstudio_url}/api/v1/models")
|
||||
if r.status_code == 200:
|
||||
result = "lm-studio"
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
_probe_failed(exc)
|
||||
if result is None:
|
||||
# Ollama exposes /api/tags and responds with {"models": [...]}
|
||||
# LM Studio returns {"error": "Unexpected endpoint"} with status 200
|
||||
@@ -905,8 +1016,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
result = "ollama"
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
_probe_failed(exc)
|
||||
if result is None:
|
||||
# llama.cpp exposes /v1/props (older builds used /props without the /v1 prefix)
|
||||
try:
|
||||
@@ -915,8 +1026,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
r = client.get(f"{server_url}/props") # fallback for older builds
|
||||
if r.status_code == 200 and "default_generation_settings" in r.text:
|
||||
result = "llamacpp"
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
_probe_failed(exc)
|
||||
if result is None:
|
||||
# vLLM: /version
|
||||
try:
|
||||
@@ -925,8 +1036,8 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
data = r.json()
|
||||
if "version" in data:
|
||||
result = "vllm"
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
_probe_failed(exc)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@@ -1120,6 +1231,12 @@ def fetch_endpoint_model_metadata(
|
||||
if cached is not None and (time.time() - cached_at) < _ENDPOINT_MODEL_CACHE_TTL:
|
||||
return cached
|
||||
|
||||
# Blackholed endpoint: every candidate below would spend its full 5s
|
||||
# connect budget. Returned empty rather than cached, so the endpoint is
|
||||
# retried as soon as the blackhole entry expires.
|
||||
if _endpoint_blackholed(normalized):
|
||||
return {}
|
||||
|
||||
candidates = [normalized]
|
||||
if normalized.endswith("/v1"):
|
||||
alternate = normalized[:-3].rstrip("/")
|
||||
@@ -1182,9 +1299,20 @@ def fetch_endpoint_model_metadata(
|
||||
return cache
|
||||
except Exception as exc:
|
||||
last_error = exc
|
||||
if _is_connect_timeout(exc):
|
||||
_note_endpoint_blackholed(normalized)
|
||||
|
||||
for candidate in candidates:
|
||||
url = candidate.rstrip("/") + "/models"
|
||||
# A connect timeout on one candidate condemns the host, not the path:
|
||||
# the remaining candidates differ only by URL suffix, so trying them
|
||||
# would repeat the same stall.
|
||||
if _endpoint_blackholed(normalized):
|
||||
break
|
||||
# normalized/candidates stay unrewritten (cache key stability); only
|
||||
# the outbound request target is IPv4-resolved to skip the multi-second
|
||||
# dual-stack IPv6 connect timeout (see _localhost_to_ipv4).
|
||||
request_candidate = _localhost_to_ipv4(candidate)
|
||||
url = request_candidate.rstrip("/") + "/models"
|
||||
response = None
|
||||
try:
|
||||
response = requests.get(
|
||||
@@ -1230,7 +1358,7 @@ def fetch_endpoint_model_metadata(
|
||||
if is_llamacpp:
|
||||
try:
|
||||
# Try /v1/props first (current llama.cpp); fall back to /props for older builds
|
||||
base = candidate.rstrip("/").replace("/v1", "")
|
||||
base = request_candidate.rstrip("/").replace("/v1", "")
|
||||
_verify = _resolve_requests_verify()
|
||||
props_resp = requests.get(base + "/v1/props", headers=headers, timeout=5, verify=_verify)
|
||||
if not props_resp.ok:
|
||||
@@ -1250,6 +1378,8 @@ def fetch_endpoint_model_metadata(
|
||||
return cache
|
||||
except Exception as exc:
|
||||
last_error = exc
|
||||
if _is_connect_timeout(exc):
|
||||
_note_endpoint_blackholed(normalized)
|
||||
finally:
|
||||
if response is not None:
|
||||
response.close()
|
||||
@@ -1813,6 +1943,9 @@ def _query_ollama_api_show_uncached(model: str, base_url: str, api_key: str = ""
|
||||
if server_url.endswith("/v1"):
|
||||
server_url = server_url[:-3]
|
||||
|
||||
if _endpoint_blackholed(server_url):
|
||||
return None
|
||||
|
||||
headers = _auth_headers(api_key)
|
||||
|
||||
try:
|
||||
@@ -1844,8 +1977,9 @@ def _query_ollama_api_show_uncached(model: str, base_url: str, api_key: str = ""
|
||||
return ctx
|
||||
except ValueError:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
if _is_connect_timeout(exc):
|
||||
_note_endpoint_blackholed(server_url)
|
||||
return None
|
||||
|
||||
|
||||
@@ -1933,6 +2067,9 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
|
||||
server_url = server_url[:-3]
|
||||
lmstudio_url = _localhost_to_ipv4(_lmstudio_server_root(base_url))
|
||||
|
||||
if _endpoint_blackholed(server_url):
|
||||
return None
|
||||
|
||||
headers = _auth_headers(api_key)
|
||||
|
||||
try:
|
||||
@@ -2003,13 +2140,39 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
|
||||
if resp.status_code == 200:
|
||||
data = resp.json()
|
||||
models_list = data.get("data", [])
|
||||
# Match by id; on single-model servers (e.g. llama.cpp) the
|
||||
# configured name rarely equals the reported id (a GGUF path),
|
||||
# so fall back to the sole model when nothing matches.
|
||||
matched = None
|
||||
for m in models_list:
|
||||
if _model_id_matches(m.get("id", ""), model):
|
||||
ctx = m.get("max_model_len") or m.get("context_length") or m.get("max_tokens")
|
||||
if ctx and isinstance(ctx, (int, float)):
|
||||
return int(ctx)
|
||||
except Exception:
|
||||
pass
|
||||
matched = m
|
||||
break
|
||||
if matched is None and len(models_list) == 1:
|
||||
matched = models_list[0]
|
||||
if matched is not None:
|
||||
# llama.cpp nests the runtime context under meta.n_ctx; the
|
||||
# vLLM/OpenAI keys are also checked. Runtime n_ctx is
|
||||
# preferred over n_ctx_train (the training maximum, which
|
||||
# can be larger than what the server actually allocates).
|
||||
for source in (matched, matched.get("meta") or {}):
|
||||
if not isinstance(source, dict):
|
||||
continue
|
||||
for key in (
|
||||
"n_ctx",
|
||||
"context_length",
|
||||
"context_window",
|
||||
"max_model_len",
|
||||
"max_context_length",
|
||||
"max_tokens",
|
||||
"n_ctx_train",
|
||||
):
|
||||
val = source.get(key)
|
||||
if isinstance(val, (int, float)) and val:
|
||||
return int(val)
|
||||
except Exception as exc:
|
||||
if _is_connect_timeout(exc):
|
||||
_note_endpoint_blackholed(server_url)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
+54
-20
@@ -482,11 +482,9 @@ class RelayTurnContext:
|
||||
default_factory=threading.RLock,
|
||||
repr=False,
|
||||
)
|
||||
_token: contextvars.Token[RelayTurnContext | None] | None = field(
|
||||
default=None,
|
||||
repr=False,
|
||||
)
|
||||
_previous_turn: RelayTurnContext | None = field(default=None, repr=False)
|
||||
_active_registered: bool = field(default=False, repr=False)
|
||||
relay_enabled: bool = True
|
||||
closed: bool = False
|
||||
|
||||
|
||||
@@ -600,7 +598,28 @@ class RelaySessionCoordinator:
|
||||
if lease.released:
|
||||
raise RuntimeError("Hermes Relay conversation lease is released")
|
||||
turn = RelayTurnContext(lease=lease, turn_id=turn_id, task_id=task_id)
|
||||
if isinstance(lease.host, RelayRuntime) and lease.session is not None:
|
||||
key = (lease.profile_key, lease.session_id)
|
||||
with self._active_turns_lock:
|
||||
active = self._active_turns.get(key)
|
||||
if active:
|
||||
# A Relay session owns one physical scope stack. Concurrent
|
||||
# Hermes turns would create sibling scopes on that stack, but
|
||||
# their completion order is not guaranteed to be LIFO.
|
||||
turn.relay_enabled = False
|
||||
logger.warning(
|
||||
"Skipping Relay instrumentation for concurrent Hermes turn "
|
||||
"%s in session %s",
|
||||
turn_id,
|
||||
lease.session_id,
|
||||
)
|
||||
else:
|
||||
self._active_turns[key] = {id(turn)}
|
||||
turn._active_registered = True
|
||||
if (
|
||||
turn.relay_enabled
|
||||
and isinstance(lease.host, RelayRuntime)
|
||||
and lease.session is not None
|
||||
):
|
||||
try:
|
||||
turn.handle = lease.host.run_in_session(
|
||||
lease.session,
|
||||
@@ -617,11 +636,8 @@ class RelaySessionCoordinator:
|
||||
)
|
||||
except Exception:
|
||||
logger.warning("Hermes Relay turn initialization failed", exc_info=True)
|
||||
turn._token = _CURRENT_TURN.set(turn)
|
||||
key = (lease.profile_key, lease.session_id)
|
||||
with self._active_turns_lock:
|
||||
self._active_turns.setdefault(key, set()).add(id(turn))
|
||||
turn._active_registered = True
|
||||
turn._previous_turn = _CURRENT_TURN.get()
|
||||
_CURRENT_TURN.set(turn)
|
||||
return turn
|
||||
|
||||
def end_turn(
|
||||
@@ -755,16 +771,18 @@ class RelaySessionCoordinator:
|
||||
|
||||
@staticmethod
|
||||
def _reset_turn_context(turn: RelayTurnContext) -> None:
|
||||
"""Reset the originating ContextVar token when called in that context."""
|
||||
if turn._token is None:
|
||||
"""Unwind ``turn`` without disturbing a newer context-local turn."""
|
||||
if _CURRENT_TURN.get() is not turn:
|
||||
return
|
||||
try:
|
||||
_CURRENT_TURN.reset(turn._token)
|
||||
except ValueError:
|
||||
# A copied async/thread context may own terminal cleanup. Keep the
|
||||
# token so the originating context can clear its stale reference.
|
||||
return
|
||||
turn._token = None
|
||||
previous = turn._previous_turn
|
||||
seen = {id(turn)}
|
||||
while previous is not None and previous.closed:
|
||||
if id(previous) in seen:
|
||||
previous = None
|
||||
break
|
||||
seen.add(id(previous))
|
||||
previous = previous._previous_turn
|
||||
_CURRENT_TURN.set(previous)
|
||||
|
||||
@staticmethod
|
||||
def release_conversation(lease: ConversationLease) -> None:
|
||||
@@ -793,10 +811,21 @@ def current_turn() -> RelayTurnContext | None:
|
||||
return _CURRENT_TURN.get()
|
||||
|
||||
|
||||
def relay_instrumentation_enabled() -> bool:
|
||||
"""Return whether this inherited turn may create Relay instrumentation."""
|
||||
turn = current_turn()
|
||||
return turn is None or (turn.relay_enabled and not turn.closed)
|
||||
|
||||
|
||||
def active_turn(session_id: str | None = None) -> RelayTurnContext | None:
|
||||
"""Return a live turn only when it belongs to the active profile/session."""
|
||||
turn = current_turn()
|
||||
if turn is None or turn.closed or turn.lease.released:
|
||||
if (
|
||||
turn is None
|
||||
or not turn.relay_enabled
|
||||
or turn.closed
|
||||
or turn.lease.released
|
||||
):
|
||||
return None
|
||||
if turn.lease.profile_key != current_profile_key():
|
||||
return None
|
||||
@@ -814,6 +843,11 @@ def resolve_execution_context(
|
||||
session_id: str,
|
||||
) -> tuple[RelayRuntime | None, RelaySession | None, Any]:
|
||||
"""Resolve one active turn/session parent for managed Relay execution."""
|
||||
inherited_turn = current_turn()
|
||||
if inherited_turn is not None and (
|
||||
not inherited_turn.relay_enabled or inherited_turn.closed
|
||||
):
|
||||
return None, None, None
|
||||
turn = active_turn(session_id)
|
||||
if (
|
||||
turn is not None
|
||||
|
||||
@@ -13,6 +13,7 @@ the conversation without modifying the system prompt (preserving prompt caching)
|
||||
Inspired by Block/goose's SubdirectoryHintTracker.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
@@ -45,6 +46,18 @@ _COMMAND_TOOLS = {"terminal"}
|
||||
# Prevents scanning all the way to / for deeply nested paths.
|
||||
_MAX_ANCESTOR_WALK = 5
|
||||
|
||||
# Directory names that never contain authoritative project context.
|
||||
# Backups, vendored deps, VCS internals, and caches routinely hold *copies* of
|
||||
# AGENTS.md; loading those duplicates real context and inflates the prompt.
|
||||
_EXCLUDED_DIR_NAMES = frozenset({
|
||||
"node_modules", "venv", ".venv", "__pycache__",
|
||||
".git", ".hg", ".svn",
|
||||
".Trash", ".cache", ".tox", ".mypy_cache", ".pytest_cache",
|
||||
"site-packages", "dist-packages",
|
||||
"backups", "backup", ".backups",
|
||||
"vendor", "third_party",
|
||||
})
|
||||
|
||||
|
||||
def _is_ancestor_or_same(a: Path, b: Path) -> bool:
|
||||
"""Check if *a* is the same as or an ancestor of *b* (parent directory check)."""
|
||||
@@ -54,6 +67,7 @@ def _is_ancestor_or_same(a: Path, b: Path) -> bool:
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
class SubdirectoryHintTracker:
|
||||
"""Track which directories the agent visits and load hints on first access.
|
||||
|
||||
@@ -70,8 +84,34 @@ class SubdirectoryHintTracker:
|
||||
def __init__(self, working_dir: Optional[str] = None):
|
||||
self.working_dir = Path(working_dir or os.getcwd()).resolve()
|
||||
self._loaded_dirs: Set[Path] = set()
|
||||
# Content digests already injected — prevents re-sending the same file
|
||||
# reachable through symlinks, hardlinks, or duplicated copies.
|
||||
self._loaded_digests: Set[str] = set()
|
||||
# Pre-mark the working dir as loaded (startup context handles it)
|
||||
self._loaded_dirs.add(self.working_dir)
|
||||
self._seed_working_dir_digest()
|
||||
|
||||
def _seed_working_dir_digest(self) -> None:
|
||||
"""Record the CWD context file's digest so it is never re-injected.
|
||||
|
||||
``prompt_builder`` already loads the working directory's context file at
|
||||
startup. Seeding its digest here means the same content reached through
|
||||
a different path (a symlink farm, a shared workspace) is recognised as a
|
||||
duplicate instead of being sent a second time.
|
||||
"""
|
||||
for filename in _HINT_FILENAMES:
|
||||
candidate = self.working_dir / filename
|
||||
try:
|
||||
if not candidate.is_file():
|
||||
continue
|
||||
content = candidate.read_text(encoding="utf-8").strip()
|
||||
except (OSError, UnicodeDecodeError):
|
||||
continue
|
||||
if content:
|
||||
self._loaded_digests.add(
|
||||
hashlib.sha256(content.encode("utf-8")).hexdigest()
|
||||
)
|
||||
break # first match wins, mirroring startup loading
|
||||
|
||||
def check_tool_call(
|
||||
self,
|
||||
@@ -193,8 +233,25 @@ class SubdirectoryHintTracker:
|
||||
# check as a best-effort safeguard.
|
||||
if not _is_ancestor_or_same(self.working_dir, path):
|
||||
return False
|
||||
if self._is_excluded(path):
|
||||
return False
|
||||
return True
|
||||
|
||||
def _is_excluded(self, path: Path) -> bool:
|
||||
"""True when the path sits inside a directory that holds copies, not context.
|
||||
|
||||
Directories the user is deliberately working inside are never excluded —
|
||||
if ``working_dir`` is itself under ``vendor/``, that segment is legitimate
|
||||
and only segments *below* the working dir are screened.
|
||||
"""
|
||||
try:
|
||||
rel_parts = path.relative_to(self.working_dir).parts
|
||||
except ValueError:
|
||||
# Paths outside the working dir are already rejected by
|
||||
# _is_valid_subdir before this runs; treat as excluded defensively.
|
||||
return True
|
||||
return any(part in _EXCLUDED_DIR_NAMES for part in rel_parts)
|
||||
|
||||
def _load_hints_for_directory(self, directory: Path) -> Optional[str]:
|
||||
"""Load hint files from a directory. Returns formatted text or None.
|
||||
|
||||
@@ -230,6 +287,19 @@ class SubdirectoryHintTracker:
|
||||
content = hint_path.read_text(encoding="utf-8").strip()
|
||||
if not content:
|
||||
continue
|
||||
# Skip content we've already injected. The same AGENTS.md is
|
||||
# routinely reachable through several paths (symlinked shared
|
||||
# workspaces, hardlinks, copied backups); re-sending it burns
|
||||
# context for zero new information.
|
||||
digest = hashlib.sha256(content.encode("utf-8")).hexdigest()
|
||||
if digest in self._loaded_digests:
|
||||
logger.debug(
|
||||
"Skipping duplicate hint content at %s (digest %s)",
|
||||
hint_path,
|
||||
digest[:12],
|
||||
)
|
||||
break
|
||||
self._loaded_digests.add(digest)
|
||||
# Same security scan as startup context loading
|
||||
content = _scan_context_content(content, filename)
|
||||
if len(content) > _MAX_HINT_CHARS:
|
||||
|
||||
+28
-14
@@ -11,14 +11,14 @@ Three tiers are joined with ``\\n\\n``:
|
||||
|
||||
* ``stable`` — identity (SOUL.md or DEFAULT_AGENT_IDENTITY), tool
|
||||
guidance, computer-use guidance, nous subscription block, tool-use
|
||||
enforcement guidance + per-model operational guidance, skills prompt,
|
||||
enforcement guidance + per-model operational guidance,
|
||||
alibaba model-name workaround, environment hints, coding guidance,
|
||||
platform hints.
|
||||
* ``context`` — caller-supplied ``system_message`` plus context files
|
||||
(AGENTS.md / .cursorrules / etc.) discovered under ``TERMINAL_CWD``,
|
||||
plus the session's coding-workspace snapshot.
|
||||
* ``volatile`` — memory snapshot, USER.md profile, external memory
|
||||
provider block, timestamp/session/model/provider line.
|
||||
* ``volatile`` — skills index, memory snapshot, USER.md profile, external
|
||||
memory provider block, timestamp/session/model/provider line.
|
||||
|
||||
Pure helpers that read the agent's state. AIAgent keeps thin forwarders.
|
||||
"""
|
||||
@@ -158,8 +158,8 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
|
||||
* ``context`` — the workspace snapshot followed by the remaining
|
||||
session-stable guidance, context files, and caller-supplied
|
||||
system_message.
|
||||
* ``volatile`` — memory snapshot, user profile, external
|
||||
memory provider block, timestamp line.
|
||||
* ``volatile`` — skills index, memory snapshot, user profile,
|
||||
external memory provider block, timestamp line.
|
||||
|
||||
Joined into a single string by :func:`build_system_prompt` and
|
||||
cached on ``agent._cached_system_prompt`` for the lifetime of the
|
||||
@@ -325,8 +325,6 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
|
||||
)
|
||||
else:
|
||||
skills_prompt = ""
|
||||
if skills_prompt:
|
||||
stable_parts.append(skills_prompt)
|
||||
|
||||
# Alibaba Coding Plan API always returns "glm-4.7" as model name regardless
|
||||
# of the requested model. Inject explicit model identity into the system prompt
|
||||
@@ -497,8 +495,22 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
|
||||
if context_files_prompt:
|
||||
context_parts.append(context_files_prompt)
|
||||
|
||||
# ── Volatile tier (changes per session/turn — never cached) ───
|
||||
# ── Volatile tier (most likely to differ on a rebuild; kept last so the stable prefix stays reusable) ──
|
||||
volatile_parts: List[str] = []
|
||||
# Skills are runtime-mutable: the agent adds and patches them across a
|
||||
# session (SKILLS_GUIDANCE tells it to patch a skill the moment it goes
|
||||
# stale). The built prompt is cached per session and only rebuilt on
|
||||
# compaction/restore (see build_system_prompt), so a skill change is not
|
||||
# byte-stable across rebuilds. With the index in the stable band, a rebuild
|
||||
# that picked up a skill change would bust the cached prefix from the index
|
||||
# down, taking the whole scaffold with it. Render it at the FRONT of the
|
||||
# volatile band instead, ahead of the turn-varying memory/timestamp tail:
|
||||
# on an implicit longest-prefix backend an unchanged index still falls
|
||||
# inside the reused prefix, and a changed one only re-prefills from here on.
|
||||
# (No effect for single-block cache_control backends, where the whole
|
||||
# system message is one cache unit regardless of internal order.)
|
||||
if skills_prompt:
|
||||
volatile_parts.append(skills_prompt)
|
||||
|
||||
if agent._memory_store:
|
||||
if agent._memory_enabled:
|
||||
@@ -556,10 +568,12 @@ def build_system_prompt(agent: Any, system_message: Optional[str] = None) -> str
|
||||
|
||||
Layers are ordered cache-friendly: stable identity/guidance first,
|
||||
then session-stable context files, then per-call volatile content
|
||||
(memory, USER profile, timestamp). The whole string is treated as
|
||||
one cached block — Hermes never rebuilds or reinjects parts of it
|
||||
mid-session, which is the only way to keep upstream prompt caches
|
||||
warm across turns.
|
||||
(skills index, memory, USER profile, timestamp). For explicit
|
||||
cache_control backends the whole string is one cached block. For
|
||||
implicit longest-prefix backends the order is what matters: the
|
||||
content most likely to change is rendered last, so when the prompt is
|
||||
rebuilt (on compaction/restore) the unchanged stable scaffold ahead of
|
||||
the change stays in the reused prefix.
|
||||
"""
|
||||
parts = build_system_prompt_parts(agent, system_message=system_message)
|
||||
joined = "\n\n".join(p for p in (parts["stable"], parts["context"], parts["volatile"]) if p)
|
||||
@@ -602,8 +616,8 @@ def reconstruct_static_prefix(
|
||||
Safety: the rebuilt stable tier is used ONLY when the stored prompt
|
||||
literally starts with it (checked here AND re-checked by
|
||||
``_apply_system_cache_markers``'s ``startswith`` gate). If any
|
||||
stable-tier input changed since the prompt was persisted (skills
|
||||
edited, identity changed), the prefix mismatches, the static stays
|
||||
stable-tier input changed since the prompt was persisted (identity
|
||||
changed, SOUL.md edited), the prefix mismatches, the static stays
|
||||
None, and requests fall back to the legacy layout with the stored
|
||||
prompt bytes untouched — never a rewritten prompt.
|
||||
|
||||
|
||||
@@ -48,6 +48,7 @@ _PARALLEL_SAFE_TOOLS = frozenset({
|
||||
"ha_get_state",
|
||||
"ha_list_entities",
|
||||
"ha_list_services",
|
||||
"image_generate",
|
||||
"read_file",
|
||||
"search_files",
|
||||
"session_search",
|
||||
|
||||
+42
-1
@@ -93,6 +93,7 @@ def _budget_for_agent(agent) -> BudgetConfig:
|
||||
# Maximum number of concurrent worker threads for parallel tool execution.
|
||||
# Mirrors the constant in ``run_agent`` for tests/imports that look here.
|
||||
_MAX_TOOL_WORKERS = 8
|
||||
_DEFAULT_IMAGE_PARALLEL_REQUESTS = 4
|
||||
# Keep this above the stock auxiliary.web_extract timeout (360s) so the batch
|
||||
# guard does not preempt a slow-but-valid summarization attempt.
|
||||
_DEFAULT_CONCURRENT_TOOL_TIMEOUT_S = 420.0
|
||||
@@ -159,6 +160,46 @@ def _flush_session_db_after_tool_progress(
|
||||
return False
|
||||
|
||||
|
||||
def _image_generate_parallel_limit() -> int:
|
||||
"""Return the configured image-generation parallelism cap.
|
||||
|
||||
Image-generation calls are slow enough that concurrent execution is useful,
|
||||
but backend bursts can hit TTFB or rate-limit failures. Keep the default
|
||||
intentionally conservative while allowing users to tune it per install.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
cfg = load_config() or {}
|
||||
image_gen = cfg.get("image_gen") if isinstance(cfg, dict) else None
|
||||
value = (
|
||||
image_gen.get("max_parallel_requests")
|
||||
if isinstance(image_gen, dict)
|
||||
else None
|
||||
)
|
||||
except Exception:
|
||||
value = None
|
||||
|
||||
try:
|
||||
limit = int(value)
|
||||
except (TypeError, ValueError):
|
||||
limit = _DEFAULT_IMAGE_PARALLEL_REQUESTS
|
||||
return max(1, min(limit, _MAX_TOOL_WORKERS))
|
||||
|
||||
|
||||
def _max_workers_for_tool_batch(runnable_calls) -> int:
|
||||
"""Return the worker cap for a concurrent tool batch."""
|
||||
if not runnable_calls:
|
||||
return 0
|
||||
max_workers = _MAX_TOOL_WORKERS
|
||||
if any(
|
||||
(call[2] if len(call) >= 3 else None) == "image_generate"
|
||||
for call in runnable_calls
|
||||
):
|
||||
max_workers = min(max_workers, _image_generate_parallel_limit())
|
||||
return min(len(runnable_calls), max_workers)
|
||||
|
||||
|
||||
def _ra():
|
||||
"""Lazy reference to ``run_agent`` so patches like ``run_agent._set_interrupt`` work."""
|
||||
import run_agent
|
||||
@@ -953,7 +994,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
|
||||
timeout_s = _resolve_concurrent_tool_timeout()
|
||||
deadline = time.monotonic() + timeout_s if timeout_s is not None else None
|
||||
if runnable_calls:
|
||||
max_workers = min(len(runnable_calls), _MAX_TOOL_WORKERS)
|
||||
max_workers = _max_workers_for_tool_batch(runnable_calls)
|
||||
# Daemon workers: an interrupted/timed-out batch is abandoned with
|
||||
# shutdown(wait=False), but stdlib ThreadPoolExecutor workers are
|
||||
# non-daemon and registered in concurrent.futures' atexit hook,
|
||||
|
||||
@@ -9,6 +9,7 @@ which has provider-specific conditionals for max_tokens defaults,
|
||||
reasoning configuration, temperature handling, and extra_body assembly.
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Any, Dict
|
||||
|
||||
from agent.lmstudio_reasoning import resolve_lmstudio_effort
|
||||
@@ -18,6 +19,56 @@ from agent.transports.base import ProviderTransport
|
||||
from agent.transports.types import NormalizedResponse, ToolCall, Usage
|
||||
|
||||
|
||||
def _static_prompt_instructions(messages: list[dict[str, Any]]) -> str:
|
||||
"""Return the stable system/developer prefix used for cache routing.
|
||||
|
||||
Chat Completions carries instructions in its message list rather than a
|
||||
separate ``instructions`` field. Only a leading system/developer message
|
||||
is static by contract; later messages are conversation state and must not
|
||||
split a warm prefix bucket on every turn.
|
||||
"""
|
||||
if not messages or not isinstance(messages[0], dict):
|
||||
return ""
|
||||
first = messages[0]
|
||||
if first.get("role") not in {"system", "developer"}:
|
||||
return ""
|
||||
content = first.get("content")
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
try:
|
||||
return json.dumps(content, sort_keys=True, ensure_ascii=False, separators=(",", ":"))
|
||||
except (TypeError, ValueError):
|
||||
return str(content or "")
|
||||
|
||||
|
||||
def _add_prompt_cache_key(
|
||||
api_kwargs: dict[str, Any],
|
||||
*,
|
||||
messages: list[dict[str, Any]],
|
||||
tools: list[dict[str, Any]] | None,
|
||||
supports_prompt_cache_key: bool,
|
||||
) -> None:
|
||||
"""Add a content-addressed key only for an explicitly capable endpoint."""
|
||||
if not supports_prompt_cache_key:
|
||||
return
|
||||
|
||||
# An explicit caller body field is authoritative too. Do not add a
|
||||
# duplicate top-level field whose SDK merge precedence could overwrite it.
|
||||
extra_body = api_kwargs.get("extra_body")
|
||||
if "prompt_cache_key" in api_kwargs or (
|
||||
isinstance(extra_body, dict) and "prompt_cache_key" in extra_body
|
||||
):
|
||||
return
|
||||
|
||||
# Reuse the Responses transport's single authoritative hash algorithm so
|
||||
# equivalent static prefixes route to the same cache bucket across modes.
|
||||
from agent.transports.codex import _content_cache_key
|
||||
|
||||
cache_key = _content_cache_key(_static_prompt_instructions(messages), tools)
|
||||
if cache_key:
|
||||
api_kwargs["prompt_cache_key"] = cache_key
|
||||
|
||||
|
||||
def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None:
|
||||
"""Return the model's wire-compatible reasoning config."""
|
||||
if not isinstance(reasoning_config, dict):
|
||||
@@ -112,6 +163,24 @@ def _is_gemini_openai_compat_base_url(base_url: Any) -> bool:
|
||||
return normalized.endswith("/openai")
|
||||
|
||||
|
||||
def _is_openai_api_base_url(base_url: Any) -> bool:
|
||||
"""True only for api.openai.com itself (exact host).
|
||||
|
||||
OpenAI documents ``prompt_cache_key`` as a first-class body field and
|
||||
GPT-5.6+ docs recommend it for reliable cache routing, so the flag is
|
||||
implied for the real endpoint. Deliberately NOT a substring match:
|
||||
Azure OpenAI and strict OpenAI-compat endpoints may reject unknown
|
||||
fields and must stay opt-in via ``supports_prompt_cache_key``.
|
||||
"""
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
|
||||
host = (urlparse(str(base_url or "").strip()).hostname or "").lower()
|
||||
except Exception:
|
||||
return False
|
||||
return host == "api.openai.com"
|
||||
|
||||
|
||||
def _model_consumes_thought_signature(model: Any) -> bool:
|
||||
"""True when the outgoing model is a Gemini family model that requires
|
||||
``extra_content`` (thought_signature) to be replayed on tool calls.
|
||||
@@ -327,6 +396,8 @@ class ChatCompletionsTransport(ProviderTransport):
|
||||
# Claude on OpenRouter/Nous max output
|
||||
anthropic_max_output: int | None
|
||||
extra_body_additions: dict | None
|
||||
supports_prompt_cache_key: bool — explicit endpoint capability for
|
||||
the top-level Chat Completions request field; defaults off.
|
||||
"""
|
||||
# Codex sanitization: drop reasoning_items / call_id / response_item_id.
|
||||
# Pass model so the Gemini thought_signature (extra_content) is kept for
|
||||
@@ -507,6 +578,14 @@ class ChatCompletionsTransport(ProviderTransport):
|
||||
if overrides:
|
||||
api_kwargs.update(overrides)
|
||||
|
||||
_add_prompt_cache_key(
|
||||
api_kwargs,
|
||||
messages=sanitized,
|
||||
tools=api_kwargs.get("tools"),
|
||||
supports_prompt_cache_key=bool(params.get("supports_prompt_cache_key"))
|
||||
or _is_openai_api_base_url(params.get("base_url")),
|
||||
)
|
||||
|
||||
return api_kwargs
|
||||
|
||||
def _build_kwargs_from_profile(self, profile, model, sanitized, tools, params):
|
||||
@@ -649,6 +728,13 @@ class ChatCompletionsTransport(ProviderTransport):
|
||||
if extra_body:
|
||||
api_kwargs["extra_body"] = extra_body
|
||||
|
||||
_add_prompt_cache_key(
|
||||
api_kwargs,
|
||||
messages=sanitized,
|
||||
tools=api_kwargs.get("tools"),
|
||||
supports_prompt_cache_key=bool(profile.supports_prompt_cache_key),
|
||||
)
|
||||
|
||||
return api_kwargs
|
||||
|
||||
def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse:
|
||||
|
||||
+79
-50
@@ -28,6 +28,52 @@ def _bounded_prompt_cache_key(value: Any) -> Optional[str]:
|
||||
return f"pck_{digest}"
|
||||
|
||||
|
||||
# Wire-name used when Hermes keeps client-side web_search on xAI Responses.
|
||||
# A function literally named ``web_search`` collides with Grok's native
|
||||
# server-side tool (incomplete hang or HTTP 400 duplicate names); this alias
|
||||
# avoids that while still dispatching through Hermes's configured provider
|
||||
# (Firecrawl / Tavily / …). Mapped back to ``web_search`` in normalize_response.
|
||||
_XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search"
|
||||
|
||||
|
||||
def _xai_prefers_native_web_search() -> bool:
|
||||
"""True when xAI Responses should use Grok's native ``web_search`` built-in.
|
||||
|
||||
Delegates to the web-search registry's provider resolution (which reads
|
||||
``web.search_backend`` / ``web.backend`` from config) and checks whether
|
||||
the resolved provider is xAI. Falls back to the legacy ``_get_search_backend``
|
||||
probe when the registry has no providers loaded. On any resolution failure,
|
||||
returns True (fail-closed to native — preserves the #48108 incomplete-hang
|
||||
fix rather than risk reintroducing it).
|
||||
"""
|
||||
try:
|
||||
from agent.web_search_registry import get_active_search_provider
|
||||
|
||||
provider = get_active_search_provider()
|
||||
if provider is not None:
|
||||
return getattr(provider, "name", None) == "xai"
|
||||
|
||||
from tools.web_tools import _get_search_backend
|
||||
|
||||
return (_get_search_backend() or "").strip().lower() == "xai"
|
||||
except Exception:
|
||||
# Fail closed to native — same behavior as pre-fix main.
|
||||
return True
|
||||
|
||||
|
||||
def _rename_client_web_search_for_xai(response_tools: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""Rename client ``web_search`` → alias so xAI won't hijack it server-side."""
|
||||
rewritten: List[Dict[str, Any]] = []
|
||||
for tool in response_tools:
|
||||
if isinstance(tool, dict) and tool.get("name") == "web_search":
|
||||
aliased = dict(tool)
|
||||
aliased["name"] = _XAI_CLIENT_WEB_SEARCH_ALIAS
|
||||
rewritten.append(aliased)
|
||||
else:
|
||||
rewritten.append(tool)
|
||||
return rewritten
|
||||
|
||||
|
||||
_EXTENDED_PROMPT_CACHE_MODELS = (
|
||||
"gpt-5.5-pro",
|
||||
"gpt-5.5",
|
||||
@@ -232,63 +278,41 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
|
||||
response_tools = _responses_tools(tools)
|
||||
|
||||
# xAI server-side web search.
|
||||
# xAI server-side web search vs Hermes web providers.
|
||||
#
|
||||
# grok models on xAI's /v1/responses surface (notably
|
||||
# grok-composer-2.5-fast on SuperGrok OAuth) have a *native*,
|
||||
# server-executed web search. When the model is handed a
|
||||
# client-side function literally named ``web_search``, it routes
|
||||
# the intent to that native engine — but because the tool is
|
||||
# declared as a plain ``function`` rather than xAI's first-class
|
||||
# ``{"type": "web_search"}`` built-in, the server-side search is
|
||||
# dispatched but never reconciled: the response streams reasoning
|
||||
# + ``web_search_call`` progress items, the searches never reach
|
||||
# ``status="completed"`` in the assembled output, no final
|
||||
# message is emitted, and ``_normalize_codex_response`` correctly
|
||||
# sees reasoning-with-no-answer and reports ``incomplete``. The
|
||||
# turn then burns 3 continuation retries and fails with "Codex
|
||||
# response remained incomplete after 3 continuation attempts".
|
||||
# Verified live against grok-composer-2.5-fast (2026-06).
|
||||
# grok models on xAI's /v1/responses surface have a *native*,
|
||||
# server-executed web search. A client-side function literally named
|
||||
# ``web_search`` collides with that engine: declared as a plain
|
||||
# ``function`` rather than ``{"type": "web_search"}``, the search
|
||||
# dispatches but never reconciles → incomplete turn + 3 retries.
|
||||
# Verified live against grok-composer-2.5-fast (2026-06); see #48108.
|
||||
#
|
||||
# Fix: when the agent HAS a client-side ``web_search`` function (i.e.
|
||||
# the user enabled the web toolset), declare xAI's native
|
||||
# ``web_search`` built-in instead so the search actually runs to
|
||||
# completion server-side and the model streams a real answer. The
|
||||
# Responses API rejects two tools sharing the name ``web_search``
|
||||
# (HTTP 400 "Duplicate tool names"), so we drop the client-side
|
||||
# ``web_search`` function for the xAI path and let the native tool
|
||||
# satisfy it. All other client-side tools (read_file, terminal,
|
||||
# web_extract, MCP tools, …) are untouched and continue to dispatch
|
||||
# through Hermes's agent loop.
|
||||
# Two modes, chosen by the user's web-search backend config:
|
||||
#
|
||||
# Scope: we ONLY swap in the native built-in when the client
|
||||
# ``web_search`` was actually present. We do NOT force-enable Grok
|
||||
# server-side search on turns where the user never had web enabled —
|
||||
# that would silently route around Hermes's web-provider config and
|
||||
# tool-trace/citation plumbing for every xai-oauth turn. The swap is
|
||||
# a 1:1 replacement of an already-requested capability, not an
|
||||
# additive grant.
|
||||
#
|
||||
# NOTE: for the swapped case this routes ``web_search`` to Grok's
|
||||
# native search engine for xAI sessions instead of Hermes's
|
||||
# configured web provider (Tavily/etc.), and those results bypass
|
||||
# Hermes's tool-trace / citation plumbing (they arrive baked into the
|
||||
# model's answer rather than as a tool result the loop observes).
|
||||
# Scoped to ``is_xai_responses`` deliberately; narrow to specific
|
||||
# models if a future grok variant should keep the client-side
|
||||
# function.
|
||||
# 1. **Native** (active/configured backend is ``xai``, or resolution
|
||||
# fails): drop the client ``web_search`` function and declare
|
||||
# xAI's built-in instead. 1:1 swap only when client ``web_search``
|
||||
# was already present — never an additive grant.
|
||||
# 2. **Client** (Firecrawl / Tavily / Exa / … configured or resolved):
|
||||
# keep Hermes dispatch so ``web.backend`` / ``web.search_backend``
|
||||
# is honored, but rename the wire tool to
|
||||
# ``hermes_web_search`` so Grok cannot hijack the name. The alias
|
||||
# is mapped back to ``web_search`` in ``normalize_response``.
|
||||
if is_xai_responses and response_tools:
|
||||
has_client_web_search = any(
|
||||
isinstance(t, dict) and t.get("name") == "web_search"
|
||||
for t in response_tools
|
||||
)
|
||||
if has_client_web_search:
|
||||
filtered = [
|
||||
t for t in response_tools
|
||||
if not (isinstance(t, dict) and t.get("name") == "web_search")
|
||||
]
|
||||
filtered.append({"type": "web_search"})
|
||||
response_tools = filtered
|
||||
if _xai_prefers_native_web_search():
|
||||
filtered = [
|
||||
t for t in response_tools
|
||||
if not (isinstance(t, dict) and t.get("name") == "web_search")
|
||||
]
|
||||
filtered.append({"type": "web_search"})
|
||||
response_tools = filtered
|
||||
else:
|
||||
response_tools = _rename_client_web_search_for_xai(response_tools)
|
||||
|
||||
# ``tools`` MUST be omitted entirely when there are no functions to
|
||||
# expose: the openai SDK's ``responses.stream()`` / ``responses.parse()``
|
||||
@@ -487,9 +511,14 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
provider_data["call_id"] = tc.call_id
|
||||
if hasattr(tc, "response_item_id") and tc.response_item_id:
|
||||
provider_data["response_item_id"] = tc.response_item_id
|
||||
name = tc.function.name if hasattr(tc, "function") else getattr(tc, "name", "")
|
||||
# Undo the xAI client-path wire alias so Hermes dispatches
|
||||
# the real ``web_search`` tool (Firecrawl / etc.).
|
||||
if name == _XAI_CLIENT_WEB_SEARCH_ALIAS:
|
||||
name = "web_search"
|
||||
tool_calls.append(ToolCall(
|
||||
id=tc.id if hasattr(tc, "id") else (tc.function.name if hasattr(tc, "function") else None),
|
||||
name=tc.function.name if hasattr(tc, "function") else getattr(tc, "name", ""),
|
||||
id=tc.id if hasattr(tc, "id") else (name or None),
|
||||
name=name,
|
||||
arguments=tc.function.arguments if hasattr(tc, "function") else getattr(tc, "arguments", "{}"),
|
||||
provider_data=provider_data or None,
|
||||
))
|
||||
|
||||
@@ -1259,6 +1259,8 @@ def _approval_choice_to_codex_decision(choice: str) -> str:
|
||||
return "accept"
|
||||
if choice in {"session", "always"}:
|
||||
return "acceptForSession"
|
||||
# "deny" and "timeout" both map to decline — codex has no wire value for
|
||||
# "prompt expired"; the Hermes-side messaging already distinguishes them.
|
||||
return "decline"
|
||||
|
||||
|
||||
|
||||
@@ -41,6 +41,7 @@ from agent.conversation_compression import (
|
||||
from agent.context_engine import automatic_compaction_status_message
|
||||
from agent.iteration_budget import IterationBudget
|
||||
from agent.memory_manager import build_memory_context_block
|
||||
from agent.memory_provider import is_trivial_prompt
|
||||
from agent.model_metadata import (
|
||||
estimate_messages_tokens_rough,
|
||||
estimate_request_tokens_rough,
|
||||
@@ -1152,11 +1153,15 @@ def build_turn_context(
|
||||
pass
|
||||
|
||||
# External memory provider: prefetch once before the tool loop.
|
||||
#
|
||||
# Skip prefetch on trivial prompts (greetings, acknowledgements) to
|
||||
# prevent memory-context injection on turns that carry no semantic signal.
|
||||
ext_prefetch_cache = ""
|
||||
if agent._memory_manager:
|
||||
try:
|
||||
_query = original_user_message if isinstance(original_user_message, str) else ""
|
||||
ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or ""
|
||||
if not is_trivial_prompt(_query):
|
||||
ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or ""
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
import { test, expect } from './test'
|
||||
|
||||
import {
|
||||
type MockBackendFixture,
|
||||
setupMockBackend,
|
||||
waitForAppReady,
|
||||
} from './fixtures'
|
||||
|
||||
let fixture: MockBackendFixture | null = null
|
||||
|
||||
test.beforeAll(async () => {
|
||||
fixture = await setupMockBackend()
|
||||
await waitForAppReady(fixture, 120_000)
|
||||
})
|
||||
|
||||
test.afterAll(async () => {
|
||||
await fixture?.cleanup()
|
||||
fixture = null
|
||||
})
|
||||
|
||||
test('persistent terminal overlay follows the pane after split dragging', async () => {
|
||||
const page = fixture!.page
|
||||
|
||||
await page.keyboard.press('Control+`')
|
||||
await page.locator('[data-terminal-slot]').waitFor({ state: 'visible', timeout: 30_000 })
|
||||
await page.locator('[data-persistent-terminal] .xterm').waitFor({ state: 'visible', timeout: 30_000 })
|
||||
|
||||
const result = await page.evaluate(async () => {
|
||||
const slot = document.querySelector('[data-terminal-slot]')
|
||||
const overlay = document.querySelector('[data-persistent-terminal]')
|
||||
|
||||
if (!slot || !overlay) {
|
||||
return { drift: -1, moved: 0, target: false }
|
||||
}
|
||||
|
||||
const before = slot.getBoundingClientRect()
|
||||
const target = [...document.querySelectorAll<HTMLElement>('[role="separator"]')]
|
||||
.map(element => {
|
||||
const box = element.getBoundingClientRect()
|
||||
const horizontal = box.width > box.height
|
||||
const center = horizontal
|
||||
? (box.top + box.bottom) / 2
|
||||
: (box.left + box.right) / 2
|
||||
const sides = horizontal
|
||||
? [before.top, before.bottom]
|
||||
: [before.left, before.right]
|
||||
|
||||
return {
|
||||
element,
|
||||
box,
|
||||
horizontal,
|
||||
score: Math.min(...sides.map(side => Math.abs(center - side))),
|
||||
}
|
||||
})
|
||||
.filter(item => item.box.width > 0 && item.box.height > 0)
|
||||
.sort((a, b) => a.score - b.score)[0]
|
||||
|
||||
if (!target) {
|
||||
return { drift: -1, moved: 0, target: false }
|
||||
}
|
||||
|
||||
const x = target.box.left + target.box.width / 2
|
||||
const y0 = target.box.top + target.box.height / 2
|
||||
const nearestSide = target.horizontal
|
||||
? Math.abs(y0 - before.top) < Math.abs(y0 - before.bottom)
|
||||
? 'top'
|
||||
: 'bottom'
|
||||
: Math.abs(x - before.left) < Math.abs(x - before.right)
|
||||
? 'left'
|
||||
: 'right'
|
||||
const deltaX = nearestSide === 'left' ? -1 : nearestSide === 'right' ? 1 : 0
|
||||
const deltaY = nearestSide === 'top' ? -1 : nearestSide === 'bottom' ? 1 : 0
|
||||
let currentX = x
|
||||
let y = y0
|
||||
const pointer = {
|
||||
bubbles: true,
|
||||
cancelable: true,
|
||||
pointerId: 71,
|
||||
pointerType: 'mouse',
|
||||
isPrimary: true,
|
||||
button: 0,
|
||||
buttons: 1,
|
||||
}
|
||||
|
||||
target.element.dispatchEvent(
|
||||
new PointerEvent('pointerdown', { ...pointer, clientX: x, clientY: y }),
|
||||
)
|
||||
|
||||
for (let index = 0; index < 24; index += 1) {
|
||||
currentX += deltaX
|
||||
y += deltaY
|
||||
window.dispatchEvent(
|
||||
new PointerEvent('pointermove', {
|
||||
...pointer,
|
||||
clientX: currentX,
|
||||
clientY: y,
|
||||
}),
|
||||
)
|
||||
await new Promise<void>(resolve => requestAnimationFrame(() => resolve()))
|
||||
}
|
||||
|
||||
window.dispatchEvent(
|
||||
new PointerEvent('pointerup', {
|
||||
...pointer,
|
||||
buttons: 0,
|
||||
clientX: currentX,
|
||||
clientY: y,
|
||||
}),
|
||||
)
|
||||
await new Promise<void>(resolve => setTimeout(resolve, 350))
|
||||
await new Promise<void>(resolve =>
|
||||
requestAnimationFrame(() => requestAnimationFrame(() => resolve())),
|
||||
)
|
||||
|
||||
const next = slot.getBoundingClientRect()
|
||||
const fixed = overlay.getBoundingClientRect()
|
||||
|
||||
return {
|
||||
drift: Math.max(
|
||||
Math.abs(next.top - fixed.top),
|
||||
Math.abs(next.left - fixed.left),
|
||||
Math.abs(next.width - fixed.width),
|
||||
Math.abs(next.height - fixed.height),
|
||||
),
|
||||
moved: Math.max(
|
||||
Math.abs(next.top - before.top),
|
||||
Math.abs(next.left - before.left),
|
||||
Math.abs(next.width - before.width),
|
||||
Math.abs(next.height - before.height),
|
||||
),
|
||||
target: true,
|
||||
}
|
||||
})
|
||||
|
||||
expect(result.target).toBe(true)
|
||||
expect(result.moved).toBeGreaterThan(10)
|
||||
expect(result.drift).toBeLessThanOrEqual(1)
|
||||
})
|
||||
@@ -5410,6 +5410,12 @@ function buildApplicationMenu() {
|
||||
{ role: 'cut' },
|
||||
{ role: 'copy' },
|
||||
{ role: 'paste' },
|
||||
// ⌘⇧V is only wired up by this item existing: an accelerator with no menu
|
||||
// entry is never translated into an editor command, so the chord was a
|
||||
// no-op in every input in the app. The composer inserts plain text on
|
||||
// every paste anyway, so this is the same result as ⌘V there — it's the
|
||||
// terminal, preview, and other editable surfaces that need the strip.
|
||||
{ role: 'pasteAndMatchStyle' },
|
||||
{ role: 'delete' },
|
||||
{ role: 'selectAll' }
|
||||
]
|
||||
|
||||
@@ -53,6 +53,7 @@ directly via `window.__PERF_DRIVE__`, so no LLM credits are spent.
|
||||
| `transcript` | ci | large-transcript mount + paint cost | (new) |
|
||||
| `render-churn` | ci | per-component render attribution + store churn while N tabs stream | (new) |
|
||||
| `idle-cost` | report | busy-but-silent tiles: idle commit rate, + fps while resizing / typing | (new) |
|
||||
| `right-pane` | report | file tree + persistent xterm tabs under chat/terminal output and split dragging | (new) |
|
||||
| `cold-start` | cold | launch → CDP → driver → first paint (fresh spawn/run) | (new) |
|
||||
| `first-token` | backend | Enter → first assistant token painted (TTFT) | (new) |
|
||||
| `submit` | backend | Enter → cleared → user msg painted, scroll jump | measure-submit, measure-jump |
|
||||
|
||||
@@ -8,6 +8,7 @@ import keystroke from './keystroke.mjs'
|
||||
import multitab from './multitab.mjs'
|
||||
import profileSwitch from './profile-switch.mjs'
|
||||
import renderChurn from './render-churn.mjs'
|
||||
import rightPane from './right-pane.mjs'
|
||||
import sessionLoad from './session-load.mjs'
|
||||
import sessionSwitch from './session-switch.mjs'
|
||||
import stream from './stream.mjs'
|
||||
@@ -22,6 +23,7 @@ export const SCENARIOS = {
|
||||
[transcript.name]: transcript,
|
||||
[multitab.name]: multitab,
|
||||
[renderChurn.name]: renderChurn,
|
||||
[rightPane.name]: rightPane,
|
||||
[idleCost.name]: idleCost,
|
||||
[coldStart.name]: coldStart,
|
||||
[firstToken.name]: firstToken,
|
||||
|
||||
@@ -0,0 +1,315 @@
|
||||
// File-tree + terminal workspace stress. This is the regression scene for the
|
||||
// desktop symptom where opening the project tree/terminal made the whole page
|
||||
// hitch while chat and PTY output continued.
|
||||
//
|
||||
// It mounts a real project tree, one PTY plus multiple persistent xterm tabs,
|
||||
// streams chat and terminal output together, mutates Git decoration state, and
|
||||
// drags the terminal split. The debug probe records the specific work we care
|
||||
// about rather than inferring it from CPU alone:
|
||||
// - fixed-overlay measurements
|
||||
// - active/hidden xterm fits
|
||||
// - ProjectTree + per-path row renders
|
||||
// - frame pacing / slow frames
|
||||
//
|
||||
// npm run perf -- right-pane --spawn --prod --runs 3
|
||||
|
||||
import { dirname, resolve } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
|
||||
import { sleep } from '../lib/cdp.mjs'
|
||||
import { frameHistogram, percentile } from '../lib/stats.mjs'
|
||||
|
||||
const DEFAULT_CWD = resolve(dirname(fileURLToPath(import.meta.url)), '../../..')
|
||||
|
||||
const RECORDERS = `
|
||||
(() => {
|
||||
window.__RP_FRAME_GEN__ = (window.__RP_FRAME_GEN__ || 0) + 1
|
||||
const generation = window.__RP_FRAME_GEN__
|
||||
window.__RP_FRAMES__ = { times: [], stop: false }
|
||||
let last = performance.now()
|
||||
const tick = () => {
|
||||
if (window.__RP_FRAME_GEN__ !== generation || window.__RP_FRAMES__.stop) return
|
||||
const now = performance.now()
|
||||
window.__RP_FRAMES__.times.push(now - last)
|
||||
last = now
|
||||
requestAnimationFrame(tick)
|
||||
}
|
||||
requestAnimationFrame(tick)
|
||||
|
||||
window.__RP_LONG__ = { entries: [], stop: false }
|
||||
try {
|
||||
const observer = new PerformanceObserver(list => {
|
||||
if (window.__RP_LONG__.stop) return
|
||||
for (const entry of list.getEntries()) {
|
||||
window.__RP_LONG__.entries.push({ duration: entry.duration, startTime: entry.startTime })
|
||||
}
|
||||
})
|
||||
observer.observe({ entryTypes: ['longtask'] })
|
||||
window.__RP_LONG__.observer = observer
|
||||
} catch {}
|
||||
return 'armed'
|
||||
})()
|
||||
`
|
||||
|
||||
const COLLECT_RECORDERS = `
|
||||
(() => {
|
||||
window.__RP_FRAMES__.stop = true
|
||||
window.__RP_LONG__.stop = true
|
||||
try { window.__RP_LONG__.observer && window.__RP_LONG__.observer.disconnect() } catch {}
|
||||
return JSON.stringify({ frames: window.__RP_FRAMES__.times, longtasks: window.__RP_LONG__.entries })
|
||||
})()
|
||||
`
|
||||
|
||||
const START_COUNTERS = `window.__RIGHT_PANE_PERF__.start(); 'recording'`
|
||||
const SNAPSHOT_COUNTERS = `
|
||||
(() => {
|
||||
window.__RIGHT_PANE_PERF__.stop()
|
||||
return JSON.stringify(window.__RIGHT_PANE_PERF__.snapshot())
|
||||
})()
|
||||
`
|
||||
|
||||
const DRAG_TERMINAL_SPLIT = `
|
||||
(async () => {
|
||||
const slot = document.querySelector('[data-terminal-slot]')
|
||||
const overlay = document.querySelector('[data-persistent-terminal]')
|
||||
if (!slot || !overlay) return JSON.stringify({ target: 'none', drift: -1, moved: 0 })
|
||||
|
||||
const slotBox = slot.getBoundingClientRect()
|
||||
const candidates = [...document.querySelectorAll('[role="separator"]')]
|
||||
.map(element => ({ element, box: element.getBoundingClientRect() }))
|
||||
.filter(item => item.box.width > item.box.height * 3)
|
||||
.sort((a, b) =>
|
||||
Math.abs((a.box.top + a.box.bottom) / 2 - slotBox.top) -
|
||||
Math.abs((b.box.top + b.box.bottom) / 2 - slotBox.top)
|
||||
)
|
||||
const target = candidates[0]
|
||||
if (!target) return JSON.stringify({ target: 'none', drift: -1, moved: 0 })
|
||||
|
||||
const x = target.box.left + target.box.width / 2
|
||||
const y0 = target.box.top + target.box.height / 2
|
||||
let y = y0
|
||||
const pointer = {
|
||||
bubbles: true, cancelable: true, pointerId: 91, pointerType: 'mouse',
|
||||
isPrimary: true, button: 0, buttons: 1
|
||||
}
|
||||
target.element.dispatchEvent(new PointerEvent('pointerdown', { ...pointer, clientX: x, clientY: y }))
|
||||
|
||||
for (let i = 0; i < 24; i += 1) {
|
||||
y -= 1
|
||||
window.dispatchEvent(new PointerEvent('pointermove', { ...pointer, clientX: x, clientY: y }))
|
||||
await new Promise(resolve => requestAnimationFrame(resolve))
|
||||
}
|
||||
for (let i = 0; i < 24; i += 1) {
|
||||
y += 1
|
||||
window.dispatchEvent(new PointerEvent('pointermove', { ...pointer, clientX: x, clientY: y }))
|
||||
await new Promise(resolve => requestAnimationFrame(resolve))
|
||||
}
|
||||
window.dispatchEvent(new PointerEvent('pointerup', { ...pointer, buttons: 0, clientX: x, clientY: y }))
|
||||
// Track-size transitions continue briefly after pointerup. Wait through
|
||||
// that animation, then give the overlay its normal two-frame calibration.
|
||||
await new Promise(resolve => setTimeout(resolve, 350))
|
||||
await new Promise(resolve => requestAnimationFrame(() => requestAnimationFrame(resolve)))
|
||||
|
||||
const a = slot.getBoundingClientRect()
|
||||
const b = overlay.getBoundingClientRect()
|
||||
const drift = Math.max(
|
||||
Math.abs(a.top - b.top),
|
||||
Math.abs(a.left - b.left),
|
||||
Math.abs(a.width - b.width),
|
||||
Math.abs(a.height - b.height)
|
||||
)
|
||||
return JSON.stringify({ target: 'horizontal-separator', drift, moved: 24 })
|
||||
})()
|
||||
`
|
||||
|
||||
async function waitFor(cdp, expression, label, timeoutMs = 20000) {
|
||||
const deadline = Date.now() + timeoutMs
|
||||
|
||||
while (Date.now() < deadline) {
|
||||
if (await cdp.eval(expression)) {
|
||||
return
|
||||
}
|
||||
|
||||
await sleep(100)
|
||||
}
|
||||
|
||||
throw new Error(`right-pane timed out waiting for ${label}`)
|
||||
}
|
||||
|
||||
const trimWarmup = (frames, warmupMs = 300) => {
|
||||
const kept = []
|
||||
let elapsed = 0
|
||||
|
||||
for (const frame of frames) {
|
||||
elapsed += frame
|
||||
|
||||
if (elapsed >= warmupMs) {
|
||||
kept.push(frame)
|
||||
}
|
||||
}
|
||||
|
||||
return kept
|
||||
}
|
||||
|
||||
export default {
|
||||
name: 'right-pane',
|
||||
tier: 'report',
|
||||
description: 'Project tree + persistent terminal tabs under chat/terminal output and split dragging.',
|
||||
async run(cdp, opts = {}) {
|
||||
const cwd = resolve(String(opts.cwd ?? DEFAULT_CWD))
|
||||
const terminalCount = Math.max(2, Number(opts.terminals ?? 3))
|
||||
const tokens = Number(opts.tokens ?? 90)
|
||||
const outputChunks = Number(opts.outputChunks ?? 160)
|
||||
|
||||
await cdp.send('Runtime.enable')
|
||||
|
||||
const ready = await cdp.eval(
|
||||
`!!(window.__PERF_DRIVE__?.rightPaneSetup && window.__RIGHT_PANE_PERF__ && window.__HERMES_LAYOUT_TREE__)`
|
||||
)
|
||||
|
||||
if (!ready) {
|
||||
throw new Error('right-pane needs a dev renderer or a production build with VITE_PERF_PROBE=1.')
|
||||
}
|
||||
|
||||
let setup
|
||||
|
||||
try {
|
||||
setup = await cdp.eval(
|
||||
`window.__PERF_DRIVE__.rightPaneSetup(${JSON.stringify({ cwd, terminals: terminalCount })})`
|
||||
)
|
||||
await cdp.eval(`window.__HERMES_LAYOUT_TREE__.reveal('files'); window.__HERMES_LAYOUT_TREE__.reveal('terminal')`)
|
||||
await waitFor(cdp, `!!document.querySelector('[data-project-tree]')`, 'project tree')
|
||||
await waitFor(
|
||||
cdp,
|
||||
`!!document.querySelector('[data-terminal-slot]') && !!document.querySelector('[data-persistent-terminal]')`,
|
||||
'persistent terminal'
|
||||
)
|
||||
await waitFor(
|
||||
cdp,
|
||||
`document.querySelectorAll('[data-terminal] .xterm').length >= ${terminalCount}`,
|
||||
`${terminalCount} mounted xterms`,
|
||||
30000
|
||||
)
|
||||
await sleep(1200)
|
||||
|
||||
// Activate every keep-alive tab once, then return to the output tab.
|
||||
// Each activation should restore exactly one fit; inactive tabs must stay
|
||||
// at zero even while another tab resizes or writes output.
|
||||
await cdp.eval(START_COUNTERS)
|
||||
|
||||
for (const id of setup.terminalIds) {
|
||||
await cdp.eval(`window.__PERF_DRIVE__.rightPaneSelect(${JSON.stringify(id)})`)
|
||||
await sleep(180)
|
||||
}
|
||||
|
||||
const activation = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS))
|
||||
|
||||
await cdp.eval(RECORDERS)
|
||||
|
||||
// Chat DOM churn is deliberately measured in its own counter window:
|
||||
// terminal positioning should receive no wakeups from transcript changes.
|
||||
await cdp.eval(START_COUNTERS)
|
||||
await cdp.eval(
|
||||
`window.__PERF_DRIVE__.stream({
|
||||
chunk: 'Right pane streaming sentence with **bold** and \`code\`.\\n\\n',
|
||||
intervalMs: 16,
|
||||
totalTokens: ${tokens},
|
||||
flushMinMs: 33
|
||||
})`
|
||||
)
|
||||
await cdp.eval(`
|
||||
(() => {
|
||||
let n = 0
|
||||
window.__RP_OUTPUT_TIMER__ = setInterval(() => {
|
||||
window.__PERF_DRIVE__.rightPaneWrite(
|
||||
${JSON.stringify(setup.procId)},
|
||||
'terminal output line ' + n + ' ........................................\\r\\n'
|
||||
)
|
||||
n += 1
|
||||
if (n >= ${outputChunks}) clearInterval(window.__RP_OUTPUT_TIMER__)
|
||||
}, 16)
|
||||
return 'writing'
|
||||
})()
|
||||
`)
|
||||
await sleep(Math.max(tokens, outputChunks) * 16 + 900)
|
||||
const stream = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS))
|
||||
|
||||
// An unrelated Git status publication should render neither the tree root
|
||||
// nor any visible row. A status for one visible path should touch only it.
|
||||
await cdp.eval(START_COUNTERS)
|
||||
await cdp.eval(`window.__PERF_DRIVE__.rightPaneGit('__right_pane_unrelated__.txt', 'modified')`)
|
||||
await sleep(250)
|
||||
const unrelatedGit = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS))
|
||||
|
||||
const visiblePath = await cdp.eval(
|
||||
`document.querySelector('[data-project-tree] [title]')?.getAttribute('title') || ''`
|
||||
)
|
||||
let affectedGit = { counts: { 'project-tree-render': 0, 'project-tree-row-render': 0 }, rows: {} }
|
||||
|
||||
if (visiblePath) {
|
||||
const relative = String(visiblePath).startsWith(`${cwd}/`)
|
||||
? String(visiblePath).slice(cwd.length + 1)
|
||||
: String(visiblePath)
|
||||
await cdp.eval(START_COUNTERS)
|
||||
await cdp.eval(`window.__PERF_DRIVE__.rightPaneGit(${JSON.stringify(relative)}, 'modified')`)
|
||||
await sleep(250)
|
||||
affectedGit = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS))
|
||||
}
|
||||
|
||||
await cdp.eval(START_COUNTERS)
|
||||
const drag = JSON.parse(await cdp.eval(DRAG_TERMINAL_SPLIT))
|
||||
const dragCounters = JSON.parse(await cdp.eval(SNAPSHOT_COUNTERS))
|
||||
const recorded = JSON.parse(await cdp.eval(COLLECT_RECORDERS))
|
||||
const frames = trimWarmup(recorded.frames)
|
||||
const longtasks = recorded.longtasks.map(entry => entry.duration)
|
||||
const streamCounts = stream.counts
|
||||
const activationCounts = activation.counts
|
||||
const unrelatedCounts = unrelatedGit.counts
|
||||
const affectedRows = Object.values(affectedGit.rows).reduce((sum, count) => sum + count, 0)
|
||||
const affectedPaths = Object.keys(affectedGit.rows).length
|
||||
|
||||
if (drag.target === 'none') {
|
||||
throw new Error('right-pane found no horizontal terminal split separator.')
|
||||
}
|
||||
|
||||
return {
|
||||
metrics: {
|
||||
chat_terminal_measures: streamCounts['terminal-measure'],
|
||||
hidden_terminal_fits: activationCounts['terminal-fit-hidden'] + streamCounts['terminal-fit-hidden'],
|
||||
activation_fit_mismatch: Math.abs(activationCounts['terminal-fit-active'] - setup.terminalIds.length),
|
||||
unrelated_tree_renders: unrelatedCounts['project-tree-render'],
|
||||
unrelated_row_renders: unrelatedCounts['project-tree-row-render'],
|
||||
affected_tree_renders: affectedGit.counts['project-tree-render'],
|
||||
affected_row_path_excess: Math.max(0, affectedPaths - 1),
|
||||
terminal_drift_px: Math.round(drag.drift * 10) / 10,
|
||||
frame_p95_ms: Math.round(percentile(frames, 0.95) * 10) / 10,
|
||||
frame_p99_ms: Math.round(percentile(frames, 0.99) * 10) / 10,
|
||||
slow_frames_33: frames.filter(frame => frame > 33).length,
|
||||
longtask_max_ms: Math.round((longtasks.length ? Math.max(...longtasks) : 0) * 10) / 10
|
||||
},
|
||||
detail: {
|
||||
cwd,
|
||||
terminals: setup.terminalIds.length,
|
||||
activation,
|
||||
stream,
|
||||
unrelatedGit,
|
||||
affectedGit,
|
||||
affectedRows,
|
||||
drag,
|
||||
dragCounters,
|
||||
frameHistogram: frameHistogram(frames),
|
||||
frames: frames.length
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
await cdp.eval(`
|
||||
(() => {
|
||||
clearInterval(window.__RP_OUTPUT_TIMER__)
|
||||
window.__RIGHT_PANE_PERF__?.stop()
|
||||
window.__PERF_DRIVE__?.reset()
|
||||
return 'cleaned'
|
||||
})()
|
||||
`)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ import { type ReactNode, useEffect, useMemo, useState } from 'react'
|
||||
|
||||
import { useElapsedSeconds } from '@/components/chat/activity-timer'
|
||||
import { ActivityTimerText } from '@/components/chat/activity-timer-text'
|
||||
import { usePaneVisible } from '@/components/pane-shell/pane-visibility'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { FadeText } from '@/components/ui/fade-text'
|
||||
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
|
||||
@@ -189,15 +190,17 @@ function SubagentTree({ tree }: { tree: SubagentNode[] }) {
|
||||
const tokens = flat.reduce((sum, n) => sum + (n.inputTokens ?? 0) + (n.outputTokens ?? 0), 0)
|
||||
const cost = flat.reduce((sum, n) => sum + (n.costUsd ?? 0), 0)
|
||||
|
||||
const visible = usePaneVisible()
|
||||
|
||||
useEffect(() => {
|
||||
if (active <= 0 || typeof window === 'undefined') {
|
||||
if (active <= 0 || !visible || typeof window === 'undefined') {
|
||||
return
|
||||
}
|
||||
|
||||
const id = window.setInterval(() => setNowMs(Date.now()), 500)
|
||||
|
||||
return () => window.clearInterval(id)
|
||||
}, [active])
|
||||
}, [active, visible])
|
||||
|
||||
if (tree.length === 0) {
|
||||
return (
|
||||
|
||||
@@ -446,18 +446,12 @@ export function ChatBar({
|
||||
const handlePaste = (event: ClipboardEvent<HTMLDivElement>) => {
|
||||
const imageBlobs = extractClipboardImageBlobs(event.clipboardData)
|
||||
|
||||
if (imageBlobs.length > 0) {
|
||||
event.preventDefault()
|
||||
if (imageBlobs.length > 0 && onAttachImageBlob) {
|
||||
triggerHaptic('selection')
|
||||
|
||||
if (onAttachImageBlob) {
|
||||
triggerHaptic('selection')
|
||||
|
||||
for (const blob of imageBlobs) {
|
||||
void onAttachImageBlob(blob)
|
||||
}
|
||||
for (const blob of imageBlobs) {
|
||||
void onAttachImageBlob(blob)
|
||||
}
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
// Trim surrounding whitespace so a copy that dragged along leading/trailing
|
||||
@@ -469,6 +463,10 @@ export function ChatBar({
|
||||
if (!pastedText) {
|
||||
event.preventDefault()
|
||||
|
||||
if (imageBlobs.length > 0) {
|
||||
return
|
||||
}
|
||||
|
||||
// Under WSL2/WSLg the Windows host clipboard doesn't bridge *images* to
|
||||
// the Linux clipboard the DOM paste event reads, so a host screenshot
|
||||
// arrives as an empty paste (no blobs, no text). Fall back to the main
|
||||
|
||||
@@ -212,6 +212,47 @@ describe('extractClipboardImageBlobs', () => {
|
||||
|
||||
expect(extractClipboardImageBlobs(clipboard)).toEqual([image])
|
||||
})
|
||||
|
||||
// A rich-text copy (Discord thread, web page, doc) carries prose plus whatever
|
||||
// inline images the page decorated it with. That is a TEXT paste: attaching the
|
||||
// page's placeholder graphics as composer images while the text vanished is the
|
||||
// "blank attachments, no message" bug.
|
||||
it('ignores inline HTML images when the copy carries its own text', () => {
|
||||
const clipboard = {
|
||||
files: { length: 0, item: () => null },
|
||||
getData: (type: string) =>
|
||||
type === 'text/html'
|
||||
? `<p>hello from the thread</p><img src="data:image/png;base64,${'A'.repeat(20_000)}">`
|
||||
: 'hello from the thread',
|
||||
items: []
|
||||
} as unknown as DataTransfer
|
||||
|
||||
expect(extractClipboardImageBlobs(clipboard)).toEqual([])
|
||||
})
|
||||
|
||||
it('keeps inline HTML images when the copy is image-only', () => {
|
||||
const clipboard = {
|
||||
files: { length: 0, item: () => null },
|
||||
getData: (type: string) =>
|
||||
type === 'text/html' ? `<img src="data:image/png;base64,${'A'.repeat(20_000)}">` : '',
|
||||
items: []
|
||||
} as unknown as DataTransfer
|
||||
|
||||
const blobs = extractClipboardImageBlobs(clipboard)
|
||||
|
||||
expect(blobs).toHaveLength(1)
|
||||
expect(blobs[0]?.type).toBe('image/png')
|
||||
})
|
||||
|
||||
it('drops sub-thumbnail inline images — spacers, trackers, blurhash placeholders', () => {
|
||||
const clipboard = {
|
||||
files: { length: 0, item: () => null },
|
||||
getData: (type: string) => (type === 'text/html' ? `<img src="data:image/png;base64,${'A'.repeat(64)}">` : ''),
|
||||
items: []
|
||||
} as unknown as DataTransfer
|
||||
|
||||
expect(extractClipboardImageBlobs(clipboard)).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('blobDedupeKey', () => {
|
||||
|
||||
@@ -70,6 +70,11 @@ const SLASH_INLINE_TRIGGER_RE = /[\s\uFFFC](\/)([a-zA-Z][\w-]*)?$/
|
||||
// `:` or `:D` smiley doesn't open a popover the user didn't ask for.
|
||||
const EMOJI_TRIGGER_RE = /(?:^|[\s\uFFFC])(:)([a-zA-Z0-9_+-]{2,})$/
|
||||
|
||||
const INLINE_IMAGE_SRC_RE = /<img\b[^>]*?\bsrc\s*=\s*["'](data:image\/[^"']+)["']/gi
|
||||
// Below this, an inline data URL is chrome rather than content — a spacer, a
|
||||
// 1×1 tracker, or a blurhash placeholder. Real pasted artwork clears it easily.
|
||||
const MIN_INLINE_IMAGE_BYTES = 4096
|
||||
|
||||
/** Stable key for paste dedupe — `items` and `files` often mirror the same image as different objects. */
|
||||
export function blobDedupeKey(blob: Blob): string {
|
||||
if (blob instanceof File) {
|
||||
@@ -125,16 +130,22 @@ export function extractClipboardImageBlobs(clipboard: DataTransfer): Blob[] {
|
||||
|
||||
if (DATA_IMAGE_URL_RE.test(text)) {
|
||||
push(dataUrlToBlob(text))
|
||||
|
||||
return blobs
|
||||
}
|
||||
|
||||
if (blobs.length === 0) {
|
||||
const html = clipboard.getData('text/html')
|
||||
// Inline `<img src="data:…">` in the clipboard's HTML — but only for a copy
|
||||
// that carried no text of its own. A rich-text copy WITH prose is a text
|
||||
// paste that happens to contain images, and its data URLs are the page's
|
||||
// decorations rather than content: Discord ships a 32×5 blurhash placeholder
|
||||
// beside every image embed, so copying a thread attached a blank thumbnail
|
||||
// and (because an image paste swallows the event) dropped the text entirely.
|
||||
if (!text) {
|
||||
for (const match of clipboard.getData('text/html').matchAll(INLINE_IMAGE_SRC_RE)) {
|
||||
const blob = dataUrlToBlob(match[1])
|
||||
|
||||
if (html) {
|
||||
const matches = html.matchAll(/<img\b[^>]*?\bsrc\s*=\s*["'](data:image\/[^"']+)["']/gi)
|
||||
|
||||
for (const match of matches) {
|
||||
push(dataUrlToBlob(match[1]))
|
||||
if (blob && blob.size >= MIN_INLINE_IMAGE_BYTES) {
|
||||
push(blob)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,164 @@
|
||||
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
|
||||
import { cleanup, fireEvent, render, screen } from '@testing-library/react'
|
||||
import { useState } from 'react'
|
||||
import { MemoryRouter } from 'react-router'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { assistantTextPart, type ChatMessage } from '@/lib/chat-messages'
|
||||
import {
|
||||
$activeSessionId,
|
||||
$awaitingResponse,
|
||||
$busy,
|
||||
$contextSuggestions,
|
||||
$currentCwd,
|
||||
$currentModel,
|
||||
$currentProvider,
|
||||
$freshDraftReady,
|
||||
$gatewayState,
|
||||
$messages,
|
||||
$selectedStoredSessionId,
|
||||
$sessions
|
||||
} from '@/store/session'
|
||||
|
||||
const threadRenderCount = vi.hoisted(() => ({ current: 0 }))
|
||||
|
||||
vi.mock('@/components/assistant-ui/thread', async () => {
|
||||
const React = await import('react')
|
||||
|
||||
return {
|
||||
Thread: () => {
|
||||
threadRenderCount.current += 1
|
||||
|
||||
return React.createElement('div', { 'data-testid': 'thread' })
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
vi.mock('@/components/Backdrop', async () => {
|
||||
const React = await import('react')
|
||||
|
||||
return { Backdrop: () => React.createElement('div', { 'data-testid': 'backdrop' }) }
|
||||
})
|
||||
|
||||
vi.mock('@/components/prompt-overlays', () => ({ PromptOverlays: () => null }))
|
||||
vi.mock('@/components/chat/vibe-hearts', () => ({ COMPOSER_HEART_CONFIG: {}, HeartField: () => null }))
|
||||
vi.mock('@/lib/model-options', () => ({
|
||||
modelOptionsQueryKey: (...parts: unknown[]) => ['model-options', ...parts],
|
||||
requestModelOptions: vi.fn(async () => ({ models: [] }))
|
||||
}))
|
||||
vi.mock('./chat-drop-overlay', () => ({ ChatDropOverlay: () => null }))
|
||||
vi.mock('./chat-swap-overlay', () => ({ ChatSwapOverlay: () => null }))
|
||||
vi.mock('./composer', () => ({ ChatBar: () => null, ChatBarFallback: () => null }))
|
||||
vi.mock('./hooks/use-file-drop-zone', () => ({
|
||||
useFileDropZone: () => ({ dragKind: null, dropHandlers: {} })
|
||||
}))
|
||||
vi.mock('./sidebar/session-actions-menu', async () => {
|
||||
const React = await import('react')
|
||||
|
||||
return {
|
||||
SessionActionsMenu: ({ children }: { children: React.ReactNode }) =>
|
||||
React.createElement('div', { 'data-testid': 'session-actions-menu' }, children)
|
||||
}
|
||||
})
|
||||
|
||||
const { ChatView } = await import('./index')
|
||||
|
||||
function assistantMessage(id: string, text: string): ChatMessage {
|
||||
return {
|
||||
id,
|
||||
parts: [assistantTextPart(text)],
|
||||
role: 'assistant'
|
||||
}
|
||||
}
|
||||
|
||||
describe('ChatView render isolation', () => {
|
||||
beforeEach(() => {
|
||||
threadRenderCount.current = 0
|
||||
$activeSessionId.set('runtime-1')
|
||||
$awaitingResponse.set(false)
|
||||
$busy.set(false)
|
||||
$contextSuggestions.set([])
|
||||
$currentCwd.set('/work')
|
||||
$currentModel.set('test-model')
|
||||
$currentProvider.set('test-provider')
|
||||
$freshDraftReady.set(false)
|
||||
$gatewayState.set('closed')
|
||||
$messages.set([assistantMessage('assistant-1', 'Stable historical answer')])
|
||||
$selectedStoredSessionId.set('stored-1')
|
||||
$sessions.set([{ id: 'stored-1', message_count: 1, title: 'Stable chat' } as never])
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
vi.restoreAllMocks()
|
||||
$activeSessionId.set(null)
|
||||
$awaitingResponse.set(false)
|
||||
$busy.set(false)
|
||||
$contextSuggestions.set([])
|
||||
$currentCwd.set('')
|
||||
$currentModel.set('')
|
||||
$currentProvider.set('')
|
||||
$freshDraftReady.set(false)
|
||||
$gatewayState.set('idle')
|
||||
$messages.set([])
|
||||
$selectedStoredSessionId.set(null)
|
||||
$sessions.set([])
|
||||
})
|
||||
|
||||
it('does not re-render chat history when an unrelated parent idle tick updates', () => {
|
||||
const props = {
|
||||
gateway: null,
|
||||
maxVoiceRecordingSeconds: 120,
|
||||
onAddContextRef: vi.fn(),
|
||||
onAddUrl: vi.fn(),
|
||||
onAttachDroppedItems: vi.fn(),
|
||||
onAttachImageBlob: vi.fn(),
|
||||
onBranchInNewChat: vi.fn(),
|
||||
onCancel: vi.fn(),
|
||||
onDeleteSelectedSession: vi.fn(),
|
||||
onEdit: vi.fn(),
|
||||
onPasteClipboardImage: vi.fn(),
|
||||
onPickFiles: vi.fn(),
|
||||
onPickFolders: vi.fn(),
|
||||
onPickImages: vi.fn(),
|
||||
onReload: vi.fn(),
|
||||
onRemoveAttachment: vi.fn(),
|
||||
onRetryResume: vi.fn(),
|
||||
onSteer: vi.fn(),
|
||||
onSubmit: vi.fn(),
|
||||
onThreadMessagesChange: vi.fn(),
|
||||
onToggleSelectedPin: vi.fn(),
|
||||
onTranscribeAudio: vi.fn()
|
||||
}
|
||||
|
||||
const queryClient = new QueryClient({
|
||||
defaultOptions: { queries: { retry: false } }
|
||||
})
|
||||
|
||||
function ParentTickHarness() {
|
||||
const [tick, setTick] = useState(0)
|
||||
|
||||
return (
|
||||
<QueryClientProvider client={queryClient}>
|
||||
<MemoryRouter initialEntries={['/stored-1']}>
|
||||
<button onClick={() => setTick(value => value + 1)} type="button">
|
||||
parent tick {tick}
|
||||
</button>
|
||||
<ChatView {...props} />
|
||||
</MemoryRouter>
|
||||
</QueryClientProvider>
|
||||
)
|
||||
}
|
||||
|
||||
render(<ParentTickHarness />)
|
||||
|
||||
expect(screen.getByTestId('thread')).toBeTruthy()
|
||||
expect(threadRenderCount.current).toBe(1)
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: /parent tick/i }))
|
||||
|
||||
// memo(ChatView) with stable props must absorb the parent's idle tick —
|
||||
// the transcript (Thread) must not re-render. This is PR #38470's contract.
|
||||
expect(threadRenderCount.current).toBe(1)
|
||||
})
|
||||
})
|
||||
@@ -3,7 +3,7 @@ import { useStore } from '@nanostores/react'
|
||||
import { useQuery } from '@tanstack/react-query'
|
||||
import type { ReadableAtom } from 'nanostores'
|
||||
import type * as React from 'react'
|
||||
import { Suspense, useCallback, useEffect, useMemo, useState } from 'react'
|
||||
import { memo, Suspense, useCallback, useEffect, useMemo, useState } from 'react'
|
||||
import { useLocation } from 'react-router'
|
||||
|
||||
import type { SubmitTextOptions } from '@/app/session/hooks/use-prompt-actions/utils'
|
||||
@@ -240,7 +240,10 @@ function ChatRuntimeBoundary({
|
||||
return <AssistantRuntimeProvider runtime={runtime}>{children}</AssistantRuntimeProvider>
|
||||
}
|
||||
|
||||
export function ChatView({
|
||||
// Memoized: the tile caller (session-tile.tsx) and the contrib surface re-render
|
||||
// on idle ticks unrelated to the chat; with stable callback props (hoisted to
|
||||
// useCallback at the call sites) memo() lets the whole chat shell skip those.
|
||||
export const ChatView = memo(function ChatView({
|
||||
className,
|
||||
gateway,
|
||||
modelMenuContent,
|
||||
@@ -596,4 +599,4 @@ export function ChatView({
|
||||
</ChatRuntimeBoundary>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
})
|
||||
|
||||
@@ -1,7 +1,18 @@
|
||||
import { Profiler, type ProfilerOnRenderCallback, type ReactNode } from 'react'
|
||||
|
||||
import { $terminalTakeover, setTerminalTakeover } from '@/app/right-sidebar/store'
|
||||
import { writeAgentTerminalChunk } from '@/app/right-sidebar/terminal/agent-terminal-stream'
|
||||
import {
|
||||
$activeTerminalId,
|
||||
$terminals,
|
||||
createTerminal,
|
||||
ensureAgentTerminal,
|
||||
selectTerminal,
|
||||
type TerminalEntry
|
||||
} from '@/app/right-sidebar/terminal/terminals'
|
||||
import { $repoStatusByCwd } from '@/store/coding-status'
|
||||
import { $gateway } from '@/store/gateway'
|
||||
import { $messages, setBusy, setMessages } from '@/store/session'
|
||||
import { $currentCwd, $messages, setBusy, setCurrentCwdTransient, setMessages } from '@/store/session'
|
||||
|
||||
type Sample = {
|
||||
id: string
|
||||
@@ -38,6 +49,12 @@ declare global {
|
||||
* backend) doesn't contaminate frame-pacing numbers.
|
||||
*/
|
||||
connected: () => boolean
|
||||
/** Mount files + multiple xterms for the synthetic right-pane scenario. */
|
||||
rightPaneSetup: (opts: { cwd: string; terminals?: number }) => { procId: string; terminalIds: string[] }
|
||||
rightPaneGit: (path: string, kind?: 'added' | 'conflicted' | 'modified') => void
|
||||
rightPaneReset: () => void
|
||||
rightPaneSelect: (id: string) => void
|
||||
rightPaneWrite: (procId: string, chunk: string) => void
|
||||
reset: () => void
|
||||
snapshotMsgs: () => number
|
||||
}
|
||||
@@ -102,11 +119,32 @@ if (typeof window !== 'undefined' && !window.__PERF_DRIVE__) {
|
||||
let baseline: ReturnType<typeof $messages.get> | null = null
|
||||
let activeHandle: SyntheticDriverHandle | null = null
|
||||
|
||||
let rightPaneBaseline: null | {
|
||||
activeTerminalId: null | string
|
||||
cwd: string
|
||||
repoStatusByCwd: ReturnType<typeof $repoStatusByCwd.get>
|
||||
takeover: boolean
|
||||
terminals: readonly TerminalEntry[]
|
||||
} = null
|
||||
|
||||
const stop = () => {
|
||||
activeHandle = null
|
||||
setBusy(false)
|
||||
}
|
||||
|
||||
const resetRightPane = () => {
|
||||
if (!rightPaneBaseline) {
|
||||
return
|
||||
}
|
||||
|
||||
setTerminalTakeover(rightPaneBaseline.takeover)
|
||||
$terminals.set(rightPaneBaseline.terminals)
|
||||
$activeTerminalId.set(rightPaneBaseline.activeTerminalId)
|
||||
$repoStatusByCwd.set(rightPaneBaseline.repoStatusByCwd)
|
||||
setCurrentCwdTransient(rightPaneBaseline.cwd)
|
||||
rightPaneBaseline = null
|
||||
}
|
||||
|
||||
// One synthetic turn's worth of mixed markdown — prose, a list, a fenced
|
||||
// code block, inline code, a link, and a short table — so a loaded transcript
|
||||
// exercises the same render cost (Streamdown blocks, code cards) a real one
|
||||
@@ -166,6 +204,69 @@ if (typeof window !== 'undefined' && !window.__PERF_DRIVE__) {
|
||||
return false
|
||||
}
|
||||
},
|
||||
rightPaneGit: (path, kind = 'modified') => {
|
||||
const file = {
|
||||
conflicted: kind === 'conflicted',
|
||||
path,
|
||||
staged: false,
|
||||
unstaged: kind === 'modified',
|
||||
untracked: kind === 'added'
|
||||
}
|
||||
|
||||
const cwd = $currentCwd.get().trim()
|
||||
$repoStatusByCwd.set({
|
||||
...$repoStatusByCwd.get(),
|
||||
[cwd]: {
|
||||
added: 0,
|
||||
ahead: 0,
|
||||
behind: 0,
|
||||
branch: 'perf',
|
||||
changed: 1,
|
||||
conflicted: kind === 'conflicted' ? 1 : 0,
|
||||
defaultBranch: 'main',
|
||||
detached: false,
|
||||
files: [file],
|
||||
removed: 0,
|
||||
staged: 0,
|
||||
unstaged: kind === 'modified' ? 1 : 0,
|
||||
untracked: kind === 'added' ? 1 : 0
|
||||
}
|
||||
})
|
||||
},
|
||||
rightPaneReset: resetRightPane,
|
||||
rightPaneSelect: selectTerminal,
|
||||
rightPaneSetup: ({ cwd, terminals = 3 }) => {
|
||||
resetRightPane()
|
||||
rightPaneBaseline = {
|
||||
activeTerminalId: $activeTerminalId.get(),
|
||||
cwd: $currentCwd.get(),
|
||||
repoStatusByCwd: $repoStatusByCwd.get(),
|
||||
takeover: $terminalTakeover.get(),
|
||||
terminals: $terminals.get()
|
||||
}
|
||||
|
||||
setCurrentCwdTransient(cwd)
|
||||
const terminalIds = [createTerminal(cwd)]
|
||||
let procId = ''
|
||||
|
||||
for (let index = 1; index < Math.max(1, terminals); index += 1) {
|
||||
procId = `right-pane-perf-${Date.now()}-${index}`
|
||||
const id = ensureAgentTerminal(procId, `perf output ${index}`)
|
||||
|
||||
if (id) {
|
||||
terminalIds.push(id)
|
||||
}
|
||||
}
|
||||
|
||||
if (procId) {
|
||||
selectTerminal(terminalIds.at(-1) ?? terminalIds[0])
|
||||
}
|
||||
|
||||
setTerminalTakeover(true)
|
||||
|
||||
return { procId, terminalIds }
|
||||
},
|
||||
rightPaneWrite: (procId, chunk) => writeAgentTerminalChunk(procId, chunk),
|
||||
loadTranscript: (turns = 200) => {
|
||||
if (!baseline) {
|
||||
baseline = $messages.get()
|
||||
@@ -190,6 +291,7 @@ if (typeof window !== 'undefined' && !window.__PERF_DRIVE__) {
|
||||
},
|
||||
reset: () => {
|
||||
activeHandle?.stop()
|
||||
resetRightPane()
|
||||
|
||||
if (baseline) {
|
||||
setMessages(baseline)
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { useStore } from '@nanostores/react'
|
||||
|
||||
import { StatusPulse } from '@/components/ui/status-pulse'
|
||||
import { type Translations, useI18n } from '@/i18n'
|
||||
import { useStoreSelector } from '@/lib/use-session-slice'
|
||||
import { cn } from '@/lib/utils'
|
||||
@@ -17,6 +18,10 @@ import { type SessionDotState, sessionDotState } from './sidebar/session-row-sta
|
||||
type DotVariant = {
|
||||
ariaLabel?: (r: Translations['sidebar']['row']) => string
|
||||
className: string
|
||||
pulse?: {
|
||||
className: string
|
||||
opacity: number
|
||||
}
|
||||
role?: 'status'
|
||||
title?: (r: Translations['sidebar']['row']) => string
|
||||
}
|
||||
@@ -24,12 +29,6 @@ type DotVariant = {
|
||||
// Shared base for every active dot; idle is smaller and uses its own class.
|
||||
const DOT_BASE = 'relative size-1.5 rounded-full'
|
||||
|
||||
// Pseudo-element ping ring that scales outward and fades — shared scaffold for
|
||||
// the two pulsing dots. The `before:bg-*` color is written inline per variant
|
||||
// (NOT interpolated here): Tailwind only generates utilities it can see as
|
||||
// complete static strings, so a `before:bg-${color}` template never emits.
|
||||
const PING = "before:absolute before:inset-0 before:animate-ping before:rounded-full before:content-['']"
|
||||
|
||||
const DOT_VARIANTS: Record<SessionDotState, DotVariant> = {
|
||||
// Amber steady — a clarify/approval is blocking the turn. Steady (not
|
||||
// pulsing) reads as "your turn", distinct from the accent pulse of a turn.
|
||||
@@ -42,14 +41,22 @@ const DOT_VARIANTS: Record<SessionDotState, DotVariant> = {
|
||||
// Accent pulse — the LLM turn is actively running.
|
||||
working: {
|
||||
ariaLabel: r => r.sessionRunning,
|
||||
className: `${DOT_BASE} bg-(--ui-accent) shadow-[0_0_0.625rem_color-mix(in_srgb,var(--ui-accent)_55%,transparent)] ${PING} before:bg-(--ui-accent) before:opacity-70`,
|
||||
className: `${DOT_BASE} bg-(--ui-accent) shadow-[0_0_0.625rem_color-mix(in_srgb,var(--ui-accent)_55%,transparent)]`,
|
||||
pulse: {
|
||||
className: 'absolute inset-0 rounded-full bg-(--ui-accent) opacity-0',
|
||||
opacity: 0.7
|
||||
},
|
||||
role: 'status'
|
||||
},
|
||||
// Quiet accent pulse — the turn is still authoritative-running, but no
|
||||
// stream activity has arrived for the watchdog window.
|
||||
stalled: {
|
||||
ariaLabel: r => r.sessionRunning,
|
||||
className: `${DOT_BASE} bg-(--ui-accent) opacity-70 ${PING} before:bg-(--ui-accent) before:opacity-40`,
|
||||
className: `${DOT_BASE} bg-(--ui-accent) opacity-70`,
|
||||
pulse: {
|
||||
className: 'absolute inset-0 rounded-full bg-(--ui-accent) opacity-0',
|
||||
opacity: 0.4
|
||||
},
|
||||
role: 'status',
|
||||
title: r => r.sessionRunning
|
||||
},
|
||||
@@ -58,7 +65,11 @@ const DOT_VARIANTS: Record<SessionDotState, DotVariant> = {
|
||||
// than muted-foreground so it's visible against the surface.
|
||||
background: {
|
||||
ariaLabel: r => r.backgroundRunning,
|
||||
className: `${DOT_BASE} bg-muted-foreground/80 ${PING} before:bg-muted-foreground/80 before:opacity-60`,
|
||||
className: `${DOT_BASE} bg-muted-foreground/80`,
|
||||
pulse: {
|
||||
className: 'absolute inset-0 rounded-full bg-muted-foreground/80 opacity-0',
|
||||
opacity: 0.6
|
||||
},
|
||||
role: 'status',
|
||||
title: r => r.backgroundRunning
|
||||
},
|
||||
@@ -123,6 +134,7 @@ export function SessionStatusDot({ storedSessionId, session, branchStem, classNa
|
||||
const hasBackground = useStoreSelector($backgroundRunningSessionIds, ids => ids.includes(storedSessionId))
|
||||
|
||||
const dotState = sessionDotState({ hasBackground, isStalled, isUnread, isWorking, needsInput })
|
||||
const variant = DOT_VARIANTS[dotState]
|
||||
|
||||
return (
|
||||
<span className={cn('flex items-center gap-0.5', className)}>
|
||||
@@ -135,11 +147,20 @@ export function SessionStatusDot({ storedSessionId, session, branchStem, classNa
|
||||
<span aria-hidden="true" className="size-1 rounded-full" style={{ backgroundColor: color }} />
|
||||
) : (
|
||||
<span
|
||||
aria-label={DOT_VARIANTS[dotState].ariaLabel?.(r)}
|
||||
className={DOT_VARIANTS[dotState].className}
|
||||
role={DOT_VARIANTS[dotState].role}
|
||||
title={DOT_VARIANTS[dotState].title?.(r)}
|
||||
/>
|
||||
aria-label={variant.ariaLabel?.(r)}
|
||||
className={variant.className}
|
||||
role={variant.role}
|
||||
title={variant.title?.(r)}
|
||||
>
|
||||
{variant.pulse ? (
|
||||
<StatusPulse
|
||||
aria-hidden="true"
|
||||
className={variant.pulse.className}
|
||||
kind="ping"
|
||||
opacity={variant.pulse.opacity}
|
||||
/>
|
||||
) : null}
|
||||
</span>
|
||||
)}
|
||||
</span>
|
||||
)
|
||||
|
||||
@@ -104,6 +104,13 @@ function buildTileView(storedSessionId: string): SessionView {
|
||||
}
|
||||
}
|
||||
|
||||
// Module-level constants so these ChatView props are referentially stable —
|
||||
// tiles have no pin/delete affordance, and transcription needs no per-tile state.
|
||||
const noop = () => undefined
|
||||
|
||||
const tileTranscribeAudio = async (audio: Blob) =>
|
||||
(await transcribeAudio(await blobToDataUrl(audio), audio.type)).transcript
|
||||
|
||||
function TileChat({
|
||||
runtimeId,
|
||||
storedSessionId,
|
||||
@@ -144,6 +151,29 @@ function TileChat({
|
||||
scope: { add: attachments.add, remove: attachments.remove, target: scope.target }
|
||||
})
|
||||
|
||||
// ChatView is memo()d — every callback prop must be referentially stable or
|
||||
// the memo never holds and each tile-level render (idle ticks, unrelated
|
||||
// store updates) re-renders the whole chat shell. The individual composer
|
||||
// functions are useCallback'd inside useComposerActions, so hoisting these
|
||||
// wrappers onto them keeps identity stable across renders.
|
||||
const { addContextRefAttachment, pasteClipboardImage, pickContextPaths, pickImages, removeAttachment } = composer
|
||||
|
||||
const onAddUrl = useCallback(
|
||||
(url: string) => addContextRefAttachment(`@url:${formatRefValue(url)}`, url),
|
||||
[addContextRefAttachment]
|
||||
)
|
||||
|
||||
const onPasteClipboardImage = useCallback(
|
||||
(opts?: { silent?: boolean }) => pasteClipboardImage(opts),
|
||||
[pasteClipboardImage]
|
||||
)
|
||||
|
||||
const onPickFiles = useCallback(() => void pickContextPaths('file'), [pickContextPaths])
|
||||
const onPickFolders = useCallback(() => void pickContextPaths('folder'), [pickContextPaths])
|
||||
const onPickImages = useCallback(() => void pickImages(), [pickImages])
|
||||
const onRemoveAttachment = useCallback((id: string) => void removeAttachment(id), [removeAttachment])
|
||||
const onRetryResume = useCallback(() => patchSessionTile(storedSessionId, { error: undefined }), [storedSessionId])
|
||||
|
||||
// Per-tile model menu — rendered under this tile's SessionView so the pill
|
||||
// + switch target THIS runtime, not the primary (which may be mid-turn).
|
||||
const modelMenuContent = useMemo(
|
||||
@@ -165,27 +195,27 @@ function TileChat({
|
||||
<ChatView
|
||||
gateway={gateway}
|
||||
modelMenuContent={modelMenuContent}
|
||||
onAddContextRef={composer.addContextRefAttachment}
|
||||
onAddUrl={url => composer.addContextRefAttachment(`@url:${formatRefValue(url)}`, url)}
|
||||
onAddContextRef={addContextRefAttachment}
|
||||
onAddUrl={onAddUrl}
|
||||
onAttachDroppedItems={composer.attachDroppedItems}
|
||||
onAttachImageBlob={composer.attachImageBlob}
|
||||
onCancel={actions.cancelRun}
|
||||
onDeleteSelectedSession={() => undefined}
|
||||
onDeleteSelectedSession={noop}
|
||||
onDismissError={actions.dismissError}
|
||||
onEdit={actions.editMessage}
|
||||
onPasteClipboardImage={opts => composer.pasteClipboardImage(opts)}
|
||||
onPickFiles={() => void composer.pickContextPaths('file')}
|
||||
onPickFolders={() => void composer.pickContextPaths('folder')}
|
||||
onPickImages={() => void composer.pickImages()}
|
||||
onPasteClipboardImage={onPasteClipboardImage}
|
||||
onPickFiles={onPickFiles}
|
||||
onPickFolders={onPickFolders}
|
||||
onPickImages={onPickImages}
|
||||
onReload={actions.reloadFromMessage}
|
||||
onRemoveAttachment={id => void composer.removeAttachment(id)}
|
||||
onRemoveAttachment={onRemoveAttachment}
|
||||
onRestoreToMessage={actions.restoreToMessage}
|
||||
onRetryResume={() => patchSessionTile(storedSessionId, { error: undefined })}
|
||||
onRetryResume={onRetryResume}
|
||||
onSteer={actions.steerPrompt}
|
||||
onSubmit={actions.submitText}
|
||||
onThreadMessagesChange={actions.handleThreadMessagesChange}
|
||||
onToggleSelectedPin={() => undefined}
|
||||
onTranscribeAudio={async audio => (await transcribeAudio(await blobToDataUrl(audio), audio.type)).transcript}
|
||||
onToggleSelectedPin={noop}
|
||||
onTranscribeAudio={tileTranscribeAudio}
|
||||
/>
|
||||
</ComposerScopeProvider>
|
||||
</SessionViewProvider>
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { useStore } from '@nanostores/react'
|
||||
import { useEffect, useMemo, useState } from 'react'
|
||||
|
||||
import { usePaneVisible } from '@/components/pane-shell/pane-visibility'
|
||||
import { ActionsContextMenu, type MenuKit, renderActionItem } from '@/components/ui/actions-menu'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { DisclosureCaret } from '@/components/ui/disclosure-caret'
|
||||
@@ -92,17 +93,19 @@ export function SidebarCronJobsSection({
|
||||
// Rows revealed so far; starts compact, grows in steps via "load more".
|
||||
const [visibleCount, setVisibleCount] = useState(INITIAL_VISIBLE_JOBS)
|
||||
|
||||
const visible = usePaneVisible()
|
||||
|
||||
// One clock for the whole section (rows are pure) so the countdowns tick
|
||||
// without re-rendering the rest of the sidebar. Only runs while expanded.
|
||||
// without re-rendering the rest of the sidebar. Only runs while expanded and visible.
|
||||
useEffect(() => {
|
||||
if (!open) {
|
||||
if (!open || !visible) {
|
||||
return
|
||||
}
|
||||
|
||||
const id = window.setInterval(() => setNowMs(Date.now()), 1000)
|
||||
|
||||
return () => window.clearInterval(id)
|
||||
}, [open])
|
||||
}, [open, visible])
|
||||
|
||||
// Upcoming first (soonest next run), jobs with no next run sink to the bottom,
|
||||
// then alphabetical for stability.
|
||||
@@ -328,6 +331,7 @@ function CronJobSidebarRuns({ jobId, onOpenRun }: { jobId: string; onOpenRun: (s
|
||||
const changeEventsAvailable = useStore($changeEventsAvailable)
|
||||
const cronChangeTick = useStore($cronChangeTick)
|
||||
const [runs, setRuns] = useState<null | SessionInfo[]>(null)
|
||||
const visible = usePaneVisible()
|
||||
|
||||
useEffect(() => {
|
||||
let cancelled = false
|
||||
@@ -345,6 +349,15 @@ function CronJobSidebarRuns({ jobId, onOpenRun }: { jobId: string; onOpenRun: (s
|
||||
}
|
||||
})
|
||||
|
||||
// Hidden pane: skip the peek entirely — no initial load, no interval.
|
||||
// `visible` is in the dep array, so becoming visible re-runs this effect
|
||||
// and starts the load + timer fresh (same shape as the section clock).
|
||||
if (!visible) {
|
||||
return () => {
|
||||
cancelled = true
|
||||
}
|
||||
}
|
||||
|
||||
void load()
|
||||
|
||||
const intervalId = window.setInterval(
|
||||
@@ -361,7 +374,7 @@ function CronJobSidebarRuns({ jobId, onOpenRun }: { jobId: string; onOpenRun: (s
|
||||
window.clearInterval(intervalId)
|
||||
}
|
||||
// cronChangeTick: a fired run reloads the peek immediately.
|
||||
}, [changeEventsAvailable, cronChangeTick, jobId])
|
||||
}, [changeEventsAvailable, cronChangeTick, jobId, visible])
|
||||
|
||||
return (
|
||||
<div className="mb-1 ml-[1.375rem] flex flex-col gap-px">
|
||||
|
||||
@@ -7,6 +7,7 @@ import {
|
||||
closeAllTreeTabs,
|
||||
closeOtherTreeTabs,
|
||||
closeTreeTabsToRight,
|
||||
reloadTreePane,
|
||||
treeTabCloseTargets
|
||||
} from '@/components/pane-shell/tree/store'
|
||||
import {
|
||||
@@ -239,12 +240,24 @@ function useSessionActions({
|
||||
})
|
||||
]
|
||||
|
||||
// TAB — close verbs that act on the strip (tabs only; a row isn't a tab).
|
||||
// TAB — verbs that act on the strip (tabs only; a row isn't a tab).
|
||||
const closeTargets = surface === 'tab' && tabPaneId ? treeTabCloseTargets(tabPaneId) : null
|
||||
|
||||
const tabCloseItems: ActionItemSpec[] =
|
||||
const tabItems: ActionItemSpec[] =
|
||||
surface === 'tab'
|
||||
? [
|
||||
...(tabPaneId
|
||||
? [
|
||||
spec({
|
||||
icon: 'refresh',
|
||||
label: t.zones.reload,
|
||||
onSelect: () => {
|
||||
triggerHaptic('selection')
|
||||
reloadTreePane(tabPaneId)
|
||||
}
|
||||
})
|
||||
]
|
||||
: []),
|
||||
...(onClose
|
||||
? [
|
||||
spec({
|
||||
@@ -342,10 +355,10 @@ function useSessionActions({
|
||||
/>
|
||||
<kit.Separator />
|
||||
{workItems.map(item => renderActionItem(kit, item))}
|
||||
{tabCloseItems.length > 0 && (
|
||||
{tabItems.length > 0 && (
|
||||
<>
|
||||
<kit.Separator />
|
||||
{tabCloseItems.map(item => renderActionItem(kit, item))}
|
||||
{tabItems.map(item => renderActionItem(kit, item))}
|
||||
</>
|
||||
)}
|
||||
<kit.Separator />
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
import { cleanup, render } from '@testing-library/react'
|
||||
import type * as React from 'react'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import type { SessionInfo } from '@/hermes'
|
||||
|
||||
import { SidebarSessionsSection, VIRTUALIZE_THRESHOLD } from './sessions-section'
|
||||
import type { VirtualSessionListProps } from './virtual-session-list'
|
||||
|
||||
afterEach(cleanup)
|
||||
|
||||
vi.mock('@/i18n', () => ({
|
||||
useI18n: () => ({
|
||||
t: {
|
||||
sidebar: {
|
||||
dateDivider: {
|
||||
earlierThisMonth: 'Earlier this month',
|
||||
lastMonth: 'Last month',
|
||||
lastWeek: 'Last week',
|
||||
older: 'Older',
|
||||
today: 'Today',
|
||||
yesterday: 'Yesterday'
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
}))
|
||||
|
||||
const mockVirtualListPropsHistory: VirtualSessionListProps[] = []
|
||||
|
||||
vi.mock('./virtual-session-list', () => ({
|
||||
VirtualSessionList: (props: VirtualSessionListProps) => {
|
||||
mockVirtualListPropsHistory.push(props)
|
||||
|
||||
return <div data-testid="virtual-session-list">Virtual List ({props.rows.length} rows)</div>
|
||||
}
|
||||
}))
|
||||
|
||||
vi.mock('./session-row', () => ({
|
||||
SidebarSessionRow: ({ session }: { session: SessionInfo }) => (
|
||||
<div data-testid={`session-row-${session.id}`}>{session.id}</div>
|
||||
)
|
||||
}))
|
||||
|
||||
function makeSession(id: string, startedAt = 1000): SessionInfo {
|
||||
return {
|
||||
handoff_platform: null,
|
||||
handoff_state: null,
|
||||
id,
|
||||
last_active: startedAt,
|
||||
profile: 'default',
|
||||
started_at: startedAt
|
||||
} as unknown as SessionInfo
|
||||
}
|
||||
|
||||
function generateSessions(count: number): SessionInfo[] {
|
||||
return Array.from({ length: count }, (_, i) => makeSession(`session-${i + 1}`, 10000 - i * 100))
|
||||
}
|
||||
|
||||
const noop = () => {}
|
||||
|
||||
describe('SidebarSessionsSection memoization & virtualizer stability', () => {
|
||||
it('memoizes flatRows and passes the exact same rows array reference across parent re-renders', () => {
|
||||
mockVirtualListPropsHistory.length = 0
|
||||
|
||||
const sessions = generateSessions(VIRTUALIZE_THRESHOLD + 5)
|
||||
|
||||
const { rerender } = render(
|
||||
<SidebarSessionsSection
|
||||
activeSessionId={null}
|
||||
emptyState={<div>Empty</div>}
|
||||
label="Sessions"
|
||||
onArchiveSession={noop}
|
||||
onDeleteSession={noop}
|
||||
onResumeSession={noop}
|
||||
onToggle={noop}
|
||||
onTogglePin={noop}
|
||||
open={true}
|
||||
pinned={false}
|
||||
sessions={sessions}
|
||||
workingSessionIdSet={new Set()}
|
||||
/>
|
||||
)
|
||||
|
||||
expect(mockVirtualListPropsHistory.length).toBe(1)
|
||||
const initialRowsRef = mockVirtualListPropsHistory[0].rows
|
||||
expect(initialRowsRef.length).toBeGreaterThan(VIRTUALIZE_THRESHOLD)
|
||||
|
||||
// Re-render parent with the exact same sessions array and props
|
||||
rerender(
|
||||
<SidebarSessionsSection
|
||||
activeSessionId={null}
|
||||
emptyState={<div>Empty</div>}
|
||||
label="Sessions"
|
||||
onArchiveSession={noop}
|
||||
onDeleteSession={noop}
|
||||
onResumeSession={noop}
|
||||
onToggle={noop}
|
||||
onTogglePin={noop}
|
||||
open={true}
|
||||
pinned={false}
|
||||
sessions={sessions}
|
||||
workingSessionIdSet={new Set()}
|
||||
/>
|
||||
)
|
||||
|
||||
expect(mockVirtualListPropsHistory.length).toBe(2)
|
||||
const nextRowsRef = mockVirtualListPropsHistory[1].rows
|
||||
|
||||
// Confirm that the flatRows array reference remains strictly identical across renders (useMemo proof)
|
||||
expect(nextRowsRef).toBe(initialRowsRef)
|
||||
})
|
||||
|
||||
it('re-computes flatRows reference when dateGrouped or sessions change', () => {
|
||||
mockVirtualListPropsHistory.length = 0
|
||||
|
||||
const initialSessions = generateSessions(VIRTUALIZE_THRESHOLD + 2)
|
||||
|
||||
const { rerender } = render(
|
||||
<SidebarSessionsSection
|
||||
activeSessionId={null}
|
||||
dateGrouped={false}
|
||||
emptyState={<div>Empty</div>}
|
||||
label="Sessions"
|
||||
onArchiveSession={noop}
|
||||
onDeleteSession={noop}
|
||||
onResumeSession={noop}
|
||||
onToggle={noop}
|
||||
onTogglePin={noop}
|
||||
open={true}
|
||||
pinned={false}
|
||||
sessions={initialSessions}
|
||||
workingSessionIdSet={new Set()}
|
||||
/>
|
||||
)
|
||||
|
||||
const firstRowsRef = mockVirtualListPropsHistory[0].rows
|
||||
|
||||
// Change dateGrouped to true
|
||||
rerender(
|
||||
<SidebarSessionsSection
|
||||
activeSessionId={null}
|
||||
dateGrouped={true}
|
||||
emptyState={<div>Empty</div>}
|
||||
label="Sessions"
|
||||
onArchiveSession={noop}
|
||||
onDeleteSession={noop}
|
||||
onResumeSession={noop}
|
||||
onToggle={noop}
|
||||
onTogglePin={noop}
|
||||
open={true}
|
||||
pinned={false}
|
||||
sessions={initialSessions}
|
||||
workingSessionIdSet={new Set()}
|
||||
/>
|
||||
)
|
||||
|
||||
const secondRowsRef = mockVirtualListPropsHistory[1].rows
|
||||
expect(secondRowsRef).not.toBe(firstRowsRef)
|
||||
|
||||
// Change sessions array identity
|
||||
const updatedSessions = generateSessions(VIRTUALIZE_THRESHOLD + 4)
|
||||
rerender(
|
||||
<SidebarSessionsSection
|
||||
activeSessionId={null}
|
||||
dateGrouped={true}
|
||||
emptyState={<div>Empty</div>}
|
||||
label="Sessions"
|
||||
onArchiveSession={noop}
|
||||
onDeleteSession={noop}
|
||||
onResumeSession={noop}
|
||||
onToggle={noop}
|
||||
onTogglePin={noop}
|
||||
open={true}
|
||||
pinned={false}
|
||||
sessions={updatedSessions}
|
||||
workingSessionIdSet={new Set()}
|
||||
/>
|
||||
)
|
||||
|
||||
const thirdRowsRef = mockVirtualListPropsHistory[2].rows
|
||||
expect(thirdRowsRef).not.toBe(secondRowsRef)
|
||||
})
|
||||
})
|
||||
@@ -1,6 +1,6 @@
|
||||
import type { useSensors } from '@dnd-kit/core'
|
||||
import type * as React from 'react'
|
||||
import { useMemo } from 'react'
|
||||
import { useCallback, useMemo } from 'react'
|
||||
|
||||
import { SidebarPanelLabel } from '@/app/shell/sidebar-label'
|
||||
import { DisclosureCaret } from '@/components/ui/disclosure-caret'
|
||||
@@ -225,52 +225,79 @@ export function SidebarSessionsSection({
|
||||
[sessions, preserveInputOrder]
|
||||
)
|
||||
|
||||
const renderRow = (session: SessionInfo, draggable: boolean, branchStem?: string) => {
|
||||
const rowProps = {
|
||||
branchStem,
|
||||
isPinned: pinned,
|
||||
isSelected: session.id === activeSessionId,
|
||||
isWorking: workingSessionIdSet.has(session.id),
|
||||
onArchive: () => onArchiveSession(session.id),
|
||||
onBranch: onBranchSession ? () => onBranchSession(session.id, session.profile) : undefined,
|
||||
onDelete: () => onDeleteSession(session.id),
|
||||
onPin: () => onTogglePin(sessionPinId(session)),
|
||||
onResume: () => onResumeSession(session.id),
|
||||
reorderable: draggable && !branchStem,
|
||||
session,
|
||||
showProfile: showProfileTags
|
||||
}
|
||||
const renderRow = useCallback(
|
||||
(session: SessionInfo, draggable: boolean, branchStem?: string) => {
|
||||
const rowProps = {
|
||||
branchStem,
|
||||
isPinned: pinned,
|
||||
isSelected: session.id === activeSessionId,
|
||||
isWorking: workingSessionIdSet.has(session.id),
|
||||
onArchive: () => onArchiveSession(session.id),
|
||||
onBranch: onBranchSession ? () => onBranchSession(session.id, session.profile) : undefined,
|
||||
onDelete: () => onDeleteSession(session.id),
|
||||
onPin: () => onTogglePin(sessionPinId(session)),
|
||||
onResume: () => onResumeSession(session.id),
|
||||
reorderable: draggable && !branchStem,
|
||||
session,
|
||||
showProfile: showProfileTags
|
||||
}
|
||||
|
||||
return draggable && !branchStem ? (
|
||||
<SortableSidebarSessionRow key={session.id} {...rowProps} />
|
||||
) : (
|
||||
<SidebarSessionRow key={session.id} {...rowProps} />
|
||||
)
|
||||
}
|
||||
return draggable && !branchStem ? (
|
||||
<SortableSidebarSessionRow key={session.id} {...rowProps} />
|
||||
) : (
|
||||
<SidebarSessionRow key={session.id} {...rowProps} />
|
||||
)
|
||||
},
|
||||
[
|
||||
activeSessionId,
|
||||
onArchiveSession,
|
||||
onBranchSession,
|
||||
onDeleteSession,
|
||||
onResumeSession,
|
||||
onTogglePin,
|
||||
pinned,
|
||||
showProfileTags,
|
||||
workingSessionIdSet
|
||||
]
|
||||
)
|
||||
|
||||
// A single flat/virtual/lane list row — either a date divider or a session.
|
||||
const renderListRow = (row: SidebarListRow, draggable: boolean) =>
|
||||
row.kind === 'divider' ? (
|
||||
<SidebarDateDivider key={row.key} label={sessionBucketLabel(row.bucket, dividerLabels)} />
|
||||
) : (
|
||||
renderRow(row.entry.session, draggable, row.entry.branchStem)
|
||||
)
|
||||
const renderListRow = useCallback(
|
||||
(row: SidebarListRow, draggable: boolean) =>
|
||||
row.kind === 'divider' ? (
|
||||
<SidebarDateDivider key={row.key} label={sessionBucketLabel(row.bucket, dividerLabels)} />
|
||||
) : (
|
||||
renderRow(row.entry.session, draggable, row.entry.branchStem)
|
||||
),
|
||||
[dividerLabels, renderRow]
|
||||
)
|
||||
|
||||
// Sessions inside repos/worktrees are date-ordered and static.
|
||||
const renderRows = (items: SessionInfo[]) =>
|
||||
flattenSessionsWithBranches(items).map(({ branchStem, session }) => renderRow(session, false, branchStem))
|
||||
const renderRows = useCallback(
|
||||
(items: SessionInfo[]) =>
|
||||
flattenSessionsWithBranches(items).map(({ branchStem, session }) => renderRow(session, false, branchStem)),
|
||||
[renderRow]
|
||||
)
|
||||
|
||||
// Same as `renderRows`, but with date dividers folded in — used for
|
||||
// entered-project lanes so a lane spanning multiple days reads
|
||||
// chronologically, matching the flat recents list.
|
||||
const renderRowsDated = (items: SessionInfo[]) => {
|
||||
const entries = flattenSessionsWithBranches(items)
|
||||
const renderRowsDated = useCallback(
|
||||
(items: SessionInfo[]) => {
|
||||
const entries = flattenSessionsWithBranches(items)
|
||||
|
||||
return (dateGrouped ? groupEntriesByRecency(entries) : toSessionRows(entries)).map(row => renderListRow(row, false))
|
||||
}
|
||||
return (dateGrouped ? groupEntriesByRecency(entries) : toSessionRows(entries)).map(row =>
|
||||
renderListRow(row, false)
|
||||
)
|
||||
},
|
||||
[dateGrouped, renderListRow]
|
||||
)
|
||||
|
||||
// Flat recents as list rows: grouped by recency when enabled, plain otherwise.
|
||||
const flatRows: SidebarListRow[] = dateGrouped ? groupEntriesByRecency(displayEntries) : toSessionRows(displayEntries)
|
||||
const flatRows: SidebarListRow[] = useMemo(
|
||||
() => (dateGrouped ? groupEntriesByRecency(displayEntries) : toSessionRows(displayEntries)),
|
||||
[dateGrouped, displayEntries]
|
||||
)
|
||||
|
||||
const flatVirtualized =
|
||||
!showEmptyState &&
|
||||
|
||||
@@ -27,7 +27,7 @@ interface SessionRowCommonProps {
|
||||
showProfile?: boolean
|
||||
}
|
||||
|
||||
interface VirtualSessionListProps {
|
||||
export interface VirtualSessionListProps {
|
||||
activeSessionId: null | string
|
||||
className?: string
|
||||
rows: SidebarListRow[]
|
||||
|
||||
@@ -77,7 +77,8 @@ describe('useSessionTileDelegate resumeTile', () => {
|
||||
expect(requestGateway).toHaveBeenCalledWith('session.resume', {
|
||||
session_id: 'stored-x',
|
||||
cols: 96,
|
||||
profile: 'ai-engineer'
|
||||
profile: 'ai-engineer',
|
||||
omit_messages: true
|
||||
})
|
||||
})
|
||||
|
||||
@@ -94,7 +95,8 @@ describe('useSessionTileDelegate resumeTile', () => {
|
||||
expect(requestGateway).toHaveBeenCalledWith('session.resume', {
|
||||
session_id: 'stored-y',
|
||||
cols: 96,
|
||||
profile: 'default'
|
||||
profile: 'default',
|
||||
omit_messages: true
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
@@ -79,6 +79,7 @@ export function useSessionTileDelegate({
|
||||
requestGateway<SessionResumeResponse>('session.resume', {
|
||||
session_id: storedSessionId,
|
||||
cols: 96,
|
||||
omit_messages: true,
|
||||
...(profile ? { profile } : {})
|
||||
})
|
||||
])
|
||||
|
||||
@@ -5,6 +5,7 @@ import type { HermesConnection } from '@/global'
|
||||
import { HermesGateway } from '@/hermes'
|
||||
import { translateNow } from '@/i18n'
|
||||
import { desktopDefaultCwd } from '@/lib/desktop-fs'
|
||||
import { reconnectBackoffDelayMs } from '@/lib/reconnect-backoff'
|
||||
import {
|
||||
$desktopBoot,
|
||||
applyDesktopBootProgress,
|
||||
@@ -42,12 +43,16 @@ import type { RpcEvent } from '@/types/hermes'
|
||||
|
||||
import { stashGatewaySurvivor, survivorIsStale, takeGatewaySurvivor } from './gateway-hmr-survivor'
|
||||
|
||||
// After this many consecutive failed reconnects (≈45s with the 1→15s backoff)
|
||||
// raise a recoverable boot error. Otherwise a dropped remote gateway loops the
|
||||
// backoff forever behind the fullscreen CONNECTING overlay with no way to reach
|
||||
// Settings / sign in / switch to local — the "lost connection breaks the app"
|
||||
// dead end. The next successful reconnect clears it.
|
||||
const RECONNECT_ESCALATE_AFTER = 6
|
||||
// After the reconnect loop has been failing for this long, raise a recoverable
|
||||
// boot error. Otherwise a dropped remote gateway loops the backoff forever
|
||||
// behind the fullscreen CONNECTING overlay with no way to reach Settings /
|
||||
// sign in / switch to local — the "lost connection breaks the app" dead end.
|
||||
// The next successful reconnect clears it. Time-based (not attempt-count)
|
||||
// because the full-jitter backoff makes attempt counts a meaningless clock:
|
||||
// six jittered attempts can elapse in ~9s, while the old deterministic
|
||||
// 1→15s ladder took ~45s to reach six failures — this threshold keeps that
|
||||
// original ~45s calibration.
|
||||
const RECONNECT_ESCALATE_AFTER_MS = 45_000
|
||||
|
||||
interface GatewayBootOptions {
|
||||
beforeConnectionSwitch: () => void
|
||||
@@ -114,13 +119,18 @@ export function useGatewayBoot({
|
||||
let reconnecting = false
|
||||
let reconnectTimer: ReturnType<typeof setTimeout> | null = null
|
||||
let reconnectAttempt = 0
|
||||
// Wall-clock start of the current disconnect episode (first failed
|
||||
// reconnect attempt); null while healthy. Drives the time-based
|
||||
// escalation below. Reset on a clean open or a manual/wake reconnect.
|
||||
let reconnectFailingSince: number | null = null
|
||||
// Surface "sign in again" once per disconnect episode, not on every backoff
|
||||
// tick — a stale OAuth ticket fails every attempt and would otherwise stack
|
||||
// identical error toasts (and their haptics). Reset on the next clean open.
|
||||
let reauthNotified = false
|
||||
// Raised once the reconnect loop crosses RECONNECT_ESCALATE_AFTER so the
|
||||
// recovery overlay replaces the dead-end CONNECTING screen. Reset on a clean
|
||||
// open or a manual/wake-driven reconnect.
|
||||
// Raised once the reconnect loop has been failing for
|
||||
// RECONNECT_ESCALATE_AFTER_MS so the recovery overlay replaces the
|
||||
// dead-end CONNECTING screen. Reset on a clean open or a manual/
|
||||
// wake-driven reconnect.
|
||||
let escalated = false
|
||||
|
||||
// Wrap the live getter in a call so TS control-flow analysis doesn't narrow
|
||||
@@ -173,6 +183,7 @@ export function useGatewayBoot({
|
||||
}
|
||||
|
||||
reconnectAttempt = 0
|
||||
reconnectFailingSince = null
|
||||
// A respawned backend re-mints (recycles) runtime ids, so any tile's
|
||||
// bound runtime id is now stale — drop them so each tile re-resumes.
|
||||
resetTileRuntimeBindings()
|
||||
@@ -192,7 +203,11 @@ export function useGatewayBoot({
|
||||
reconnecting = false
|
||||
|
||||
if (!cancelled && !gatewayOpen() && !$gatewaySwitching.get()) {
|
||||
if (reconnectAttempt >= RECONNECT_ESCALATE_AFTER && !escalated) {
|
||||
if (reconnectFailingSince === null) {
|
||||
reconnectFailingSince = Date.now()
|
||||
}
|
||||
|
||||
if (Date.now() - reconnectFailingSince >= RECONNECT_ESCALATE_AFTER_MS && !escalated) {
|
||||
escalated = true
|
||||
failDesktopBoot(translateNow('boot.errors.gatewayConnectionLost'))
|
||||
}
|
||||
@@ -207,8 +222,11 @@ export function useGatewayBoot({
|
||||
return
|
||||
}
|
||||
|
||||
// 1s, 2s, 4s … capped at 15s.
|
||||
const delay = Math.min(15_000, 1_000 * 2 ** Math.min(reconnectAttempt, 4))
|
||||
// Full-jitter exponential backoff (300ms base, 15s cap) so a gateway
|
||||
// restart doesn't get redialed by every desktop client in lockstep —
|
||||
// an immediate-retry reconnect storm can exhaust the gateway's file
|
||||
// descriptors while it's still coming back up.
|
||||
const delay = reconnectBackoffDelayMs(reconnectAttempt)
|
||||
reconnectAttempt += 1
|
||||
reconnectTimer = setTimeout(() => {
|
||||
reconnectTimer = null
|
||||
@@ -223,6 +241,7 @@ export function useGatewayBoot({
|
||||
|
||||
clearReconnectTimer()
|
||||
reconnectAttempt = 0
|
||||
reconnectFailingSince = null
|
||||
escalated = false
|
||||
reconnectSecondaryGateways()
|
||||
|
||||
@@ -269,6 +288,7 @@ export function useGatewayBoot({
|
||||
$gatewaySwitching.set(true)
|
||||
clearReconnectTimer()
|
||||
reconnectAttempt = 0
|
||||
reconnectFailingSince = null
|
||||
escalated = false
|
||||
reauthNotified = false
|
||||
callbacksRef.current.beforeConnectionSwitch()
|
||||
@@ -371,6 +391,7 @@ export function useGatewayBoot({
|
||||
|
||||
if (st === 'open') {
|
||||
reconnectAttempt = 0
|
||||
reconnectFailingSince = null
|
||||
reauthNotified = false
|
||||
escalated = false
|
||||
clearReconnectTimer()
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { projectTreeViewportSize } from './tree'
|
||||
|
||||
describe('projectTreeViewportSize', () => {
|
||||
it('uses ResizeObserver contentRect without forcing another layout read', () => {
|
||||
const element = document.createElement('div')
|
||||
const getBoundingClientRect = vi.spyOn(element, 'getBoundingClientRect')
|
||||
const contentRect = { height: 480, width: 320 } as DOMRectReadOnly
|
||||
|
||||
expect(
|
||||
projectTreeViewportSize([{ contentRect, target: element } as unknown as ResizeObserverEntry], element)
|
||||
).toEqual({
|
||||
height: 480,
|
||||
width: 320
|
||||
})
|
||||
expect(getBoundingClientRect).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('falls back to a rect when ResizeObserver is unavailable', () => {
|
||||
const element = document.createElement('div')
|
||||
vi.spyOn(element, 'getBoundingClientRect').mockReturnValue({
|
||||
height: 240,
|
||||
width: 160
|
||||
} as DOMRect)
|
||||
|
||||
expect(projectTreeViewportSize([], element)).toEqual({ height: 240, width: 160 })
|
||||
})
|
||||
})
|
||||
@@ -1,12 +1,14 @@
|
||||
import { useStore } from '@nanostores/react'
|
||||
import { type KeyboardEvent as ReactKeyboardEvent, useCallback, useEffect, useRef, useState } from 'react'
|
||||
import { useMemo } from 'react'
|
||||
import { type NodeApi, type NodeRendererProps, type RowRendererProps, Tree, type TreeApi } from 'react-arborist'
|
||||
|
||||
import { TreeSkeleton } from '@/components/chat/skeletons'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { markRightPanePerf } from '@/debug/right-pane-events'
|
||||
import { useResizeObserver } from '@/hooks/use-resize-observer'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { $repoChangeByPath, type RepoChangeKind } from '@/store/coding-status'
|
||||
import { type RepoChangeKind, repoChangeKindForPath } from '@/store/coding-status'
|
||||
import { $renamingPath, beginInlineRename } from '@/store/file-actions'
|
||||
import { $revealInTreeRequest } from '@/store/layout'
|
||||
|
||||
@@ -55,19 +57,20 @@ export function ProjectTree({
|
||||
onPreviewFile,
|
||||
openState
|
||||
}: ProjectTreeProps) {
|
||||
markRightPanePerf('project-tree-render')
|
||||
|
||||
const containerRef = useRef<HTMLDivElement | null>(null)
|
||||
const treeRef = useRef<TreeApi<TreeNode> | null>(null)
|
||||
const [size, setSize] = useState({ height: 0, width: 0 })
|
||||
const changeByPath = useStore($repoChangeByPath)
|
||||
|
||||
const syncTreeSize = useCallback(() => {
|
||||
const syncTreeSize = useCallback((entries: readonly ResizeObserverEntry[]) => {
|
||||
const el = containerRef.current
|
||||
|
||||
if (!el) {
|
||||
return
|
||||
}
|
||||
|
||||
const { height, width } = el.getBoundingClientRect()
|
||||
const { height, width } = projectTreeViewportSize(entries, el)
|
||||
|
||||
setSize(prev => {
|
||||
if (prev.height === height && prev.width === width) {
|
||||
@@ -175,7 +178,12 @@ export function ProjectTree({
|
||||
}, [])
|
||||
|
||||
return (
|
||||
<div className="min-h-0 flex-1 overflow-hidden" onKeyDownCapture={handleRenameShortcut} ref={containerRef}>
|
||||
<div
|
||||
className="min-h-0 flex-1 overflow-hidden"
|
||||
data-project-tree=""
|
||||
onKeyDownCapture={handleRenameShortcut}
|
||||
ref={containerRef}
|
||||
>
|
||||
{size.height > 0 && size.width > 0 ? (
|
||||
<Tree<TreeNode>
|
||||
childrenAccessor={node => (node?.isDirectory ? (node.children ?? []) : null)}
|
||||
@@ -200,7 +208,6 @@ export function ProjectTree({
|
||||
{props => (
|
||||
<ProjectTreeRow
|
||||
{...props}
|
||||
changeKind={props.node.data ? changeByPath.get(props.node.data.id) : undefined}
|
||||
onAttachFile={onActivateFile}
|
||||
onAttachFolder={onActivateFolder}
|
||||
onPreviewFile={onPreviewFile}
|
||||
@@ -215,6 +222,16 @@ export function ProjectTree({
|
||||
)
|
||||
}
|
||||
|
||||
export function projectTreeViewportSize(
|
||||
entries: readonly ResizeObserverEntry[],
|
||||
element: HTMLElement
|
||||
): { height: number; width: number } {
|
||||
const entry = entries.find(item => item.target === element)
|
||||
const box = entry?.contentRect ?? element.getBoundingClientRect()
|
||||
|
||||
return { height: box.height, width: box.width }
|
||||
}
|
||||
|
||||
function TreeSizingState() {
|
||||
return <TreeSkeleton />
|
||||
}
|
||||
@@ -244,7 +261,6 @@ const CHANGE_TINT: Record<RepoChangeKind, string> = {
|
||||
}
|
||||
|
||||
function ProjectTreeRow({
|
||||
changeKind,
|
||||
dragHandle,
|
||||
node,
|
||||
onAttachFile,
|
||||
@@ -253,13 +269,17 @@ function ProjectTreeRow({
|
||||
relativeTo,
|
||||
style
|
||||
}: NodeRendererProps<TreeNode> & {
|
||||
changeKind?: RepoChangeKind
|
||||
onAttachFile: (path: string) => void
|
||||
onAttachFolder: (path: string) => void
|
||||
onPreviewFile?: (path: string) => void
|
||||
relativeTo?: null | string
|
||||
}) {
|
||||
const renamingPath = useStore($renamingPath)
|
||||
const path = node.data?.id ?? ''
|
||||
const changeStore = useMemo(() => repoChangeKindForPath(path), [path])
|
||||
const changeKind: RepoChangeKind | undefined = useStore(changeStore)
|
||||
|
||||
markRightPanePerf('project-tree-row-render', path)
|
||||
|
||||
if (!node.data) {
|
||||
return <div style={style} />
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { observeActiveTerminalResize } from './active-resize'
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllGlobals()
|
||||
})
|
||||
|
||||
function installRaf() {
|
||||
let nextId = 1
|
||||
const frames = new Map<number, FrameRequestCallback>()
|
||||
|
||||
vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => {
|
||||
const id = nextId++
|
||||
frames.set(id, callback)
|
||||
|
||||
return id
|
||||
})
|
||||
vi.stubGlobal('cancelAnimationFrame', (id: number) => frames.delete(id))
|
||||
|
||||
return {
|
||||
flush() {
|
||||
const pending = [...frames.entries()]
|
||||
frames.clear()
|
||||
pending.forEach(([, callback]) => callback(0))
|
||||
},
|
||||
pending: () => frames.size
|
||||
}
|
||||
}
|
||||
|
||||
describe('observeActiveTerminalResize', () => {
|
||||
it('fits once on activation and coalesces later resize bursts', () => {
|
||||
const raf = installRaf()
|
||||
const resize = { current: null as ResizeObserverCallback | null }
|
||||
const disconnect = vi.fn()
|
||||
|
||||
vi.stubGlobal(
|
||||
'ResizeObserver',
|
||||
class {
|
||||
constructor(callback: ResizeObserverCallback) {
|
||||
resize.current = callback
|
||||
}
|
||||
|
||||
disconnect = disconnect
|
||||
observe = vi.fn((target: Element) => {
|
||||
resize.current?.([{ target } as ResizeObserverEntry], this as unknown as ResizeObserver)
|
||||
})
|
||||
unobserve = vi.fn()
|
||||
} as unknown as typeof ResizeObserver
|
||||
)
|
||||
|
||||
const onFit = vi.fn()
|
||||
const onActivate = vi.fn()
|
||||
const host = document.createElement('div')
|
||||
const dispose = observeActiveTerminalResize(host, { onActivate, onFit })
|
||||
|
||||
// ResizeObserver's initial delivery is absorbed by the activation frame.
|
||||
expect(raf.pending()).toBe(1)
|
||||
raf.flush()
|
||||
expect(onFit).toHaveBeenCalledTimes(1)
|
||||
expect(onActivate).toHaveBeenCalledTimes(1)
|
||||
|
||||
resize.current?.([], {} as ResizeObserver)
|
||||
resize.current?.([], {} as ResizeObserver)
|
||||
resize.current?.([], {} as ResizeObserver)
|
||||
expect(raf.pending()).toBe(1)
|
||||
raf.flush()
|
||||
expect(onFit).toHaveBeenCalledTimes(2)
|
||||
|
||||
dispose()
|
||||
expect(disconnect).toHaveBeenCalledTimes(1)
|
||||
})
|
||||
|
||||
it('cancels activation without fitting when hidden before the first frame', () => {
|
||||
const raf = installRaf()
|
||||
|
||||
vi.stubGlobal(
|
||||
'ResizeObserver',
|
||||
class {
|
||||
disconnect = vi.fn()
|
||||
observe = vi.fn()
|
||||
unobserve = vi.fn()
|
||||
} as unknown as typeof ResizeObserver
|
||||
)
|
||||
|
||||
const onFit = vi.fn()
|
||||
|
||||
const dispose = observeActiveTerminalResize(document.createElement('div'), {
|
||||
onActivate: vi.fn(),
|
||||
onFit
|
||||
})
|
||||
|
||||
dispose()
|
||||
|
||||
raf.flush()
|
||||
expect(onFit).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('absorbs a real browser-style initial resize delivered after activation', () => {
|
||||
const raf = installRaf()
|
||||
const resize = { current: null as ResizeObserverCallback | null }
|
||||
|
||||
vi.stubGlobal(
|
||||
'ResizeObserver',
|
||||
class {
|
||||
constructor(callback: ResizeObserverCallback) {
|
||||
resize.current = callback
|
||||
}
|
||||
|
||||
disconnect = vi.fn()
|
||||
observe = vi.fn()
|
||||
unobserve = vi.fn()
|
||||
} as unknown as typeof ResizeObserver
|
||||
)
|
||||
|
||||
const onFit = vi.fn()
|
||||
observeActiveTerminalResize(document.createElement('div'), { onActivate: vi.fn(), onFit })
|
||||
|
||||
raf.flush()
|
||||
expect(onFit).toHaveBeenCalledTimes(1)
|
||||
|
||||
// Browser initial delivery: the activation fit already covered this size.
|
||||
resize.current?.([], {} as ResizeObserver)
|
||||
expect(raf.pending()).toBe(0)
|
||||
|
||||
// A later real resize schedules exactly one fit.
|
||||
resize.current?.([], {} as ResizeObserver)
|
||||
expect(raf.pending()).toBe(1)
|
||||
raf.flush()
|
||||
expect(onFit).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
|
||||
it('reuses a first-mount fit without fitting again on activation', () => {
|
||||
const raf = installRaf()
|
||||
|
||||
vi.stubGlobal(
|
||||
'ResizeObserver',
|
||||
class {
|
||||
disconnect = vi.fn()
|
||||
observe = vi.fn()
|
||||
unobserve = vi.fn()
|
||||
} as unknown as typeof ResizeObserver
|
||||
)
|
||||
|
||||
const onActivate = vi.fn()
|
||||
const onFit = vi.fn()
|
||||
|
||||
observeActiveTerminalResize(document.createElement('div'), {
|
||||
fitOnActivate: false,
|
||||
onActivate,
|
||||
onFit
|
||||
})
|
||||
|
||||
raf.flush()
|
||||
|
||||
expect(onActivate).toHaveBeenCalledOnce()
|
||||
expect(onFit).not.toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,78 @@
|
||||
interface ActiveTerminalResizeOptions {
|
||||
fitOnActivate?: boolean
|
||||
onActivate: () => void
|
||||
onFit: () => void
|
||||
}
|
||||
|
||||
/**
|
||||
* Observe one visible xterm host.
|
||||
*
|
||||
* Inactive terminals never call this helper, so their preserved DOM/PTY stays
|
||||
* mounted without paying for ResizeObserver delivery or FitAddon work. The
|
||||
* first frame owns activation and ignores the observer's initial delivery;
|
||||
* later resize bursts are coalesced to one fit per animation frame.
|
||||
*/
|
||||
export function observeActiveTerminalResize(
|
||||
host: HTMLElement,
|
||||
{ fitOnActivate = true, onActivate, onFit }: ActiveTerminalResizeOptions
|
||||
): () => void {
|
||||
let activated = false
|
||||
let frame = 0
|
||||
let initialResizeDelivered = false
|
||||
let stopped = false
|
||||
|
||||
const scheduleFit = () => {
|
||||
if (!activated || stopped || frame !== 0) {
|
||||
return
|
||||
}
|
||||
|
||||
frame = window.requestAnimationFrame(() => {
|
||||
frame = 0
|
||||
|
||||
if (!stopped) {
|
||||
onFit()
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
const observer = new ResizeObserver(() => {
|
||||
// ResizeObserver's initial delivery is asynchronous in browsers and may
|
||||
// arrive before OR after the activation rAF. Activation already fits the
|
||||
// current box, so absorb that first delivery in either ordering.
|
||||
if (!initialResizeDelivered) {
|
||||
initialResizeDelivered = true
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
scheduleFit()
|
||||
})
|
||||
|
||||
observer.observe(host)
|
||||
|
||||
frame = window.requestAnimationFrame(() => {
|
||||
frame = 0
|
||||
|
||||
if (stopped) {
|
||||
return
|
||||
}
|
||||
|
||||
activated = true
|
||||
|
||||
if (fitOnActivate) {
|
||||
onFit()
|
||||
}
|
||||
|
||||
onActivate()
|
||||
})
|
||||
|
||||
return () => {
|
||||
stopped = true
|
||||
observer.disconnect()
|
||||
|
||||
if (frame !== 0) {
|
||||
window.cancelAnimationFrame(frame)
|
||||
frame = 0
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,8 @@ import { act, type ReactNode } from 'react'
|
||||
import { createRoot, type Root } from 'react-dom/client'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { $paneStates } from '@/store/panes'
|
||||
|
||||
import { PersistentTerminal, TerminalSlot } from './persistent'
|
||||
|
||||
vi.mock('./terminals', () => ({
|
||||
@@ -212,7 +214,8 @@ describe('PersistentTerminal rect tracking', () => {
|
||||
|
||||
render(<Harness />)
|
||||
|
||||
expect(mutationObserveCalls.some(call => call.options?.subtree === true)).toBe(true)
|
||||
expect(mutationObserveCalls.length).toBeGreaterThan(0)
|
||||
expect(mutationObserveCalls.every(call => call.options?.subtree === false)).toBe(true)
|
||||
|
||||
act(() => {
|
||||
raf.runNext()
|
||||
@@ -244,6 +247,27 @@ describe('PersistentTerminal rect tracking', () => {
|
||||
expect(raf.pending()).toBe(0)
|
||||
})
|
||||
|
||||
it('remeasures from an explicit pane-layout state change', () => {
|
||||
const raf = installRaf()
|
||||
const before = $paneStates.get()
|
||||
vi.spyOn(HTMLElement.prototype, 'getBoundingClientRect').mockReturnValue(rect(10, 20, 200, 100))
|
||||
|
||||
render(<Harness />)
|
||||
raf.runNext()
|
||||
expect(raf.pending()).toBe(0)
|
||||
|
||||
act(() => {
|
||||
$paneStates.set({ ...before, __terminal_rect_test__: { open: true } })
|
||||
})
|
||||
|
||||
expect(raf.pending()).toBe(1)
|
||||
|
||||
act(() => {
|
||||
raf.runNext()
|
||||
$paneStates.set(before)
|
||||
})
|
||||
})
|
||||
|
||||
it('does not schedule rect RAFs while the Electron window is paused, then resumes when visible', () => {
|
||||
const raf = installRaf()
|
||||
vi.spyOn(HTMLElement.prototype, 'getBoundingClientRect').mockReturnValue(rect(10, 20, 200, 100))
|
||||
|
||||
@@ -2,7 +2,10 @@ import { useStore } from '@nanostores/react'
|
||||
import { atom } from 'nanostores'
|
||||
import { type CSSProperties, useEffect, useLayoutEffect, useRef, useState } from 'react'
|
||||
|
||||
import { $layoutTree } from '@/components/pane-shell/tree/store'
|
||||
import { markRightPanePerf } from '@/debug/right-pane-events'
|
||||
import { createRendererLoopPauseController } from '@/lib/renderer-loop-pause'
|
||||
import { $paneStates } from '@/store/panes'
|
||||
|
||||
import { $terminalTakeover } from '../store'
|
||||
|
||||
@@ -39,7 +42,7 @@ export function TerminalSlot({ className = SLOT_CLASS }: { className?: string })
|
||||
}
|
||||
}, [])
|
||||
|
||||
return <div className={className} ref={ref} />
|
||||
return <div className={className} data-terminal-slot="" ref={ref} />
|
||||
}
|
||||
|
||||
interface PersistentTerminalProps {
|
||||
@@ -86,6 +89,7 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP
|
||||
let prev: Rect | null = null
|
||||
let frame = 0
|
||||
let stopped = false
|
||||
let pendingReason = 'initial'
|
||||
let pauseController: ReturnType<typeof createRendererLoopPauseController> | null = null
|
||||
|
||||
const rendererPaused = () => pauseController?.isPaused() ?? document.visibilityState === 'hidden'
|
||||
@@ -97,11 +101,12 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP
|
||||
}
|
||||
}
|
||||
|
||||
const measure = (): boolean => {
|
||||
const measure = (reason: string): boolean => {
|
||||
if (rendererPaused()) {
|
||||
return false
|
||||
}
|
||||
|
||||
markRightPanePerf('terminal-measure', reason)
|
||||
const r = slot.getBoundingClientRect()
|
||||
// floor top/left + ceil right/bottom: overlay always covers the slot's
|
||||
// full pixel footprint, so half-pixel rects can't leak page bg through.
|
||||
@@ -123,16 +128,18 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP
|
||||
return false
|
||||
}
|
||||
|
||||
const scheduleMeasure = () => {
|
||||
const scheduleMeasure = (reason = 'unknown') => {
|
||||
if (stopped || rendererPaused() || frame !== 0) {
|
||||
return
|
||||
}
|
||||
|
||||
pendingReason = reason
|
||||
frame = window.requestAnimationFrame(() => {
|
||||
frame = 0
|
||||
const reason = pendingReason
|
||||
|
||||
if (measure()) {
|
||||
scheduleMeasure()
|
||||
if (measure(reason)) {
|
||||
scheduleMeasure('settle')
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -144,50 +151,69 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP
|
||||
return
|
||||
}
|
||||
|
||||
scheduleMeasure()
|
||||
scheduleMeasure('visibility')
|
||||
}
|
||||
|
||||
const observer =
|
||||
typeof ResizeObserver === 'undefined'
|
||||
? null
|
||||
: new ResizeObserver(() => {
|
||||
scheduleMeasure()
|
||||
scheduleMeasure('resize-observer')
|
||||
})
|
||||
|
||||
const positionObserver =
|
||||
typeof MutationObserver === 'undefined'
|
||||
? null
|
||||
: new MutationObserver(() => {
|
||||
scheduleMeasure()
|
||||
scheduleMeasure('ancestor-mutation')
|
||||
})
|
||||
|
||||
pauseController = createRendererLoopPauseController(handleVisibilityChange)
|
||||
|
||||
if (measure()) {
|
||||
scheduleMeasure()
|
||||
if (measure('initial')) {
|
||||
scheduleMeasure('settle')
|
||||
}
|
||||
|
||||
observer?.observe(slot)
|
||||
|
||||
const handleScroll = () => scheduleMeasure('scroll')
|
||||
const scrollTargets: Array<HTMLElement | Window> = [window]
|
||||
window.addEventListener('scroll', handleScroll)
|
||||
|
||||
for (let node: HTMLElement | null = slot; node; node = node.parentElement) {
|
||||
positionObserver?.observe(node, {
|
||||
attributeFilter: ['class', 'style', 'hidden', 'aria-hidden', 'data-state'],
|
||||
attributes: true,
|
||||
childList: true,
|
||||
subtree: true
|
||||
subtree: false
|
||||
})
|
||||
// Scroll does not bubble. Listen only on the slot's own ancestor chain,
|
||||
// so a transcript/file-tree/xterm viewport scroll elsewhere cannot wake
|
||||
// terminal positioning.
|
||||
node.addEventListener('scroll', handleScroll)
|
||||
scrollTargets.push(node)
|
||||
}
|
||||
|
||||
window.addEventListener('resize', scheduleMeasure)
|
||||
window.addEventListener('scroll', scheduleMeasure, true)
|
||||
// Nested layout-tree and pane-state commits can move the slot without
|
||||
// changing its own size. Subscribe to the actual layout authorities instead
|
||||
// of observing every descendant mutation under every ancestor (chat stream
|
||||
// and file-tree updates are unrelated and used to wake this tracker).
|
||||
const unsubscribeLayout = $layoutTree.listen(() => scheduleMeasure('layout-tree'))
|
||||
const unsubscribePanes = $paneStates.listen(() => scheduleMeasure('pane-state'))
|
||||
|
||||
const handleResize = () => scheduleMeasure('window-resize')
|
||||
|
||||
window.addEventListener('resize', handleResize)
|
||||
|
||||
return () => {
|
||||
stopped = true
|
||||
cancelFrame()
|
||||
observer?.disconnect()
|
||||
positionObserver?.disconnect()
|
||||
window.removeEventListener('resize', scheduleMeasure)
|
||||
window.removeEventListener('scroll', scheduleMeasure, true)
|
||||
unsubscribeLayout()
|
||||
unsubscribePanes()
|
||||
window.removeEventListener('resize', handleResize)
|
||||
scrollTargets.forEach(target => target.removeEventListener('scroll', handleScroll))
|
||||
pauseController?.dispose()
|
||||
}
|
||||
}, [slot])
|
||||
@@ -215,7 +241,7 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP
|
||||
// booting xterm/node-pty at 0×0 starts the shell at 80×24 and spawns a visible
|
||||
// conhost on Windows. After that `mounted` latches: shells persist while hidden.
|
||||
return (
|
||||
<div aria-hidden={!visible} style={style}>
|
||||
<div aria-hidden={!visible} data-persistent-terminal="" style={style}>
|
||||
{mounted && <TerminalWorkspace onAddSelectionToChat={onAddSelectionToChat} />}
|
||||
</div>
|
||||
)
|
||||
|
||||
@@ -5,9 +5,11 @@ import { Terminal } from '@xterm/xterm'
|
||||
import { useEffect, useRef } from 'react'
|
||||
|
||||
import { writeClipboardText } from '@/components/ui/copy-button'
|
||||
import { markRightPanePerf } from '@/debug/right-pane-events'
|
||||
import { triggerHaptic } from '@/lib/haptics'
|
||||
import { useTheme } from '@/themes/context'
|
||||
|
||||
import { observeActiveTerminalResize } from './active-resize'
|
||||
import { registerAgentTerminalWriter } from './agent-terminal-stream'
|
||||
import { makeTerminalReader, registerTerminalReader } from './buffer'
|
||||
import { mirrorSelection, terminalClipboardIntent } from './clipboard'
|
||||
@@ -24,7 +26,8 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id:
|
||||
const hostRef = useRef<HTMLDivElement | null>(null)
|
||||
const termRef = useRef<Terminal | null>(null)
|
||||
const webglRef = useRef<WebglAddon | null>(null)
|
||||
const fitRef = useRef<(() => void) | null>(null)
|
||||
const fitRef = useRef<((visible: boolean) => void) | null>(null)
|
||||
const initialActiveFitRef = useRef(false)
|
||||
const { latestFontFamilyRef, mountedRef } = useTerminalFontController({ fitRef, termRef, webglRef })
|
||||
|
||||
const surfaceTheme = () => {
|
||||
@@ -47,7 +50,6 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id:
|
||||
}
|
||||
|
||||
let disposed = false
|
||||
let observer: ResizeObserver | null = null
|
||||
|
||||
let unregister = () => {}
|
||||
|
||||
@@ -101,10 +103,11 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id:
|
||||
return false
|
||||
})
|
||||
|
||||
fitRef.current = () => {
|
||||
fitRef.current = visible => {
|
||||
if (host.clientWidth > 0 && host.clientHeight > 0) {
|
||||
try {
|
||||
fit.fit()
|
||||
markRightPanePerf(visible ? 'terminal-fit-active' : 'terminal-fit-hidden', id)
|
||||
} catch {
|
||||
// Mid-transition layout — the next observer tick refits.
|
||||
}
|
||||
@@ -132,9 +135,8 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id:
|
||||
// No WebGL — xterm falls back to the DOM renderer.
|
||||
}
|
||||
|
||||
fitRef.current?.()
|
||||
observer = new ResizeObserver(() => fitRef.current?.())
|
||||
observer.observe(host)
|
||||
fitRef.current?.(active)
|
||||
initialActiveFitRef.current = active
|
||||
|
||||
// Stream live output straight into the terminal (replays backlog on attach).
|
||||
unregister = registerAgentTerminalWriter(procId, chunk => term.write(chunk))
|
||||
@@ -159,7 +161,6 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id:
|
||||
unregister()
|
||||
unregisterReader()
|
||||
selectionDisposable.dispose()
|
||||
observer?.disconnect()
|
||||
fitRef.current = null
|
||||
term.dispose()
|
||||
termRef.current = null
|
||||
@@ -184,25 +185,39 @@ export function useAgentTerminal({ active, id, procId }: { active: boolean; id:
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [renderedMode, themeName])
|
||||
|
||||
// A visibility:hidden xterm doesn't paint — refit + redraw on re-activation.
|
||||
// Keep inactive agent terminals mounted for their backlog, but do not observe
|
||||
// or fit them until they become the visible tab.
|
||||
// eslint-disable-next-line no-restricted-syntax -- lifecycle flag prevents a duplicate first-mount fit
|
||||
useEffect(() => {
|
||||
if (!active) {
|
||||
initialActiveFitRef.current = false
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
const frame = requestAnimationFrame(() => {
|
||||
const term = termRef.current
|
||||
const host = hostRef.current
|
||||
|
||||
fitRef.current?.()
|
||||
webglRef.current?.clearTextureAtlas()
|
||||
term?.refresh(0, term.rows - 1)
|
||||
// Take focus on activation (parity with the user terminal) so the active
|
||||
// agent tab holds focus and ⌘W's isFocusWithin('[data-terminal]') routes
|
||||
// the close to this tab rather than to a preview.
|
||||
term?.focus()
|
||||
if (!host) {
|
||||
return
|
||||
}
|
||||
|
||||
const fitOnActivate = !initialActiveFitRef.current
|
||||
initialActiveFitRef.current = false
|
||||
|
||||
return observeActiveTerminalResize(host, {
|
||||
fitOnActivate,
|
||||
onFit: () => fitRef.current?.(true),
|
||||
onActivate: () => {
|
||||
const term = termRef.current
|
||||
|
||||
webglRef.current?.clearTextureAtlas()
|
||||
term?.refresh(0, term.rows - 1)
|
||||
// Take focus on activation (parity with the user terminal) so the active
|
||||
// agent tab holds focus and ⌘W's isFocusWithin('[data-terminal]') routes
|
||||
// the close to this tab rather than to a preview.
|
||||
term?.focus()
|
||||
}
|
||||
})
|
||||
|
||||
return () => cancelAnimationFrame(frame)
|
||||
}, [active])
|
||||
|
||||
return { hostRef }
|
||||
|
||||
@@ -7,7 +7,7 @@ import type { RefObject } from 'react'
|
||||
import { $terminalFontFamily, applyTerminalFontFamily, resolveTerminalFontFamily } from './terminal-font'
|
||||
|
||||
interface TerminalFontControllerOptions {
|
||||
fitRef: RefObject<(() => void) | null>
|
||||
fitRef: RefObject<((visible: boolean) => void) | null>
|
||||
termRef: RefObject<Terminal | null>
|
||||
webglRef: RefObject<WebglAddon | null>
|
||||
}
|
||||
@@ -38,7 +38,7 @@ export function useTerminalFontController({ fitRef, termRef, webglRef }: Termina
|
||||
|
||||
void applyTerminalFontFamily({
|
||||
clearTextureAtlas: () => webglRef.current?.clearTextureAtlas(),
|
||||
fit: () => fitRef.current?.(),
|
||||
fit: () => fitRef.current?.(true),
|
||||
fontFamily,
|
||||
isCurrent: () => !cancelled && generationRef.current === generation,
|
||||
term
|
||||
|
||||
@@ -7,12 +7,14 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||
import type { CSSProperties } from 'react'
|
||||
|
||||
import { writeClipboardText } from '@/components/ui/copy-button'
|
||||
import { markRightPanePerf } from '@/debug/right-pane-events'
|
||||
import { triggerHaptic } from '@/lib/haptics'
|
||||
import { $previewTarget } from '@/store/preview'
|
||||
import { useTheme } from '@/themes/context'
|
||||
|
||||
import { $terminalInjection } from '../store'
|
||||
|
||||
import { observeActiveTerminalResize } from './active-resize'
|
||||
import { makeTerminalReader, registerTerminalReader } from './buffer'
|
||||
import { mirrorSelection, terminalClipboardIntent } from './clipboard'
|
||||
import { terminalLinkHandler, terminalWebLinksAddon } from './links'
|
||||
@@ -413,6 +415,7 @@ export function useTerminalSession({
|
||||
// drag-and-drop paths, or an injected command). Gates idle-buffer handling in
|
||||
// persistSnapshot so an untouched tab never re-saves an accumulating snapshot.
|
||||
const hasSessionActivityRef = useRef(false)
|
||||
const initialActiveRef = useRef(active)
|
||||
const shellNameRef = useRef('shell')
|
||||
const selectionLabelRef = useRef('')
|
||||
const selectionRef = useRef('')
|
||||
@@ -420,7 +423,8 @@ export function useTerminalSession({
|
||||
const onShellRef = useRef(onShell)
|
||||
// Re-fit on activation: a tab hidden via display:none has a 0×0 host, so its
|
||||
// last fit is stale by the time it's shown again.
|
||||
const fitRef = useRef<(() => void) | null>(null)
|
||||
const fitRef = useRef<((visible: boolean) => void) | null>(null)
|
||||
const initialActiveFitRef = useRef(false)
|
||||
const { latestFontFamilyRef, mountedRef } = useTerminalFontController({ fitRef, termRef, webglRef })
|
||||
const [status, setStatus] = useState<TerminalStatus>('starting')
|
||||
const [selection, setSelection] = useState('')
|
||||
@@ -742,56 +746,28 @@ export function useTerminalSession({
|
||||
term.write(next)
|
||||
}
|
||||
|
||||
const fitAndResize = () => {
|
||||
const fitAndResize = (visible: boolean) => {
|
||||
if (disposed || !host.isConnected || host.clientWidth <= 0 || host.clientHeight <= 0) {
|
||||
return
|
||||
}
|
||||
|
||||
try {
|
||||
fit.fit()
|
||||
markRightPanePerf(visible ? 'terminal-fit-active' : 'terminal-fit-hidden', id)
|
||||
} catch {
|
||||
return
|
||||
}
|
||||
|
||||
const id = sessionIdRef.current
|
||||
const sessionId = sessionIdRef.current
|
||||
|
||||
if (id && (lastSentSize?.cols !== term.cols || lastSentSize?.rows !== term.rows)) {
|
||||
if (sessionId && (lastSentSize?.cols !== term.cols || lastSentSize?.rows !== term.rows)) {
|
||||
lastSentSize = { cols: term.cols, rows: term.rows }
|
||||
void terminalApi.resize(id, { cols: term.cols, rows: term.rows })
|
||||
void terminalApi.resize(sessionId, { cols: term.cols, rows: term.rows })
|
||||
}
|
||||
}
|
||||
|
||||
fitRef.current = fitAndResize
|
||||
|
||||
// Coalesce ResizeObserver bursts through rAF — running fit.fit()
|
||||
// synchronously while sibling panes are mid-transition (e.g. file browser
|
||||
// collapsing to 0px) crashes the WebGL renderer mid texture-atlas rebuild.
|
||||
let pendingFrame = 0
|
||||
|
||||
const scheduleResize = () => {
|
||||
if (pendingFrame) {
|
||||
return
|
||||
}
|
||||
|
||||
pendingFrame = window.requestAnimationFrame(() => {
|
||||
pendingFrame = 0
|
||||
|
||||
if (!disposed) {
|
||||
fitAndResize()
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
const resizeObserver = new ResizeObserver(scheduleResize)
|
||||
resizeObserver.observe(host)
|
||||
cleanup.push(() => {
|
||||
resizeObserver.disconnect()
|
||||
|
||||
if (pendingFrame) {
|
||||
window.cancelAnimationFrame(pendingFrame)
|
||||
}
|
||||
})
|
||||
|
||||
const dataDisposable = term.onData(data => {
|
||||
hasSessionActivityRef.current = true
|
||||
const id = sessionIdRef.current
|
||||
@@ -901,9 +877,7 @@ export function useTerminalSession({
|
||||
)
|
||||
|
||||
window.requestAnimationFrame(() => {
|
||||
fitAndResize()
|
||||
term.clearSelection() // drop any selection painted over transient boot rows
|
||||
term.focus()
|
||||
})
|
||||
})
|
||||
.catch(error => {
|
||||
@@ -938,7 +912,8 @@ export function useTerminalSession({
|
||||
console.warn('[hermes-terminal] WebGL unavailable; falling back to DOM', err)
|
||||
}
|
||||
|
||||
fitAndResize()
|
||||
fitAndResize(initialActiveRef.current)
|
||||
initialActiveFitRef.current = initialActiveRef.current
|
||||
startSession()
|
||||
}
|
||||
|
||||
@@ -1013,24 +988,39 @@ export function useTerminalSession({
|
||||
return term ? registerTerminalReader(id, makeTerminalReader(term)) : undefined
|
||||
}, [id, status])
|
||||
|
||||
// On (re)activation: a WebGL terminal doesn't paint while visibility:hidden, so
|
||||
// it reveals a stale/garbled frame. Refit, rebuild the glyph atlas, and force a
|
||||
// full redraw against the live buffer, then focus.
|
||||
// Only the active terminal observes its host. Every terminal stays mounted
|
||||
// (PTY + scrollback preserved), but hidden tabs do no FitAddon/layout work.
|
||||
// Re-activation owns one fit + atlas rebuild + redraw.
|
||||
// eslint-disable-next-line no-restricted-syntax -- lifecycle flag prevents a duplicate first-mount fit
|
||||
useEffect(() => {
|
||||
if (!active || status !== 'open') {
|
||||
if (!active) {
|
||||
initialActiveFitRef.current = false
|
||||
}
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
const frame = requestAnimationFrame(() => {
|
||||
const term = termRef.current
|
||||
const host = hostRef.current
|
||||
|
||||
fitRef.current?.()
|
||||
webglRef.current?.clearTextureAtlas()
|
||||
term?.refresh(0, term.rows - 1)
|
||||
term?.focus()
|
||||
if (!host) {
|
||||
return
|
||||
}
|
||||
|
||||
const fitOnActivate = !initialActiveFitRef.current
|
||||
initialActiveFitRef.current = false
|
||||
|
||||
return observeActiveTerminalResize(host, {
|
||||
fitOnActivate,
|
||||
onFit: () => fitRef.current?.(true),
|
||||
onActivate: () => {
|
||||
const term = termRef.current
|
||||
|
||||
webglRef.current?.clearTextureAtlas()
|
||||
term?.refresh(0, term.rows - 1)
|
||||
term?.focus()
|
||||
}
|
||||
})
|
||||
|
||||
return () => cancelAnimationFrame(frame)
|
||||
}, [active, status])
|
||||
|
||||
// Flush a queued command (e.g. a provider-disconnect) into the live session.
|
||||
|
||||
@@ -1,11 +1,14 @@
|
||||
import { QueryClient } from '@tanstack/react-query'
|
||||
import { act, cleanup, render } from '@testing-library/react'
|
||||
import { useEffect, useRef } from 'react'
|
||||
import { type MutableRefObject, useEffect, useRef } from 'react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import type { ClientSessionState } from '@/app/types'
|
||||
import type { ChatMessage } from '@/lib/chat-messages'
|
||||
import { createClientSessionState } from '@/lib/chat-runtime'
|
||||
|
||||
import { useSessionStateCache } from '../use-session-state-cache'
|
||||
|
||||
import { useMessageStream } from './index'
|
||||
|
||||
const SID = 'session-1'
|
||||
@@ -90,6 +93,42 @@ describe('useMessageStream delta flush scheduling', () => {
|
||||
expect(assistantText()).toBe('still streaming')
|
||||
})
|
||||
|
||||
it('flushes queued text immediately when a hidden window becomes visible', () => {
|
||||
vi.mocked(performance.now).mockReturnValue(0)
|
||||
mountStream()
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'caught up on focus'))
|
||||
expect(assistantText()).toBe('')
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
Object.defineProperty(globalThis.document, 'visibilityState', {
|
||||
configurable: true,
|
||||
value: 'visible'
|
||||
})
|
||||
|
||||
act(() => globalThis.document.dispatchEvent(new Event('visibilitychange')))
|
||||
|
||||
expect(assistantText()).toBe('caught up on focus')
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
})
|
||||
|
||||
it('flushes queued text on focus when visibility remains visible', () => {
|
||||
vi.mocked(performance.now).mockReturnValue(0)
|
||||
Object.defineProperty(globalThis.document, 'visibilityState', {
|
||||
configurable: true,
|
||||
value: 'visible'
|
||||
})
|
||||
mountStream()
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'focused without visibility change'))
|
||||
expect(assistantText()).toBe('')
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => globalThis.window.dispatchEvent(new Event('focus')))
|
||||
|
||||
expect(assistantText()).toBe('focused without visibility change')
|
||||
})
|
||||
|
||||
it('cancels the pending timer on unmount and flushes exactly once', async () => {
|
||||
vi.mocked(performance.now).mockReturnValue(0)
|
||||
mountStream()
|
||||
@@ -108,4 +147,275 @@ describe('useMessageStream delta flush scheduling', () => {
|
||||
expect(updateSessionState).toHaveBeenCalledTimes(updatesAfterUnmount)
|
||||
expect(window.requestAnimationFrame).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('stretches the flush gap when the deferred commit frame is expensive', async () => {
|
||||
// The streaming-path $messages publish (React commit + Streamdown
|
||||
// re-parse) is deferred to a view-sync rAF inside updateSessionState, so
|
||||
// the flush cost must be measured through that frame. Simulate one
|
||||
// expensive frame and expect the next gap to adapt to 3x the frame cost.
|
||||
let now = 1000
|
||||
vi.mocked(performance.now).mockImplementation(() => now)
|
||||
const rafCallbacks: FrameRequestCallback[] = []
|
||||
vi.mocked(window.requestAnimationFrame).mockImplementation(cb => {
|
||||
rafCallbacks.push(cb)
|
||||
|
||||
return rafCallbacks.length
|
||||
})
|
||||
|
||||
mountStream()
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'first'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('first')
|
||||
expect(rafCallbacks).toHaveLength(1)
|
||||
|
||||
// Frame started at 1040, the measurement callback runs at 1100: 60ms of
|
||||
// in-frame work (view sync + commit), so the next floor is 180ms.
|
||||
now = 1100
|
||||
act(() => rafCallbacks[0](1040))
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'second'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(79)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('first')
|
||||
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(1)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('firstsecond')
|
||||
})
|
||||
|
||||
it('keeps the write-cost floor when no frame fires (hidden renderer)', async () => {
|
||||
// A parked renderer never runs rAF callbacks. The cost must stay at the
|
||||
// synchronous store-write measurement so the gap falls back to the fixed
|
||||
// 33ms floor instead of waiting on a frame that will never come.
|
||||
let now = 1000
|
||||
vi.mocked(performance.now).mockImplementation(() => now)
|
||||
vi.mocked(window.requestAnimationFrame).mockImplementation(() => 1)
|
||||
|
||||
mountStream()
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'first'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('first')
|
||||
|
||||
// 100ms later (well past the 33ms floor): the next flush is immediate.
|
||||
now = 1100
|
||||
act(() => appendAssistantDelta!(SID, 'second'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('firstsecond')
|
||||
})
|
||||
|
||||
it('ignores a late frame measurement once a newer flush has started', async () => {
|
||||
let now = 1000
|
||||
vi.mocked(performance.now).mockImplementation(() => now)
|
||||
const rafCallbacks: FrameRequestCallback[] = []
|
||||
vi.mocked(window.requestAnimationFrame).mockImplementation(cb => {
|
||||
rafCallbacks.push(cb)
|
||||
|
||||
return rafCallbacks.length
|
||||
})
|
||||
|
||||
mountStream()
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'a'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
// A second flush starts before the first flush's frame lands.
|
||||
now = 1010
|
||||
act(() => appendAssistantDelta!(SID, 'b'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(23)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('ab')
|
||||
expect(rafCallbacks).toHaveLength(2)
|
||||
|
||||
// The stale callback must not overwrite the newer flush's cost. If it
|
||||
// did, cost would read 30ms and the next gap would stretch to 70ms.
|
||||
now = 1030
|
||||
act(() => rafCallbacks[0](1000))
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'c'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(13)
|
||||
})
|
||||
|
||||
expect(assistantText()).toBe('abc')
|
||||
})
|
||||
})
|
||||
|
||||
describe('useMessageStream composed with the real useSessionStateCache', () => {
|
||||
// The tests above mock updateSessionState, so they validate the adaptive
|
||||
// arithmetic but not the production ordering contract: runFlush's
|
||||
// measurement rAF must be registered AFTER the view-sync rAF that the real
|
||||
// updateSessionState schedules inside syncSessionStateToView, so the
|
||||
// measured frame cost includes the deferred $messages commit it adapts to.
|
||||
let cache: ReturnType<typeof useSessionStateCache> | null = null
|
||||
let published: ChatMessage[]
|
||||
|
||||
function ComposedHarness() {
|
||||
const busyRef: MutableRefObject<boolean> = { current: false }
|
||||
const queryClientRef = useRef(new QueryClient())
|
||||
|
||||
const sessionCache = useSessionStateCache({
|
||||
activeSessionId: SID,
|
||||
busyRef,
|
||||
selectedStoredSessionId: null,
|
||||
setAwaitingResponse: () => undefined,
|
||||
setBusy: () => undefined,
|
||||
setMessages: messages => {
|
||||
published = messages
|
||||
}
|
||||
})
|
||||
|
||||
const stream = useMessageStream({
|
||||
activeSessionIdRef: sessionCache.activeSessionIdRef,
|
||||
hydrateFromStoredSession: vi.fn(async () => undefined),
|
||||
queryClient: queryClientRef.current,
|
||||
refreshHermesConfig: vi.fn(async () => undefined),
|
||||
refreshSessions: vi.fn(async () => undefined),
|
||||
sessionStateByRuntimeIdRef: sessionCache.sessionStateByRuntimeIdRef,
|
||||
updateSessionState: sessionCache.updateSessionState
|
||||
})
|
||||
|
||||
useEffect(() => {
|
||||
appendAssistantDelta = stream.appendAssistantDelta
|
||||
cache = sessionCache
|
||||
}, [stream.appendAssistantDelta, sessionCache])
|
||||
|
||||
return null
|
||||
}
|
||||
|
||||
function cachedText() {
|
||||
const message = cache?.sessionStateByRuntimeIdRef.current.get(SID)?.messages.at(-1)
|
||||
const part = message?.parts.at(-1)
|
||||
|
||||
return part?.type === 'text' ? part.text : ''
|
||||
}
|
||||
|
||||
function publishedText() {
|
||||
const part = published.at(-1)?.parts.at(-1)
|
||||
|
||||
return part?.type === 'text' ? part.text : ''
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers()
|
||||
appendAssistantDelta = null
|
||||
cache = null
|
||||
published = []
|
||||
vi.spyOn(performance, 'now').mockReturnValue(100)
|
||||
vi.spyOn(window, 'requestAnimationFrame').mockImplementation(() => 1)
|
||||
vi.spyOn(window, 'cancelAnimationFrame').mockImplementation(() => undefined)
|
||||
vi.spyOn(document, 'hasFocus').mockReturnValue(false)
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
vi.useRealTimers()
|
||||
vi.restoreAllMocks()
|
||||
})
|
||||
|
||||
it('measures the frame cost through the real view-sync rAF and adapts the next gap', async () => {
|
||||
let now = 1000
|
||||
vi.mocked(performance.now).mockImplementation(() => now)
|
||||
const rafCallbacks: FrameRequestCallback[] = []
|
||||
vi.mocked(window.requestAnimationFrame).mockImplementation(cb => {
|
||||
rafCallbacks.push(cb)
|
||||
|
||||
return rafCallbacks.length
|
||||
})
|
||||
|
||||
render(<ComposedHarness />)
|
||||
expect(appendAssistantDelta).not.toBeNull()
|
||||
|
||||
// Mid-turn state: busy keeps the view sync on the deferred rAF path
|
||||
// (terminal/needing-input states flush synchronously instead).
|
||||
act(() => {
|
||||
cache!.updateSessionState(SID, state => ({ ...state, busy: true }))
|
||||
})
|
||||
expect(rafCallbacks).toHaveLength(1)
|
||||
// Drain the seed's own view-sync rAF so the flush below starts clean.
|
||||
act(() => rafCallbacks.shift()!(now))
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'first'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
// The store write landed synchronously, but the $messages publish is
|
||||
// deferred: exactly two rAF callbacks are pending — first the cache's
|
||||
// view-sync, then runFlush's measurement.
|
||||
expect(cachedText()).toBe('first')
|
||||
expect(publishedText()).toBe('')
|
||||
expect(rafCallbacks).toHaveLength(2)
|
||||
|
||||
// Draining the FIRST registered callback must be what publishes the
|
||||
// deferred commit; that identity is the ordering contract. It runs until
|
||||
// 60ms into the frame (React commit + Streamdown re-parse).
|
||||
now = 1100
|
||||
act(() => rafCallbacks[0](1040))
|
||||
expect(publishedText()).toBe('first')
|
||||
|
||||
// The measurement callback closes the same frame: 60ms of in-frame work,
|
||||
// so the next adaptive floor is 3x = 180ms.
|
||||
act(() => rafCallbacks[1](1040))
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'second'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(79)
|
||||
})
|
||||
|
||||
expect(cachedText()).toBe('first')
|
||||
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(1)
|
||||
})
|
||||
|
||||
expect(cachedText()).toBe('firstsecond')
|
||||
})
|
||||
|
||||
it('keeps the write-cost fallback when the parked renderer never fires rAF', async () => {
|
||||
let now = 1000
|
||||
vi.mocked(performance.now).mockImplementation(() => now)
|
||||
// Parked renderer: rAF callbacks are accepted but never run.
|
||||
vi.mocked(window.requestAnimationFrame).mockImplementation(() => 1)
|
||||
|
||||
render(<ComposedHarness />)
|
||||
|
||||
act(() => {
|
||||
cache!.updateSessionState(SID, state => ({ ...state, busy: true }))
|
||||
})
|
||||
|
||||
act(() => appendAssistantDelta!(SID, 'first'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
expect(cachedText()).toBe('first')
|
||||
|
||||
// 100ms later (well past the 33ms floor): the next flush is immediate.
|
||||
now = 1100
|
||||
act(() => appendAssistantDelta!(SID, 'second'))
|
||||
await act(async () => {
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
})
|
||||
|
||||
expect(cachedText()).toBe('firstsecond')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -188,6 +188,9 @@ export function useMessageStream({
|
||||
// What the previous flush cost on the main thread — drives the adaptive
|
||||
// flush floor in scheduleDeltaFlush so multi-stream load yields to input.
|
||||
const lastFlushCostRef = useRef<number>(0)
|
||||
// The pending commit-cost measurement rAF, so a newer flush (or unmount)
|
||||
// can cancel it instead of letting parked callbacks pile up while hidden.
|
||||
const measureRafRef = useRef<number | null>(null)
|
||||
const nativeSubagentSessionsRef = useRef<Set<string>>(new Set())
|
||||
// Turns that auto-compacted: skip post-turn hydrate so live scrollback survives.
|
||||
const compactedTurnRef = useRef<Set<string>>(new Set())
|
||||
@@ -257,6 +260,8 @@ export function useMessageStream({
|
||||
// keeps the thread ~75% idle for input at any load: cheap flushes stay at
|
||||
// 30fps of text growth, expensive multi-stream flushes degrade text fps
|
||||
// instead of interactivity — capped so text never updates slower than 4/s.
|
||||
// The cost has to include the deferred view-sync frame where the commit
|
||||
// actually happens; see runFlush below.
|
||||
const sinceLast = performance.now() - lastFlushAtRef.current
|
||||
|
||||
const adaptiveFloor = Math.min(
|
||||
@@ -269,7 +274,39 @@ export function useMessageStream({
|
||||
const startedAt = performance.now()
|
||||
lastFlushAtRef.current = startedAt
|
||||
flushQueuedDeltas()
|
||||
lastFlushCostRef.current = performance.now() - startedAt
|
||||
// The store write above is only the cheap half of a flush. While a
|
||||
// session streams, syncSessionStateToView defers the $messages publish
|
||||
// (and with it the React commit + Streamdown re-parse the floor is meant
|
||||
// to account for) to its own rAF inside updateSessionState, which runs
|
||||
// after this timer task. Stopping the clock here pins lastFlushCostRef
|
||||
// near zero and collapses the adaptive floor to 33ms no matter the load.
|
||||
// Our rAF is registered after the view-sync one, so it runs in the same
|
||||
// frame right after that commit; its timestamp marks frame start, so
|
||||
// (now - frameStart) counts only work done inside the frame, not the
|
||||
// vsync wait. A hidden renderer never fires rAF, so the write cost
|
||||
// stays as the fallback.
|
||||
const writeCost = performance.now() - startedAt
|
||||
lastFlushCostRef.current = writeCost
|
||||
|
||||
// At most one measurement rAF may be pending: only the newest flush's
|
||||
// measurement matters (the guard below discards stale frames), and a
|
||||
// hidden renderer parks rAF callbacks — without cancellation a long
|
||||
// hidden stream at the floor would accumulate thousands of parked
|
||||
// closures that all fire in the first frame on refocus.
|
||||
if (measureRafRef.current !== null) {
|
||||
window.cancelAnimationFrame(measureRafRef.current)
|
||||
}
|
||||
|
||||
measureRafRef.current = window.requestAnimationFrame(frameStart => {
|
||||
measureRafRef.current = null
|
||||
|
||||
// A newer flush already started; its own measurement wins.
|
||||
if (lastFlushAtRef.current !== startedAt) {
|
||||
return
|
||||
}
|
||||
|
||||
lastFlushCostRef.current = writeCost + Math.max(0, performance.now() - frameStart)
|
||||
})
|
||||
}
|
||||
|
||||
// Always a timer, never requestAnimationFrame. Chromium pauses rAF for a
|
||||
@@ -312,11 +349,46 @@ export function useMessageStream({
|
||||
}
|
||||
|
||||
flushHandleRef.current = null
|
||||
|
||||
if (measureRafRef.current !== null && typeof window !== 'undefined') {
|
||||
window.cancelAnimationFrame(measureRafRef.current)
|
||||
}
|
||||
|
||||
measureRafRef.current = null
|
||||
flushQueuedDeltas()
|
||||
},
|
||||
[flushQueuedDeltas]
|
||||
)
|
||||
|
||||
// Page Visibility does not report every Windows/Linux focus transition.
|
||||
// Flush queued deltas on both signals so returning to a chat cannot leave a
|
||||
// completed chunk waiting for the next throttled timer.
|
||||
// eslint-disable-next-line no-restricted-syntax -- timer-handle clear inside effect, not an atom mirror
|
||||
useEffect(() => {
|
||||
const flushPendingDeltas = () => {
|
||||
if (flushHandleRef.current !== null) {
|
||||
window.clearTimeout(flushHandleRef.current)
|
||||
flushHandleRef.current = null
|
||||
}
|
||||
|
||||
flushQueuedDeltas()
|
||||
}
|
||||
|
||||
const flushWhenVisible = () => {
|
||||
if (document.visibilityState === 'visible') {
|
||||
flushPendingDeltas()
|
||||
}
|
||||
}
|
||||
|
||||
document.addEventListener('visibilitychange', flushWhenVisible)
|
||||
window.addEventListener('focus', flushPendingDeltas)
|
||||
|
||||
return () => {
|
||||
document.removeEventListener('visibilitychange', flushWhenVisible)
|
||||
window.removeEventListener('focus', flushPendingDeltas)
|
||||
}
|
||||
}, [flushQueuedDeltas])
|
||||
|
||||
const appendAssistantDelta = useCallback(
|
||||
(sessionId: string, delta: string) => {
|
||||
if (!delta) {
|
||||
|
||||
@@ -87,7 +87,10 @@ describe('stream delta delivery', () => {
|
||||
})
|
||||
|
||||
expect(states.get(SID)?.messages.at(-1)?.parts).toEqual([{ type: 'text', text: 'first and the rest' }])
|
||||
// The flush must not have depended on a frame at all.
|
||||
expect(rafSpy).not.toHaveBeenCalled()
|
||||
// The flush must not have depended on a frame: this mock parks every rAF
|
||||
// callback, yet the text arrived. runFlush still registers its
|
||||
// adaptive-floor measurement callback here; that one is allowed to wait
|
||||
// for a frame that may never come.
|
||||
expect(rafSpy).toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1797,7 +1797,8 @@ describe('usePromptActions submit / queue drain semantics', () => {
|
||||
expect(accepted).toBe(true)
|
||||
expect(requestGateway).toHaveBeenCalledWith('session.resume', {
|
||||
session_id: 'stored-session-b',
|
||||
source: 'desktop'
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
expect(requestGateway).toHaveBeenCalledWith(
|
||||
'prompt.submit',
|
||||
@@ -1933,7 +1934,8 @@ describe('usePromptActions submit / queue drain semantics', () => {
|
||||
// Must resume the correct stored session to get the right runtime id.
|
||||
expect(requestGateway).toHaveBeenCalledWith('session.resume', {
|
||||
session_id: 'stored-session-a',
|
||||
source: 'desktop'
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
// The prompt must land in the resumed session, NOT the foreground.
|
||||
expect(requestGateway).toHaveBeenCalledWith(
|
||||
@@ -2194,7 +2196,7 @@ describe('usePromptActions redirectPrompt', () => {
|
||||
expect(await handle!.redirectPrompt('reconnect nudge')).toBe(true)
|
||||
expect(calls.map(c => c.method)).toEqual(['session.redirect', 'session.resume', 'session.redirect'])
|
||||
expect(calls[0]?.params).toEqual({ session_id: RUNTIME_SESSION_ID, text: 'reconnect nudge' })
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' })
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', omit_messages: true })
|
||||
expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID, text: 'reconnect nudge' })
|
||||
expect(handle!.activeSessionIdRef.current).toBe(RECOVERED_SESSION_ID)
|
||||
})
|
||||
@@ -2735,7 +2737,7 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
expect(ok).toBe(true)
|
||||
// First submit (stale id) → session.resume (stored id) → retry submit (fresh id).
|
||||
expect(calls.map(c => c.method)).toEqual(['prompt.submit', 'session.resume', 'prompt.submit'])
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' })
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', omit_messages: true })
|
||||
expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID, text: 'message after wake' })
|
||||
})
|
||||
|
||||
@@ -2779,7 +2781,12 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
)
|
||||
|
||||
expect(await handle!.submitText('message after wake')).toBe(true)
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', profile: 'work' })
|
||||
expect(calls[1]?.params).toEqual({
|
||||
session_id: STORED_SESSION_ID,
|
||||
source: 'desktop',
|
||||
omit_messages: true,
|
||||
profile: 'work'
|
||||
})
|
||||
|
||||
setSessions(() => [])
|
||||
})
|
||||
@@ -2826,7 +2833,12 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
)
|
||||
|
||||
expect(await handle!.submitText('message after wake')).toBe(true)
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop', profile: 'work' })
|
||||
expect(calls[1]?.params).toEqual({
|
||||
session_id: STORED_SESSION_ID,
|
||||
source: 'desktop',
|
||||
omit_messages: true,
|
||||
profile: 'work'
|
||||
})
|
||||
|
||||
vi.mocked(getSession).mockReset()
|
||||
setSessions(() => [])
|
||||
@@ -2887,7 +2899,11 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
session_id: 'rt-background-stale',
|
||||
text: 'queued background message after wake'
|
||||
})
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' })
|
||||
expect(calls[1]?.params).toEqual({
|
||||
session_id: STORED_SESSION_ID,
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
expect(calls[2]?.params).toEqual({
|
||||
queued: true,
|
||||
session_id: RECOVERED_SESSION_ID,
|
||||
@@ -2935,7 +2951,11 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
|
||||
expect(calls.map(c => c.method)).toEqual(['session.interrupt', 'session.resume', 'session.interrupt'])
|
||||
expect(calls[0]?.params).toEqual({ session_id: RUNTIME_SESSION_ID })
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' })
|
||||
expect(calls[1]?.params).toEqual({
|
||||
session_id: STORED_SESSION_ID,
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
expect(calls[2]?.params).toEqual({ session_id: RECOVERED_SESSION_ID })
|
||||
})
|
||||
|
||||
@@ -3067,7 +3087,11 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
|
||||
expect(ok).toBe(true)
|
||||
expect(calls.map(c => c.method)).toEqual(['prompt.submit', 'session.resume', 'prompt.submit'])
|
||||
expect(calls[1]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' })
|
||||
expect(calls[1]?.params).toEqual({
|
||||
session_id: STORED_SESSION_ID,
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
expect(calls[2]?.params).toEqual({
|
||||
session_id: RECOVERED_SESSION_ID,
|
||||
text: 'message during starved loop'
|
||||
@@ -3110,7 +3134,11 @@ describe('usePromptActions sleep/wake session recovery', () => {
|
||||
expect(ok).toBe(true)
|
||||
expect(createBackendSessionForSend).not.toHaveBeenCalled()
|
||||
expect(calls.map(c => c.method)).toEqual(['session.resume', 'prompt.submit'])
|
||||
expect(calls[0]?.params).toEqual({ session_id: STORED_SESSION_ID, source: 'desktop' })
|
||||
expect(calls[0]?.params).toEqual({
|
||||
session_id: STORED_SESSION_ID,
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
expect(calls[1]?.params).toMatchObject({ session_id: RECOVERED_SESSION_ID })
|
||||
})
|
||||
|
||||
@@ -3421,7 +3449,8 @@ describe('usePromptActions submit session-context isolation (#54527)', () => {
|
||||
expect(calls.some(c => c.method === 'prompt.submit')).toBe(false)
|
||||
expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({
|
||||
session_id: STORED_SESSION_A,
|
||||
source: 'desktop'
|
||||
source: 'desktop',
|
||||
omit_messages: true
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -639,6 +639,7 @@ export function usePromptActions({
|
||||
const resumed = await requestGateway<{ session_id: string }>('session.resume', {
|
||||
session_id: selectedStoredSessionIdRef.current,
|
||||
source: 'desktop',
|
||||
omit_messages: true,
|
||||
...(resumeProfile ? { profile: resumeProfile } : {})
|
||||
})
|
||||
|
||||
@@ -744,6 +745,7 @@ export function usePromptActions({
|
||||
const resumed = await requestGateway<{ session_id: string }>('session.resume', {
|
||||
session_id: selectedStoredSessionIdRef.current,
|
||||
source: 'desktop',
|
||||
omit_messages: true,
|
||||
...(resumeProfile ? { profile: resumeProfile } : {})
|
||||
})
|
||||
|
||||
|
||||
@@ -484,6 +484,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) {
|
||||
const resumed = await requestGateway<{ session_id: string }>('session.resume', {
|
||||
session_id: targetStoredSessionId,
|
||||
source: 'desktop',
|
||||
omit_messages: true,
|
||||
...(resumeProfile ? { profile: resumeProfile } : {})
|
||||
})
|
||||
|
||||
@@ -637,6 +638,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) {
|
||||
const resumed = await requestGateway<{ session_id: string }>('session.resume', {
|
||||
session_id: recoverStoredSessionId,
|
||||
source: 'desktop',
|
||||
omit_messages: true,
|
||||
...(resumeProfile ? { profile: resumeProfile } : {})
|
||||
})
|
||||
|
||||
|
||||
@@ -896,7 +896,7 @@ describe('resumeSession failure recovery', () => {
|
||||
|
||||
expect(resumeParams).not.toHaveProperty('lazy')
|
||||
expect(resumeParams).not.toHaveProperty('eager_build')
|
||||
expect(resumeParams).toMatchObject({ source: 'desktop' })
|
||||
expect(resumeParams).toMatchObject({ source: 'desktop', omit_messages: true })
|
||||
})
|
||||
|
||||
it('arms the failure latch when resume succeeds with an empty transcript for a non-empty stored session', async () => {
|
||||
@@ -1431,6 +1431,10 @@ describe('resumeSession warm-cache mapping integrity', () => {
|
||||
expect(methods).toContain('session.activate')
|
||||
expect(methods).not.toContain('session.resume')
|
||||
expect(getSessionMessages).toHaveBeenCalledWith('stored-A', undefined)
|
||||
expect(requestGateway).toHaveBeenCalledWith(
|
||||
'session.activate',
|
||||
expect.objectContaining({ omit_messages: true, session_id: 'rt-A' })
|
||||
)
|
||||
expect(runtimeIdByStoredSessionIdRef.current.get('stored-A')).toBe('rt-A')
|
||||
})
|
||||
|
||||
|
||||
@@ -700,7 +700,8 @@ export function useSessionActions({
|
||||
try {
|
||||
activated = await requestGateway<SessionResumeResponse>('session.activate', {
|
||||
session_id: cachedRuntimeId,
|
||||
cols: 96
|
||||
cols: 96,
|
||||
omit_messages: true
|
||||
})
|
||||
} catch (error) {
|
||||
// Compatibility for older backends. Modern backends require
|
||||
@@ -866,12 +867,14 @@ export function useSessionActions({
|
||||
session_id: storedSessionId,
|
||||
cols: 96,
|
||||
source: 'desktop',
|
||||
// REST is the transcript authority for Desktop. Avoid duplicating a
|
||||
// potentially huge compression lineage in the WebSocket response.
|
||||
// Watch windows attach lazily (live mirror). Every other cold resume
|
||||
// gets the gateway's default deferred build: the RPC returns the
|
||||
// transcript immediately instead of blocking the switch on _make_agent
|
||||
// (MCP discovery / prompt build), and the agent pre-warms in the
|
||||
// background while the prefetch above paints the transcript.
|
||||
...(watchWindow ? { lazy: true } : {}),
|
||||
...(watchWindow ? { lazy: true } : { omit_messages: true }),
|
||||
...(sessionProfile ? { profile: sessionProfile } : {})
|
||||
})
|
||||
|
||||
|
||||
+66
@@ -73,4 +73,70 @@ describe('reconcileResumeMessages — structural parts on a mid-turn switch', ()
|
||||
|
||||
expect(assistant.parts.filter(p => p.type === 'tool-call')).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('keeps live-tail structure when the flat dump is not a strict text extension', () => {
|
||||
// Mid-turn sandwich path: cache holds reasoning/tools; resume returns a
|
||||
// longer non-extending dump. Structure source must be live-tail.
|
||||
const cached: ChatMessage[] = [
|
||||
{
|
||||
id: 'assistant-stream-1',
|
||||
pending: true,
|
||||
parts: [
|
||||
{ type: 'reasoning', text: 'thinking about tools' },
|
||||
{ type: 'tool-call', toolCallId: 'c1', toolName: 'terminal', args: {} },
|
||||
{ type: 'text', text: 'partial' }
|
||||
],
|
||||
role: 'assistant'
|
||||
}
|
||||
]
|
||||
|
||||
const authoritative: ChatMessage[] = [
|
||||
{
|
||||
id: 'assistant-stream-1',
|
||||
pending: true,
|
||||
parts: [{ type: 'text', text: 'thinking about tools\nRan terminal\npartial and more dump' }],
|
||||
role: 'assistant'
|
||||
}
|
||||
]
|
||||
|
||||
const [assistant] = reconcileResumeMessages(authoritative, cached)
|
||||
|
||||
expect(assistant.parts.some(part => part.type === 'reasoning')).toBe(true)
|
||||
expect(assistant.parts.some(part => part.type === 'tool-call')).toBe(true)
|
||||
expect(assistant.parts.filter(part => part.type === 'text').map(part => ('text' in part ? part.text : ''))).toEqual(
|
||||
['partial']
|
||||
)
|
||||
})
|
||||
|
||||
it('does not graft historical structure onto a live text-only row after compression rewrote ordinals', () => {
|
||||
// Previous cache still has a completed structured assistant at ordinal 0.
|
||||
// Resume after compression returns a new live text-only assistant at the
|
||||
// same role ordinal for an unrelated turn — must not inherit foreign parts.
|
||||
const cached: ChatMessage[] = [
|
||||
{
|
||||
id: 'old-assistant',
|
||||
parts: [
|
||||
{ type: 'reasoning', text: 'old thinking' },
|
||||
{ type: 'tool-call', toolCallId: 'old-call', toolName: 'terminal', args: {} },
|
||||
{ type: 'text', text: 'old answer' }
|
||||
],
|
||||
role: 'assistant'
|
||||
}
|
||||
]
|
||||
|
||||
const authoritative: ChatMessage[] = [
|
||||
{
|
||||
id: 'assistant-stream-runtime-1',
|
||||
pending: true,
|
||||
parts: [{ type: 'text', text: 'brand new partial' }],
|
||||
role: 'assistant'
|
||||
}
|
||||
]
|
||||
|
||||
const [assistant] = reconcileResumeMessages(authoritative, cached)
|
||||
|
||||
expect(assistant.parts.some(part => part.type === 'reasoning')).toBe(false)
|
||||
expect(assistant.parts.some(part => part.type === 'tool-call')).toBe(false)
|
||||
expect(assistant.parts).toEqual([{ type: 'text', text: 'brand new partial' }])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
|
||||
|
||||
import type { ChatMessage } from '@/lib/chat-messages'
|
||||
import { textWithoutReferenceLines, WIRE_REFERENCE_KINDS } from '@/components/assistant-ui/reference-kinds'
|
||||
import { type ChatMessage, type ChatMessagePart, chatMessageText } from '@/lib/chat-messages'
|
||||
import { $approvalModes, approvalModeForProfile } from '@/store/approval-mode'
|
||||
import { $desktopOnboarding } from '@/store/onboarding'
|
||||
import { $activeGatewayProfile } from '@/store/profile'
|
||||
@@ -24,6 +25,21 @@ import {
|
||||
const msg = (id: string, role: ChatMessage['role'], text: string, extra: Partial<ChatMessage> = {}): ChatMessage =>
|
||||
({ id, role, parts: [{ type: 'text', text }], ...extra }) as ChatMessage
|
||||
|
||||
// A live assistant row carrying the structure the gateway's text-only inflight
|
||||
// snapshot cannot: reasoning and tool calls, with or without any text yet.
|
||||
const streamingMsg = (id: string, text: string, extra: Partial<ChatMessage> = {}): ChatMessage =>
|
||||
({
|
||||
id,
|
||||
role: 'assistant',
|
||||
parts: [
|
||||
{ type: 'reasoning', text: 'planning' },
|
||||
{ type: 'tool-call', toolCallId: 'call-1', toolName: 'terminal', result: 'done' },
|
||||
...(text ? [{ type: 'text', text } as ChatMessagePart] : [])
|
||||
],
|
||||
pending: true,
|
||||
...extra
|
||||
}) as ChatMessage
|
||||
|
||||
const session = (over: Partial<SessionInfo>): SessionInfo => over as SessionInfo
|
||||
|
||||
describe('applyRuntimeInfo approval mode', () => {
|
||||
@@ -394,6 +410,126 @@ describe('reconcileResumeMessages', () => {
|
||||
|
||||
expect(out.attachmentRefs).toBeUndefined()
|
||||
})
|
||||
|
||||
// #75825: switching sessions mid-stream can re-hydrate an empty inflight shell
|
||||
// at the same ordinal as the live stream row that still holds the full reply.
|
||||
it('prefers a richer local pending assistant over an empty projection shell', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'hello from stream', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: true })]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(reconciled[1]).toMatchObject({ id: 'assistant-stream-live', pending: true })
|
||||
expect(chatMessageText(reconciled[1])).toBe('hello from stream')
|
||||
})
|
||||
|
||||
it('prefers a richer local pending assistant when the projection lags mid-stream', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'hello world', { pending: true })
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'hello', { pending: true })
|
||||
]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(chatMessageText(reconciled[1])).toBe('hello world')
|
||||
expect(reconciled[1].id).toBe('assistant-stream-live')
|
||||
})
|
||||
|
||||
it('does not override when the authoritative assistant has advanced further', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'hello', { pending: true })
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'hello world', { pending: true })
|
||||
]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(chatMessageText(reconciled[1])).toBe('hello world')
|
||||
expect(reconciled[1].id).toBe('assistant-stream-sess')
|
||||
})
|
||||
|
||||
// The reported "no inference traces or tool calls": mid tool-work, the local
|
||||
// row holds reasoning + tool calls and NO text yet, so both bodies are empty
|
||||
// text and a text-length comparison cannot tell them apart.
|
||||
it('prefers a traces-only local pending row over an empty shell', () => {
|
||||
const previous = [msg('1-user', 'user', 'run the tools'), streamingMsg('assistant-stream-live', '')]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'run the tools'),
|
||||
msg('assistant-stream-sess', 'assistant', '', { pending: true })
|
||||
]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(reconciled[1].id).toBe('assistant-stream-live')
|
||||
expect(reconciled[1].parts.map(part => part.type)).toEqual(['reasoning', 'tool-call'])
|
||||
})
|
||||
|
||||
// A longer local body that is NOT an extension of the authoritative text is a
|
||||
// different turn at the same ordinal (compression rewrites history) and must
|
||||
// not hijack the slot.
|
||||
it('leaves a shorter non-prefix authoritative assistant intact', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'a long local reply about something else entirely', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('9-assistant', 'assistant', 'short authoritative answer')]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(reconciled[1].id).toBe('9-assistant')
|
||||
expect(chatMessageText(reconciled[1])).toBe('short authoritative answer')
|
||||
})
|
||||
|
||||
// A retained failure snapshot (`inflight.error`) is projected with empty text.
|
||||
// Preferring the local partial over it would erase the error and repaint the
|
||||
// turn as healthy.
|
||||
it('does not treat an errored authoritative row as an empty shell', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'do the thing'),
|
||||
msg('assistant-stream-live', 'assistant', 'partial answer before the failure', { pending: true })
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'do the thing'),
|
||||
msg('assistant-stream-sess', 'assistant', '', { error: 'model call failed: 500' })
|
||||
]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(reconciled[1].error).toBe('model call failed: 500')
|
||||
})
|
||||
|
||||
// Content comes from the renderer; liveness stays the backend's call. A
|
||||
// settled shell (queued turn behind a finished inflight one) must not leave
|
||||
// the preserved reply spinning forever.
|
||||
it('takes the local body but the authoritative settled state', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'streamed body', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: false })]
|
||||
|
||||
const reconciled = reconcileResumeMessages(next, previous)
|
||||
|
||||
expect(reconciled[1]).toMatchObject({ id: 'assistant-stream-live', pending: false })
|
||||
expect(chatMessageText(reconciled[1])).toBe('streamed body')
|
||||
})
|
||||
})
|
||||
|
||||
describe('preserveLocalPendingTurnMessages', () => {
|
||||
@@ -556,7 +692,7 @@ describe('preserveLocalPendingTurnMessages', () => {
|
||||
// `attachmentRefs`. A naive text compare (chatMessageText a === b) therefore
|
||||
// always mismatched whenever an image was attached and re-appended the
|
||||
// optimistic row as a distinct, duplicate user bubble. Both sides must now
|
||||
// reduce to the same visible text via textWithoutImageRefs.
|
||||
// reduce to the same visible text via textWithoutReferenceLines.
|
||||
it('does not duplicate the optimistic image turn when the persisted turn carries @image refs', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'first'),
|
||||
@@ -575,6 +711,91 @@ describe('preserveLocalPendingTurnMessages', () => {
|
||||
expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next)
|
||||
})
|
||||
|
||||
it('does not duplicate the optimistic file turn when the persisted turn carries @file refs', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'first'),
|
||||
msg('2-assistant', 'assistant', 'first answer'),
|
||||
msg('user-optimistic', 'user', 'text', {
|
||||
attachmentRefs: ['@file:X']
|
||||
})
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user-stored', 'user', 'first'),
|
||||
msg('2-assistant-stored', 'assistant', 'first answer'),
|
||||
msg('3-user-stored', 'user', '@file:X\n\ntext')
|
||||
]
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next)
|
||||
})
|
||||
|
||||
it.each(WIRE_REFERENCE_KINDS.filter(kind => kind !== 'file' && kind !== 'image'))(
|
||||
'does not duplicate the optimistic %s turn when the persisted turn carries its directive',
|
||||
kind => {
|
||||
const ref = `@${kind}:X`
|
||||
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'first'),
|
||||
msg('2-assistant', 'assistant', 'first answer'),
|
||||
msg('user-optimistic', 'user', 'text', {
|
||||
attachmentRefs: [ref]
|
||||
})
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user-stored', 'user', 'first'),
|
||||
msg('2-assistant-stored', 'assistant', 'first answer'),
|
||||
msg('3-user-stored', 'user', `${ref}\n\ntext`)
|
||||
]
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next)
|
||||
}
|
||||
)
|
||||
|
||||
it('does not duplicate a directive-only file turn', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'first'),
|
||||
msg('2-assistant', 'assistant', 'first answer'),
|
||||
msg('user-optimistic', 'user', '', {
|
||||
attachmentRefs: ['@file:X']
|
||||
})
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user-stored', 'user', 'first'),
|
||||
msg('2-assistant-stored', 'assistant', 'first answer'),
|
||||
msg('3-user-stored', 'user', '@file:X')
|
||||
]
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next)
|
||||
})
|
||||
|
||||
it('does not duplicate a turn with multiple CRLF directives and Unicode payloads', () => {
|
||||
const refs = ['@file:`資料/über notes.md`', '@url:`https://example.com/café?q=✓`']
|
||||
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'first'),
|
||||
msg('2-assistant', 'assistant', 'first answer'),
|
||||
msg('user-optimistic', 'user', 'text', {
|
||||
attachmentRefs: refs
|
||||
})
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user-stored', 'user', 'first'),
|
||||
msg('2-assistant-stored', 'assistant', 'first answer'),
|
||||
msg('3-user-stored', 'user', `${refs.join('\r\n')}\r\n\r\ntext`)
|
||||
]
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next)
|
||||
})
|
||||
|
||||
it('strips only complete reference lines from visible text', () => {
|
||||
expect(textWithoutReferenceLines('see @file:X here')).toBe('see @file:X here')
|
||||
expect(textWithoutReferenceLines('@file:X trailing prose')).toBe('@file:X trailing prose')
|
||||
expect(textWithoutReferenceLines(' @file:X')).toBe('@file:X')
|
||||
})
|
||||
|
||||
it('still keeps a genuinely uncommitted optimistic image turn when the persisted text differs', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'first'),
|
||||
@@ -599,6 +820,177 @@ describe('preserveLocalPendingTurnMessages', () => {
|
||||
'user-optimistic'
|
||||
])
|
||||
})
|
||||
|
||||
// #75825: an empty inflight projection shell at the same ordinal must not
|
||||
// discard the local pending assistant that still holds the streamed content.
|
||||
// Replace the shell (do not append) so the transcript shows one reply.
|
||||
it('replaces an empty inflight shell with a fuller local pending assistant', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'partial answer so far', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: true })]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(preserved.map(message => message.id)).toEqual(['1-user', 'assistant-stream-live'])
|
||||
expect(chatMessageText(preserved[1])).toBe('partial answer so far')
|
||||
expect(preserved[1].pending).toBe(true)
|
||||
})
|
||||
|
||||
it('replaces a lagging same-id shell with the fuller local pending body', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'full streamed content', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: true })]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(preserved.map(message => message.id)).toEqual(['1-user', 'assistant-stream-sess'])
|
||||
expect(chatMessageText(preserved[1])).toBe('full streamed content')
|
||||
})
|
||||
|
||||
it('still drops local pending when authoritative text is at least as complete', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'partial', { pending: true })
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'partial and more', { pending: true })
|
||||
]
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next)
|
||||
})
|
||||
|
||||
// Mid tool-work both bodies are empty text, so only the parts distinguish the
|
||||
// live row from the shell — the reported "no inference traces or tool calls".
|
||||
it('replaces an empty shell with a traces-only local pending row', () => {
|
||||
const previous = [msg('1-user', 'user', 'run the tools'), streamingMsg('assistant-stream-live', '')]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'run the tools'),
|
||||
msg('assistant-stream-sess', 'assistant', '', { pending: true })
|
||||
]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(preserved).toHaveLength(2)
|
||||
expect(preserved[1].parts.map(part => part.type)).toEqual(['reasoning', 'tool-call'])
|
||||
})
|
||||
|
||||
// Length alone is not identity: a longer local row that does not extend the
|
||||
// authoritative text belongs to another turn and must not take its slot — by
|
||||
// ordinal or by reusing the stream id.
|
||||
it('leaves a shorter non-prefix authoritative assistant intact', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'a long local reply about something else entirely', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('9-assistant', 'assistant', 'short authoritative answer')]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(preserved.map(message => message.id)).toEqual(['1-user', '9-assistant'])
|
||||
expect(chatMessageText(preserved[1])).toBe('short authoritative answer')
|
||||
})
|
||||
|
||||
it('leaves a shorter non-prefix authoritative assistant intact on the same stream id', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'a long local reply about something else entirely', { pending: true })
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'short authoritative answer')
|
||||
]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(chatMessageText(preserved[1])).toBe('short authoritative answer')
|
||||
})
|
||||
|
||||
it('does not erase a retained failure with the local partial', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'do the thing'),
|
||||
msg('assistant-stream-live', 'assistant', 'partial answer before the failure', { pending: true })
|
||||
]
|
||||
|
||||
const next = [
|
||||
msg('1-user', 'user', 'do the thing'),
|
||||
msg('assistant-stream-sess', 'assistant', '', { error: 'model call failed: 500' })
|
||||
]
|
||||
|
||||
const assistant = preserveLocalPendingTurnMessages(next, previous).find(message => message.role === 'assistant')
|
||||
|
||||
expect(assistant?.error).toBe('model call failed: 500')
|
||||
expect(assistant?.pending).not.toBe(true)
|
||||
})
|
||||
|
||||
it('takes the local body but the authoritative settled state', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-live', 'assistant', 'streamed body', { pending: true })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question'), msg('assistant-stream-sess', 'assistant', '', { pending: false })]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(preserved[1]).toMatchObject({ id: 'assistant-stream-live', pending: false })
|
||||
expect(chatMessageText(preserved[1])).toBe('streamed body')
|
||||
})
|
||||
|
||||
// #70209: history committed the reply under its own id, so the settled local
|
||||
// stream row sits at a later ordinal, pairs with nothing, and gets appended —
|
||||
// the same answer twice.
|
||||
it('does not re-append a settled stream row the authoritative history already carries', () => {
|
||||
const next = [msg('1-user-stored', 'user', 'question'), msg('2-assistant-stored', 'assistant', 'answer')]
|
||||
const settledLocalStream = msg('assistant-stream-runtime-1', 'assistant', 'answer', { pending: false })
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, [...next, settledLocalStream])).toBe(next)
|
||||
})
|
||||
|
||||
// The reply finished locally but the gateway had not committed it when the
|
||||
// session was reopened — the local row is the only copy and must survive.
|
||||
it('keeps a settled stream row the authoritative history has not committed', () => {
|
||||
const previous = [
|
||||
msg('1-user', 'user', 'question'),
|
||||
msg('assistant-stream-sess', 'assistant', 'the finished reply', { pending: false })
|
||||
]
|
||||
|
||||
const next = [msg('1-user', 'user', 'question')]
|
||||
|
||||
expect(preserveLocalPendingTurnMessages(next, previous).map(message => message.id)).toEqual([
|
||||
'1-user',
|
||||
'assistant-stream-sess'
|
||||
])
|
||||
})
|
||||
|
||||
// The whole point of replacing rather than appending: one reply on screen,
|
||||
// and the committed history around the live turn untouched.
|
||||
it('does not duplicate or rewrite committed history around the live turn', () => {
|
||||
const history = [
|
||||
msg('1-user', 'user', 'first question'),
|
||||
msg('2-assistant', 'assistant', 'first answer'),
|
||||
msg('3-user', 'user', 'run the tools')
|
||||
]
|
||||
|
||||
const previous = [...history, streamingMsg('assistant-stream-live', 'here is the full reply')]
|
||||
const next = [...history, msg('assistant-stream-sess', 'assistant', '', { pending: true })]
|
||||
|
||||
const preserved = preserveLocalPendingTurnMessages(next, previous)
|
||||
|
||||
expect(preserved).toHaveLength(4)
|
||||
expect(chatMessageText(preserved[1])).toBe('first answer')
|
||||
expect(preserved.filter(message => message.role === 'assistant')).toHaveLength(2)
|
||||
})
|
||||
})
|
||||
|
||||
describe('appendLiveSessionProjection', () => {
|
||||
@@ -730,4 +1122,73 @@ describe('appendLiveSessionProjection', () => {
|
||||
|
||||
expect(appendLiveSessionProjection(stored, { session_id: 'runtime-1' })).toBe(stored)
|
||||
})
|
||||
|
||||
it('does not sandwich a structured mid-turn row with the inflight flat dump (#76444)', () => {
|
||||
const stored: ChatMessage[] = [
|
||||
msg('stored-user', 'user', 'do the work'),
|
||||
{
|
||||
id: 'live-assistant',
|
||||
role: 'assistant',
|
||||
pending: true,
|
||||
parts: [
|
||||
{ type: 'reasoning', text: 'thinking about tools' },
|
||||
{ type: 'tool-call', toolCallId: 'c1', toolName: 'terminal', args: {} },
|
||||
{ type: 'text', text: 'partial' }
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
const restored = appendLiveSessionProjection(stored, {
|
||||
session_id: 'runtime-1',
|
||||
inflight: {
|
||||
user: 'do the work',
|
||||
// Flat dump includes thinking chatter + tool narration — longer than
|
||||
// the answer text alone, which is how the sandwich used to grow.
|
||||
assistant: 'thinking about tools\nRan terminal\npartial and more dump',
|
||||
streaming: true
|
||||
}
|
||||
})
|
||||
|
||||
const assistants = restored.filter(message => message.role === 'assistant')
|
||||
expect(assistants).toHaveLength(1)
|
||||
expect(assistants[0].id).toBe('live-assistant')
|
||||
expect(assistants[0].parts.some(part => part.type === 'reasoning')).toBe(true)
|
||||
expect(assistants[0].parts.some(part => part.type === 'tool-call')).toBe(true)
|
||||
// Answer text stays the structured row's text, not the dump.
|
||||
expect(
|
||||
assistants[0].parts.filter(part => part.type === 'text').map(part => ('text' in part ? part.text : ''))
|
||||
).toEqual(['partial'])
|
||||
})
|
||||
|
||||
it('still projects inflight when only a completed historical tool reply has structure', () => {
|
||||
// Older completed assistants keep reasoning/tool parts in the full
|
||||
// transcript; they must not suppress a new turn's text projection.
|
||||
const stored: ChatMessage[] = [
|
||||
msg('old-user', 'user', 'previous task'),
|
||||
{
|
||||
id: 'old-assistant',
|
||||
role: 'assistant',
|
||||
parts: [
|
||||
{ type: 'tool-call', toolCallId: 'old', toolName: 'terminal', args: {} },
|
||||
{ type: 'text', text: 'done earlier' }
|
||||
]
|
||||
},
|
||||
msg('new-user', 'user', 'new task')
|
||||
]
|
||||
|
||||
const restored = appendLiveSessionProjection(stored, {
|
||||
session_id: 'runtime-1',
|
||||
inflight: {
|
||||
user: 'new task',
|
||||
assistant: 'working on it',
|
||||
streaming: true
|
||||
}
|
||||
})
|
||||
|
||||
expect(restored.map(message => message.id)).toContain('assistant-stream-runtime-1')
|
||||
expect(restored.at(-1)).toMatchObject({
|
||||
id: 'assistant-stream-runtime-1',
|
||||
pending: true
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
import { textWithoutReferenceLines } from '@/components/assistant-ui/reference-kinds'
|
||||
import { getSession } from '@/hermes'
|
||||
import { assistantTextPart, type ChatMessage, chatMessageText, textPart } from '@/lib/chat-messages'
|
||||
import { normalizePersonalityValue } from '@/lib/chat-runtime'
|
||||
import { embeddedImageUrls, textWithoutEmbeddedImages, textWithoutImageRefs } from '@/lib/embedded-images'
|
||||
import { embeddedImageUrls, textWithoutEmbeddedImages } from '@/lib/embedded-images'
|
||||
import { reconcileApprovalModeForProfile } from '@/store/approval-mode'
|
||||
import { requestDesktopOnboardingForCredentialWarning } from '@/store/onboarding'
|
||||
import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/profile'
|
||||
@@ -46,6 +47,41 @@ function withAppendedText(message: ChatMessage, suffix: string): ChatMessage {
|
||||
return appended ? { ...message, parts } : message
|
||||
}
|
||||
|
||||
/** Reasoning / tool-call parts that the gateway inflight dump cannot express. */
|
||||
function hasStructuralParts(message: ChatMessage): boolean {
|
||||
return message.parts.some(part => part.type === 'reasoning' || part.type === 'tool-call')
|
||||
}
|
||||
|
||||
/**
|
||||
* A live-turn row — the gateway's text-only `inflight` projection, a
|
||||
* still-streaming local bubble, or an interim row sealed inside the running
|
||||
* turn — as opposed to a committed transcript row.
|
||||
*/
|
||||
function isLiveTailRow(message: ChatMessage): boolean {
|
||||
return (
|
||||
message.pending === true ||
|
||||
message.id.startsWith('assistant-stream-') ||
|
||||
message.id.startsWith('inflight-assistant-') ||
|
||||
message.interim === true
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* True when `next` is a pure forward extension of the previous *answer* text.
|
||||
* Empty previous answer never accepts a dump as an extension — that is how the
|
||||
* mid-turn inflight flat dump used to sandwich structured rows (#76444).
|
||||
*/
|
||||
export function isStrictAnswerTextExtension(next: string, previous: string): boolean {
|
||||
const n = next.trim()
|
||||
const p = previous.trim()
|
||||
|
||||
if (!p || !n) {
|
||||
return false
|
||||
}
|
||||
|
||||
return n.startsWith(p)
|
||||
}
|
||||
|
||||
/**
|
||||
* Carry structural parts an authoritative row cannot express.
|
||||
*
|
||||
@@ -250,8 +286,20 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes
|
||||
const nextText = chatMessageText(message).trim()
|
||||
const previousText = chatMessageText(previous)
|
||||
const previousVisibleText = textWithoutEmbeddedImages(previousText)
|
||||
const previousTrimmed = previousVisibleText.trim()
|
||||
let preserved = message
|
||||
|
||||
// #75825: resume can project an empty (or lagging) inflight assistant shell
|
||||
// at the same role-ordinal as the live stream row that still holds the
|
||||
// streamed text, reasoning and tool calls. Prefer that richer pending row
|
||||
// instead of painting the shell — otherwise the reply vanishes until
|
||||
// restart. Guarded to the same reply further along (see
|
||||
// localPendingSupersedes) so a different turn at the same ordinal cannot
|
||||
// hijack the slot.
|
||||
if (localPendingSupersedes(previous, message)) {
|
||||
return withAuthoritativeTurnState(previous, message)
|
||||
}
|
||||
|
||||
const sameText = nextText === previousVisibleText || nextText === previousText.trim()
|
||||
|
||||
// Mid-turn, the authoritative text has advanced past the cached copy by one
|
||||
@@ -260,12 +308,36 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes
|
||||
// for structural carry-over. Attachment refs and image re-appending stay on
|
||||
// the strict equality path — they reconcile a SETTLED row, and a growing
|
||||
// row is by definition not settled.
|
||||
//
|
||||
// Live-tail identity: structure-only same-turn carry is allowed only when
|
||||
// the *structure-bearing cached row* is still the in-flight stream
|
||||
// (pending / stream id / interim). Marking only the text-only next row
|
||||
// live is not enough — after compression a new live assistant can share a
|
||||
// role ordinal with an unrelated historical structured row and must not
|
||||
// inherit its reasoning/tool parts (#76444 review / salvage).
|
||||
const sameTurn =
|
||||
sameText ||
|
||||
(nextText.length > 0 && previousVisibleText.length > 0 && nextText.startsWith(previousVisibleText.trim()))
|
||||
(nextText.length > 0 && previousTrimmed.length > 0 && isStrictAnswerTextExtension(nextText, previousTrimmed)) ||
|
||||
(message.role === 'assistant' &&
|
||||
previous.role === 'assistant' &&
|
||||
hasStructuralParts(previous) &&
|
||||
!hasStructuralParts(message) &&
|
||||
isLiveTailRow(previous))
|
||||
|
||||
if (sameTurn) {
|
||||
preserved = preserveStructuralParts(preserved, previous)
|
||||
|
||||
// Never replace structured answer text with a non-extending flat dump.
|
||||
if (
|
||||
message.role === 'assistant' &&
|
||||
hasStructuralParts(previous) &&
|
||||
!hasStructuralParts(message) &&
|
||||
!isStrictAnswerTextExtension(nextText, previousVisibleText)
|
||||
) {
|
||||
const nonText = preserved.parts.filter(part => part.type !== 'text')
|
||||
const priorAnswer = previous.parts.filter(part => part.type === 'text')
|
||||
preserved = { ...preserved, parts: [...nonText, ...priorAnswer] }
|
||||
}
|
||||
}
|
||||
|
||||
if (
|
||||
@@ -315,8 +387,10 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes
|
||||
* history window. Preserve only the newest optimistic user row: compression
|
||||
* rewrites past context, so older `user-*` rows in a warm cache are stale
|
||||
* history, not in-flight work. The latest authoritative user confirms whether
|
||||
* that tail has persisted; any authoritative assistant at the same ordinal
|
||||
* supersedes the local stream.
|
||||
* that tail has persisted. An authoritative assistant at the same ordinal
|
||||
* supersedes the local stream only when it is at least as complete; an empty
|
||||
* or lagging inflight shell must not discard a fuller local pending reply
|
||||
* (#75825).
|
||||
*
|
||||
* Gateway bookkeeping markers (the model-switch / personality notices written
|
||||
* by tui_gateway/server.py) are persisted as role=user but are not user turns.
|
||||
@@ -328,6 +402,64 @@ export function reconcileResumeMessages(nextMessages: ChatMessage[], previousMes
|
||||
const isGatewaySystemMarker = (message: ChatMessage): boolean =>
|
||||
message.role === 'user' && chatMessageText(message).trimStart().startsWith('[System:')
|
||||
|
||||
/**
|
||||
* Does the row carry anything a viewer would miss — streamed answer text, or
|
||||
* the reasoning / tool-call structure the gateway's flat dump cannot express?
|
||||
* An empty inflight shell carries none of it.
|
||||
*/
|
||||
const hasStreamedContent = (message: ChatMessage): boolean =>
|
||||
chatMessageText(message).trim().length > 0 || hasStructuralParts(message)
|
||||
|
||||
/**
|
||||
* May the cached local row stand in for this authoritative assistant?
|
||||
*
|
||||
* Only for a live projection of the SAME reply that the local copy is further
|
||||
* along on: an empty shell, or text the local row strictly extends. Comparing
|
||||
* lengths alone lets an unrelated (merely longer) local row hijack the ordinal
|
||||
* — or the stream id — of a genuine stored reply. A retained failure snapshot
|
||||
* (`inflight.error`, projected with empty text) is never a shell: repainting it
|
||||
* from the local partial would hide the error and mark the turn healthy again.
|
||||
*/
|
||||
const localPendingSupersedes = (local: ChatMessage, authoritative: ChatMessage): boolean => {
|
||||
if (local.role !== 'assistant' || !isLiveTailRow(local)) {
|
||||
return false
|
||||
}
|
||||
|
||||
if (!isLiveTailRow(authoritative) || authoritative.error) {
|
||||
return false
|
||||
}
|
||||
|
||||
const authoritativeText = chatMessageText(authoritative).trim()
|
||||
|
||||
if (!authoritativeText.length) {
|
||||
return hasStreamedContent(local)
|
||||
}
|
||||
|
||||
const localText = chatMessageText(local).trim()
|
||||
|
||||
return localText.length > authoritativeText.length && isStrictAnswerTextExtension(localText, authoritativeText)
|
||||
}
|
||||
|
||||
/**
|
||||
* Take the cached row's content, but never its liveness. The renderer holds the
|
||||
* only copy of the streamed parts; the gateway remains the authority on whether
|
||||
* the turn is still running and on durable row identity — so a settled shell
|
||||
* must not repaint the reply as perpetually streaming.
|
||||
*/
|
||||
const withAuthoritativeTurnState = (local: ChatMessage, authoritative: ChatMessage): ChatMessage => {
|
||||
const merged: ChatMessage = { ...local, pending: authoritative.pending === true }
|
||||
|
||||
if (local.rowId === undefined && authoritative.rowId !== undefined) {
|
||||
merged.rowId = authoritative.rowId
|
||||
}
|
||||
|
||||
if (local.reactions === undefined && authoritative.reactions?.length) {
|
||||
merged.reactions = [...authoritative.reactions]
|
||||
}
|
||||
|
||||
return merged
|
||||
}
|
||||
|
||||
export function preserveLocalPendingTurnMessages(
|
||||
nextMessages: ChatMessage[],
|
||||
previousMessages: ChatMessage[]
|
||||
@@ -378,6 +510,9 @@ export function preserveLocalPendingTurnMessages(
|
||||
|
||||
const latestAuthoritativeUser = [...nextMessages].reverse().find(message => message.role === 'user')
|
||||
const preserved: ChatMessage[] = []
|
||||
// Authoritative id → richer local pending row. Replacing (not appending)
|
||||
// avoids painting both the empty inflight shell and the full stream bubble.
|
||||
const replacements = new Map<string, ChatMessage>()
|
||||
|
||||
for (const message of previousMessages) {
|
||||
if (isGatewaySystemMarker(message)) {
|
||||
@@ -392,7 +527,21 @@ export function preserveLocalPendingTurnMessages(
|
||||
const isPendingAssistant =
|
||||
message.role === 'assistant' && (message.pending === true || message.id.startsWith('assistant-stream-'))
|
||||
|
||||
if ((!isOptimisticUser && !isPendingAssistant) || nextIds.has(message.id)) {
|
||||
if (!isOptimisticUser && !isPendingAssistant) {
|
||||
continue
|
||||
}
|
||||
|
||||
// Same id already present: still prefer a strictly more complete local
|
||||
// pending body over an empty/stale shell that reused the stream id.
|
||||
if (nextIds.has(message.id)) {
|
||||
if (isPendingAssistant) {
|
||||
const existing = nextMessages.find(candidate => candidate.id === message.id)
|
||||
|
||||
if (existing && localPendingSupersedes(message, existing)) {
|
||||
replacements.set(message.id, withAuthoritativeTurnState(message, existing))
|
||||
}
|
||||
}
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -403,19 +552,50 @@ export function preserveLocalPendingTurnMessages(
|
||||
if (
|
||||
isOptimisticUser &&
|
||||
latestAuthoritativeUser &&
|
||||
textWithoutImageRefs(chatMessageText(latestAuthoritativeUser)) === textWithoutImageRefs(chatMessageText(message))
|
||||
textWithoutReferenceLines(chatMessageText(latestAuthoritativeUser)) ===
|
||||
textWithoutReferenceLines(chatMessageText(message))
|
||||
) {
|
||||
continue
|
||||
}
|
||||
|
||||
const authoritative = nextByRoleOrdinal.get(`${message.role}:${ordinal}`)
|
||||
|
||||
// A settled stream row (`pending: false` after message.complete) whose reply
|
||||
// the authoritative transcript already carries under its committed id is
|
||||
// stale: ordinal pairing can't see it, because the commit shifted the row
|
||||
// one ordinal earlier, and re-appending it renders the same answer twice
|
||||
// (#70209). Only text-identical rows are dropped — a settled row the backend
|
||||
// has NOT committed yet is the only copy of that reply and must survive.
|
||||
if (
|
||||
isPendingAssistant &&
|
||||
message.pending !== true &&
|
||||
nextMessages.some(
|
||||
candidate =>
|
||||
candidate.role === 'assistant' &&
|
||||
textWithoutReferenceLines(chatMessageText(candidate)) === textWithoutReferenceLines(chatMessageText(message))
|
||||
)
|
||||
) {
|
||||
continue
|
||||
}
|
||||
|
||||
if (authoritative) {
|
||||
if (isPendingAssistant) {
|
||||
// Keep the local pending row when it is the same reply further along
|
||||
// and the authoritative row is an empty projection shell or a prefix.
|
||||
// #75825
|
||||
if (!localPendingSupersedes(message, authoritative)) {
|
||||
continue
|
||||
}
|
||||
|
||||
replacements.set(authoritative.id, withAuthoritativeTurnState(message, authoritative))
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
if (textWithoutImageRefs(chatMessageText(authoritative)) === textWithoutImageRefs(chatMessageText(message))) {
|
||||
if (
|
||||
textWithoutReferenceLines(chatMessageText(authoritative)) ===
|
||||
textWithoutReferenceLines(chatMessageText(message))
|
||||
) {
|
||||
continue
|
||||
}
|
||||
}
|
||||
@@ -423,7 +603,10 @@ export function preserveLocalPendingTurnMessages(
|
||||
preserved.push(message)
|
||||
}
|
||||
|
||||
return preserved.length ? [...nextMessages, ...preserved] : nextMessages
|
||||
const withReplacements =
|
||||
replacements.size > 0 ? nextMessages.map(message => replacements.get(message.id) ?? message) : nextMessages
|
||||
|
||||
return preserved.length ? [...withReplacements, ...preserved] : withReplacements
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -484,7 +667,9 @@ export function appendLiveSessionProjection(
|
||||
}
|
||||
|
||||
const persistedInLatestRun = (text: string): boolean =>
|
||||
latestUserRun.some(message => textWithoutImageRefs(chatMessageText(message)) === textWithoutImageRefs(text))
|
||||
latestUserRun.some(
|
||||
message => textWithoutReferenceLines(chatMessageText(message)) === textWithoutReferenceLines(text)
|
||||
)
|
||||
|
||||
const inflightUserAlreadyPersisted = Boolean(inflightUser) && persistedInLatestRun(inflightUser)
|
||||
|
||||
@@ -514,14 +699,54 @@ export function appendLiveSessionProjection(
|
||||
|
||||
// Keep a pending assistant boundary even before the first delta when a
|
||||
// queued user turn follows it. This preserves the two distinct turns.
|
||||
//
|
||||
// When the *current live turn* already holds a structured mid-turn assistant
|
||||
// row (reasoning / tool-call from the live stream or journal), do NOT append
|
||||
// a pure-text projection of `inflight.assistant` — that flat dump re-renders
|
||||
// thinking as answer text and sandwiches the structured parts (#76444).
|
||||
// Only inspect the live tail after the latest user run — never a completed
|
||||
// historical tool-bearing reply earlier in the transcript (review feedback).
|
||||
const liveStreamId = `assistant-stream-${sessionId}`
|
||||
|
||||
const liveAssistantOfCurrentTurn = ((): ChatMessage | null => {
|
||||
const byStreamId = messages.find(message => message.id === liveStreamId)
|
||||
|
||||
if (byStreamId) {
|
||||
return byStreamId
|
||||
}
|
||||
|
||||
// Assistants after the latest user row belong to this turn's tail.
|
||||
if (latestUserIndex < 0) {
|
||||
return null
|
||||
}
|
||||
|
||||
for (let index = messages.length - 1; index > latestUserIndex; index -= 1) {
|
||||
if (messages[index].role === 'assistant') {
|
||||
return messages[index]
|
||||
}
|
||||
}
|
||||
|
||||
return null
|
||||
})()
|
||||
|
||||
const turnAlreadyStructured = Boolean(
|
||||
liveAssistantOfCurrentTurn &&
|
||||
hasStructuralParts(liveAssistantOfCurrentTurn) &&
|
||||
isLiveTailRow(liveAssistantOfCurrentTurn)
|
||||
)
|
||||
|
||||
if (inflightAssistant || inflightStreaming || inflightError || (inflightUser && queuedUser)) {
|
||||
projected.push({
|
||||
id: `assistant-stream-${sessionId}`,
|
||||
role: 'assistant',
|
||||
parts: inflightAssistant ? [assistantTextPart(inflightAssistant)] : [],
|
||||
pending: inflightStreaming,
|
||||
...(inflightError ? { error: inflightError } : {})
|
||||
})
|
||||
if (turnAlreadyStructured && !inflightError) {
|
||||
// Structure is authoritative; skip the text-only dump row.
|
||||
} else {
|
||||
projected.push({
|
||||
id: liveStreamId,
|
||||
role: 'assistant',
|
||||
parts: inflightAssistant ? [assistantTextPart(inflightAssistant)] : [],
|
||||
pending: inflightStreaming,
|
||||
...(inflightError ? { error: inflightError } : {})
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
if (queuedUser) {
|
||||
|
||||
@@ -157,3 +157,17 @@ const REFERENCE_PATTERN = /@(file|folder|url|image|tool|line|terminal|session):(
|
||||
export function referenceRe(): RegExp {
|
||||
return new RegExp(REFERENCE_PATTERN.source, 'g')
|
||||
}
|
||||
|
||||
/** Remove reference-only lines when comparing visible message text. */
|
||||
// Anchored + non-global: no shared `lastIndex` state (the hazard referenceRe()
|
||||
// exists to avoid), and hoisting skips a RegExp construction per call — this
|
||||
// runs on both sides of every message comparison in the reconcile loops.
|
||||
const REFERENCE_LINE_RE = new RegExp(`^(?:${REFERENCE_PATTERN.source})$`)
|
||||
|
||||
export function textWithoutReferenceLines(text: string): string {
|
||||
return text
|
||||
.split('\n')
|
||||
.filter(line => !REFERENCE_LINE_RE.test(line.trimEnd()))
|
||||
.join('\n')
|
||||
.trim()
|
||||
}
|
||||
|
||||
@@ -8,7 +8,8 @@ import {
|
||||
liveTailStart,
|
||||
type MessageGroup,
|
||||
messageRenderWeight,
|
||||
RENDER_WEIGHT_CHARS
|
||||
RENDER_WEIGHT_CHARS,
|
||||
resolveThreadScrollTarget
|
||||
} from './list'
|
||||
|
||||
// Signature rows are `${index}:${id}:${role}:${weight}` (see the useAuiState
|
||||
@@ -62,6 +63,53 @@ describe('buildGroups', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('resolveThreadScrollTarget', () => {
|
||||
const context = (scrollElement: Pick<HTMLElement, 'scrollTop'>) => ({
|
||||
contentElement: document.createElement('div'),
|
||||
scrollElement: scrollElement as HTMLElement
|
||||
})
|
||||
|
||||
it('settles when the browser clamps the requested bottom within half a CSS pixel', () => {
|
||||
let actualScrollTop = 0
|
||||
let writes = 0
|
||||
|
||||
const scrollElement = {
|
||||
get scrollTop() {
|
||||
return actualScrollTop
|
||||
},
|
||||
set scrollTop(value: number) {
|
||||
writes += 1
|
||||
actualScrollTop = value - 0.125
|
||||
}
|
||||
}
|
||||
|
||||
const target = 899
|
||||
|
||||
const requested = resolveThreadScrollTarget(target, context(scrollElement))
|
||||
scrollElement.scrollTop = requested
|
||||
const settled = resolveThreadScrollTarget(target, context(scrollElement))
|
||||
|
||||
expect(requested).toBe(target)
|
||||
expect(actualScrollTop).toBe(898.875)
|
||||
expect(settled).toBe(actualScrollTop)
|
||||
expect(actualScrollTop < settled).toBe(false)
|
||||
expect(writes).toBe(1)
|
||||
})
|
||||
|
||||
it('keeps following while more than half a CSS pixel remains', () => {
|
||||
const scrollElement = { scrollTop: 898.25 }
|
||||
|
||||
expect(resolveThreadScrollTarget(899, context(scrollElement))).toBe(899)
|
||||
})
|
||||
|
||||
it('re-arms after streaming content increases the target', () => {
|
||||
const scrollElement = { scrollTop: 898.875 }
|
||||
|
||||
expect(resolveThreadScrollTarget(899, context(scrollElement))).toBe(898.875)
|
||||
expect(resolveThreadScrollTarget(999, context(scrollElement))).toBe(999)
|
||||
})
|
||||
})
|
||||
|
||||
describe('firstVisibleGroupIndex', () => {
|
||||
const group = (id: string, weight: number): MessageGroup => ({ id, index: 0, kind: 'standalone', weight })
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ import {
|
||||
useRef,
|
||||
useState
|
||||
} from 'react'
|
||||
import { useStickToBottom } from 'use-stick-to-bottom'
|
||||
import { type GetTargetScrollTop, useStickToBottom } from 'use-stick-to-bottom'
|
||||
|
||||
import { useI18n } from '@/i18n'
|
||||
import { cn } from '@/lib/utils'
|
||||
@@ -62,6 +62,20 @@ const MAX_MEASURED_MESSAGE_CHARS = RENDER_BUDGET * RENDER_WEIGHT_CHARS
|
||||
// blocks the click-to-paint path.
|
||||
const FIRST_PAINT_BUDGET = 20
|
||||
|
||||
// Browsers may quantize a requested scrollTop to a nearby device-pixel
|
||||
// boundary. use-stick-to-bottom otherwise compares the lower actual value to
|
||||
// the integer target forever, re-requesting the same instant scroll every
|
||||
// frame. Treat a subpixel remainder as achieved; larger gaps still follow new
|
||||
// streamed content normally.
|
||||
const SCROLL_TARGET_EPSILON_PX = 0.5
|
||||
|
||||
export const resolveThreadScrollTarget: GetTargetScrollTop = (targetScrollTop, { scrollElement }) => {
|
||||
const currentScrollTop = scrollElement.scrollTop
|
||||
const remaining = targetScrollTop - currentScrollTop
|
||||
|
||||
return remaining >= 0 && remaining <= SCROLL_TARGET_EPSILON_PX ? currentScrollTop : targetScrollTop
|
||||
}
|
||||
|
||||
const contentWeightCache = new WeakMap<object, number>()
|
||||
const NON_RENDERED_CONTENT_FIELDS = new Set(['id', 'role', 'toolCallId', 'toolName', 'type'])
|
||||
|
||||
@@ -297,7 +311,8 @@ const ThreadMessageListInner: FC<ThreadMessageListProps> = ({
|
||||
// settling. Its refs hang off our own DOM so the sticky human bubbles survive.
|
||||
const { scrollRef, contentRef, isAtBottom, scrollToBottom, stopScroll } = useStickToBottom({
|
||||
initial: 'instant',
|
||||
resize: 'instant'
|
||||
resize: 'instant',
|
||||
targetScrollTop: resolveThreadScrollTarget
|
||||
})
|
||||
|
||||
const [renderBudget, setRenderBudget] = useState(FIRST_PAINT_BUDGET)
|
||||
|
||||
@@ -9,6 +9,7 @@ import { ActivityTimerText } from '@/components/chat/activity-timer-text'
|
||||
import { SCAFFOLD_LABEL_CLASS } from '@/components/chat/scaffold-row'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { Loader } from '@/components/ui/loader'
|
||||
import { StatusPulse } from '@/components/ui/status-pulse'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { $backgroundResume } from '@/store/background-delegation'
|
||||
@@ -130,7 +131,11 @@ export const ResponseLoadingIndicator: FC = () => {
|
||||
|
||||
return (
|
||||
<StatusRow data-slot="aui_response-loading" label={hint || t.assistant.thread.loadingResponse}>
|
||||
<span aria-hidden="true" className="dither inline-block size-3 rounded-[2px] text-midground/80 animate-pulse" />
|
||||
<StatusPulse
|
||||
aria-hidden="true"
|
||||
className="dither inline-block size-3 rounded-[2px] text-midground/80"
|
||||
kind="opacity"
|
||||
/>
|
||||
{hint && <HintText>{hint}</HintText>}
|
||||
<ActivityTimerText seconds={elapsed} />
|
||||
</StatusRow>
|
||||
@@ -236,7 +241,11 @@ export const StreamStallIndicator: FC = () => {
|
||||
|
||||
return (
|
||||
<StatusRow data-slot="aui_stream-stall" label={hint || 'Hermes is thinking'}>
|
||||
<span aria-hidden="true" className="dither inline-block size-3 rounded-[2px] text-midground/80 animate-pulse" />
|
||||
<StatusPulse
|
||||
aria-hidden="true"
|
||||
className="dither inline-block size-3 rounded-[2px] text-midground/80"
|
||||
kind="opacity"
|
||||
/>
|
||||
{hint && <HintText>{hint}</HintText>}
|
||||
<ActivityTimerText seconds={elapsed} />
|
||||
</StatusRow>
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
// Right-click a tab -> Reload remounts THAT pane's content: its epoch (the
|
||||
// React key the zone renderer hands the contribution) advances, and no other
|
||||
// pane's does. The layout tree itself must never move.
|
||||
|
||||
describe('reloadTreePane', () => {
|
||||
beforeEach(() => {
|
||||
window.localStorage.clear()
|
||||
vi.resetModules()
|
||||
})
|
||||
|
||||
it('advances only the reloaded pane epoch and leaves the tree alone', async () => {
|
||||
const tree = await import('@/components/pane-shell/tree/store')
|
||||
const model = await import('@/components/pane-shell/tree/model')
|
||||
|
||||
tree.declareDefaultTree(model.group(['workspace', 'files'], { active: 'workspace', id: 'grp-main' }))
|
||||
|
||||
const before = tree.$layoutTree.get()
|
||||
|
||||
tree.reloadTreePane('workspace')
|
||||
tree.reloadTreePane('workspace')
|
||||
|
||||
expect(tree.$treePaneEpochs.get().workspace).toBe(2)
|
||||
expect(tree.$treePaneEpochs.get().files).toBeUndefined()
|
||||
expect(tree.$layoutTree.get()).toBe(before)
|
||||
})
|
||||
})
|
||||
@@ -33,6 +33,7 @@ import {
|
||||
$newSessionTabAction,
|
||||
$panesWithCloser,
|
||||
$treeDragging,
|
||||
$treePaneEpochs,
|
||||
activateTreePane,
|
||||
closeAllTreeTabs,
|
||||
closeOtherTreeTabs,
|
||||
@@ -42,6 +43,7 @@ import {
|
||||
isCollapsePane,
|
||||
isSessionStripPane,
|
||||
noteActiveTreeGroup,
|
||||
reloadTreePane,
|
||||
restoreTreePane,
|
||||
SESSION_TILE_DRAG,
|
||||
setTreeGroupHeaderHidden,
|
||||
@@ -96,6 +98,12 @@ function ZoneMenu({
|
||||
|
||||
return (
|
||||
<>
|
||||
{renderActionItem(kit, {
|
||||
icon: 'refresh',
|
||||
label: t.zones.reload,
|
||||
onSelect: () => reloadTreePane(targetPane())
|
||||
})}
|
||||
<kit.Separator />
|
||||
{paneId !== undefined &&
|
||||
renderActionItem(kit, {
|
||||
icon: 'close',
|
||||
@@ -178,6 +186,9 @@ export function TreeGroup({
|
||||
const narrow = useStore($narrowViewport)
|
||||
const newSessionTabAction = useStore($newSessionTabAction)
|
||||
const panesWithCloser = useStore($panesWithCloser)
|
||||
// Reload epochs: only an explicit tab-menu Reload writes here, so this
|
||||
// subscription costs nothing on a normal render.
|
||||
const paneEpochs = useStore($treePaneEpochs)
|
||||
|
||||
const paneFor = (id: string) => panes.find(p => p.id === id)
|
||||
|
||||
@@ -557,9 +568,14 @@ export function TreeGroup({
|
||||
// can gate its hot (per-token) subscriptions while hidden;
|
||||
// the group id identifies the ZONE it lives in, for state
|
||||
// that is per-zone rather than per-tab (composer pop-out).
|
||||
// The reload epoch keys the CONTENT, not this layer: a
|
||||
// Reload remounts the contribution (effects re-run, state
|
||||
// resets) while the layer — and every other tab — stays.
|
||||
<PaneGroupContext.Provider value={node.id}>
|
||||
<PaneVisibleContext.Provider value={isActive}>
|
||||
<ContribBoundary id={pane.id}>{pane.render()}</ContribBoundary>
|
||||
<ContribBoundary id={pane.id} key={paneEpochs[paneId] ?? 0}>
|
||||
{pane.render()}
|
||||
</ContribBoundary>
|
||||
</PaneVisibleContext.Provider>
|
||||
</PaneGroupContext.Provider>
|
||||
) : (
|
||||
|
||||
@@ -450,6 +450,22 @@ export function treeTabCloseTargets(paneId: string): { all: number; others: numb
|
||||
return { all: others.length + (isUncloseablePane(paneId) ? 0 : 1), others: others.length, right: right.length }
|
||||
}
|
||||
|
||||
/**
|
||||
* RELOAD — a pane's remount counter, the tab menu's Reload (browser parity:
|
||||
* right-click a tab, reload what's in it). The zone renderer keys a pane's
|
||||
* body layer on its epoch, so bumping it unmounts the contribution and mounts
|
||||
* it fresh — data effects re-run, measurements are retaken — while the layout
|
||||
* tree, the tab's position, and every other tab stay exactly as they were.
|
||||
* Absent until a pane is first reloaded (no key churn on a normal boot).
|
||||
*/
|
||||
export const $treePaneEpochs = atom<Readonly<Record<string, number>>>({})
|
||||
|
||||
export function reloadTreePane(paneId: string): void {
|
||||
const epochs = $treePaneEpochs.get()
|
||||
|
||||
$treePaneEpochs.set({ ...epochs, [paneId]: (epochs[paneId] ?? 0) + 1 })
|
||||
}
|
||||
|
||||
/** Close a tab the way its kind expects: a tool panel leaves the strip (and
|
||||
* syncs its toggle), everything else routes through its owning Close. */
|
||||
export function closeTabPane(paneId: string) {
|
||||
@@ -1577,7 +1593,7 @@ export function resetLayoutTree() {
|
||||
}
|
||||
|
||||
// Dev hook for automation.
|
||||
if (import.meta.env.DEV && typeof window !== 'undefined') {
|
||||
if ((import.meta.env.DEV || import.meta.env.VITE_PERF_PROBE === '1') && typeof window !== 'undefined') {
|
||||
;(window as unknown as Record<string, unknown>).__HERMES_LAYOUT_TREE__ = {
|
||||
close: closeTreePane,
|
||||
dismissed: () => $dismissedPanes.get(),
|
||||
|
||||
@@ -13,7 +13,10 @@ import {
|
||||
$petRoam,
|
||||
$petRoamDir,
|
||||
clearPetUnread,
|
||||
hasPetSpriteForMeta,
|
||||
mergePetInfoMeta,
|
||||
type PetInfo,
|
||||
type PetInfoMeta,
|
||||
petProfile,
|
||||
setPetInfo
|
||||
} from '@/store/pet'
|
||||
@@ -39,25 +42,6 @@ interface Point {
|
||||
y: number
|
||||
}
|
||||
|
||||
interface PetInfoMeta {
|
||||
enabled: boolean
|
||||
slug?: string
|
||||
displayName?: string
|
||||
scale?: number
|
||||
spritesheetRevision?: string
|
||||
}
|
||||
|
||||
function samePetRevision(info: PetInfo, meta: PetInfoMeta): boolean {
|
||||
return (
|
||||
info.enabled &&
|
||||
Boolean(info.spritesheetBase64) &&
|
||||
info.slug === meta.slug &&
|
||||
info.displayName === meta.displayName &&
|
||||
info.scale === meta.scale &&
|
||||
info.spritesheetRevision === meta.spritesheetRevision
|
||||
)
|
||||
}
|
||||
|
||||
// Keep a w×h box fully inside the viewport. Pre-pet-load callers pass a nominal
|
||||
// size; the live size flows in once `info` arrives.
|
||||
function clampPoint(x: number, y: number, w: number, h: number): Point {
|
||||
@@ -161,7 +145,7 @@ export function FloatingPet() {
|
||||
// pet.changed already carries the meta payload — an enabled=false
|
||||
// broadcast clears the mascot with zero round-trips, and an unchanged
|
||||
// revision (scale-only move still changes the sig) short-circuits below
|
||||
// via samePetRevision.
|
||||
// via hasPetSpriteForMeta + mergePetInfoMeta.
|
||||
if (changeEventsAvailable && petChange.tick > 0 && petChange.meta?.enabled === false) {
|
||||
setPetInfo({ enabled: false })
|
||||
|
||||
@@ -184,7 +168,15 @@ export function FloatingPet() {
|
||||
return
|
||||
}
|
||||
|
||||
if (samePetRevision($petInfo.get(), meta)) {
|
||||
const current = $petInfo.get()
|
||||
|
||||
if (hasPetSpriteForMeta(current, meta)) {
|
||||
const merged = mergePetInfoMeta(current, meta)
|
||||
|
||||
if (merged !== current) {
|
||||
setPetInfo(merged)
|
||||
}
|
||||
|
||||
return
|
||||
}
|
||||
} catch {
|
||||
@@ -223,7 +215,14 @@ export function FloatingPet() {
|
||||
// so no timer. Legacy backend: the historical poll.
|
||||
const timer = changeEventsAvailable
|
||||
? null
|
||||
: window.setInterval(() => void pull(), active ? PET_ACTIVE_REFRESH_MS : PET_POLL_MS)
|
||||
: window.setInterval(
|
||||
() => {
|
||||
if (document.visibilityState === 'visible') {
|
||||
void pull()
|
||||
}
|
||||
},
|
||||
active ? PET_ACTIVE_REFRESH_MS : PET_POLL_MS
|
||||
)
|
||||
|
||||
return () => {
|
||||
cancelled = true
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
import { act, render, screen } from '@testing-library/react'
|
||||
import { Profiler, type ProfilerOnRenderCallback } from 'react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { PaneVisibleContext } from '@/components/pane-shell/pane-visibility'
|
||||
|
||||
import { GlyphSpinner } from './glyph-spinner'
|
||||
|
||||
describe('GlyphSpinner', () => {
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers()
|
||||
vi.spyOn(globalThis.document, 'hasFocus').mockReturnValue(true)
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllTimers()
|
||||
vi.restoreAllMocks()
|
||||
vi.useRealTimers()
|
||||
})
|
||||
|
||||
it('advances its glyph without an update-phase React commit', () => {
|
||||
let updateCommits = 0
|
||||
|
||||
const onRender: ProfilerOnRenderCallback = (_id, phase) => {
|
||||
if (phase !== 'mount') {
|
||||
updateCommits += 1
|
||||
}
|
||||
}
|
||||
|
||||
render(
|
||||
<Profiler id="glyph-spinner" onRender={onRender}>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</Profiler>
|
||||
)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(status.textContent).toBe('⠋')
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
|
||||
expect(status.textContent).toBe('⠙')
|
||||
expect(updateCommits).toBe(0)
|
||||
})
|
||||
|
||||
it('does not tick while its kept-alive pane is hidden', () => {
|
||||
const { rerender } = render(
|
||||
<PaneVisibleContext.Provider value={false}>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</PaneVisibleContext.Provider>
|
||||
)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
|
||||
expect(status.textContent).toBe('⠋')
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
rerender(
|
||||
<PaneVisibleContext.Provider value>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</PaneVisibleContext.Provider>
|
||||
)
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).toBe('⠙')
|
||||
|
||||
rerender(
|
||||
<PaneVisibleContext.Provider value={false}>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</PaneVisibleContext.Provider>
|
||||
)
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
})
|
||||
|
||||
it('suspends animation while the Desktop window is inactive', () => {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => window.dispatchEvent(new Event('blur')))
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
|
||||
act(() => window.dispatchEvent(new Event('focus')))
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).not.toBe(frozen)
|
||||
})
|
||||
|
||||
it('suspends animation while the Electron window is minimized or hidden, then resumes on restore', () => {
|
||||
let windowStateCallback: ((payload: { isMinimized?: boolean; isVisible?: boolean }) => void) | null = null
|
||||
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
value: {
|
||||
onWindowStateChanged: vi.fn((callback: typeof windowStateCallback) => {
|
||||
windowStateCallback = callback
|
||||
|
||||
return () => {
|
||||
if (windowStateCallback === callback) {
|
||||
windowStateCallback = null
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
})
|
||||
|
||||
try {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(windowStateCallback).not.toBeNull()
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => windowStateCallback?.({ isMinimized: true, isVisible: false }))
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
|
||||
act(() => windowStateCallback?.({ isMinimized: false, isVisible: true }))
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).not.toBe(frozen)
|
||||
} finally {
|
||||
delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop
|
||||
}
|
||||
})
|
||||
|
||||
it('suspends animation while the document is hidden', () => {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
Object.defineProperty(document, 'visibilityState', { configurable: true, value: 'hidden' })
|
||||
|
||||
try {
|
||||
act(() => document.dispatchEvent(new Event('visibilitychange')))
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
|
||||
Object.defineProperty(document, 'visibilityState', { configurable: true, value: 'visible' })
|
||||
act(() => document.dispatchEvent(new Event('visibilitychange')))
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).not.toBe(frozen)
|
||||
} finally {
|
||||
Object.defineProperty(document, 'visibilityState', { configurable: true, value: 'visible' })
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -1,7 +1,8 @@
|
||||
import { useEffect, useState } from 'react'
|
||||
import { useEffect, useRef } from 'react'
|
||||
import spinners, { type BrailleSpinnerName as SpinnerName } from 'unicode-animations'
|
||||
|
||||
import { usePaneVisible } from '@/components/pane-shell/pane-visibility'
|
||||
import { createRendererLoopPauseController } from '@/lib/renderer-loop-pause'
|
||||
import { cn } from '@/lib/utils'
|
||||
|
||||
export type { SpinnerName }
|
||||
@@ -43,29 +44,66 @@ interface GlyphSpinnerProps {
|
||||
*/
|
||||
export function GlyphSpinner({ ariaLabel = 'Loading', className, spinner = 'braille' }: GlyphSpinnerProps) {
|
||||
const spin = FRAMES_BY_NAME[spinner] ?? FRAMES_BY_NAME.braille!
|
||||
const [frame, setFrame] = useState(0)
|
||||
const glyphRef = useRef<HTMLSpanElement>(null)
|
||||
// Pause when this surface is a hidden (kept-alive) tab: N mounted tabs each
|
||||
// ticking a setInterval + setState burn CPU for pixels nobody can see.
|
||||
// ticking a setInterval burns CPU for pixels nobody can see.
|
||||
const visible = usePaneVisible()
|
||||
|
||||
useEffect(() => {
|
||||
if (!visible) {
|
||||
const glyph = glyphRef.current
|
||||
|
||||
if (!visible || !glyph) {
|
||||
return
|
||||
}
|
||||
|
||||
setFrame(0)
|
||||
const id = window.setInterval(() => setFrame(f => (f + 1) % spin.frames.length), spin.interval)
|
||||
let frame = 0
|
||||
let timer: number | undefined
|
||||
let pauseController: ReturnType<typeof createRendererLoopPauseController> | undefined
|
||||
glyph.textContent = spin.frames[frame]
|
||||
|
||||
return () => window.clearInterval(id)
|
||||
const stopAnimation = () => {
|
||||
if (timer === undefined) {
|
||||
return
|
||||
}
|
||||
|
||||
window.clearInterval(timer)
|
||||
timer = undefined
|
||||
}
|
||||
|
||||
const syncAnimation = () => {
|
||||
if (pauseController?.isPaused()) {
|
||||
stopAnimation()
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if (timer !== undefined) {
|
||||
return
|
||||
}
|
||||
|
||||
timer = window.setInterval(() => {
|
||||
frame = (frame + 1) % spin.frames.length
|
||||
glyph.textContent = spin.frames[frame]
|
||||
}, spin.interval)
|
||||
}
|
||||
|
||||
pauseController = createRendererLoopPauseController(syncAnimation)
|
||||
syncAnimation()
|
||||
|
||||
return () => {
|
||||
pauseController.dispose()
|
||||
stopAnimation()
|
||||
}
|
||||
}, [spin, visible])
|
||||
|
||||
return (
|
||||
<span
|
||||
aria-label={ariaLabel}
|
||||
className={cn('inline-flex items-center justify-center font-mono leading-none tabular-nums', className)}
|
||||
ref={glyphRef}
|
||||
role="status"
|
||||
>
|
||||
{spin.frames[frame]}
|
||||
{spin.frames[0]}
|
||||
</span>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
import { act, cleanup, render } from '@testing-library/react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { StatusPulse } from './status-pulse'
|
||||
|
||||
interface PlayedAnimation {
|
||||
cancel: ReturnType<typeof vi.fn>
|
||||
keyframes: Keyframe[]
|
||||
options: KeyframeAnimationOptions
|
||||
}
|
||||
|
||||
interface WindowStatePayload {
|
||||
isMinimized?: boolean
|
||||
isVisible?: boolean
|
||||
}
|
||||
|
||||
let windowStateCallback: ((payload: WindowStatePayload) => void) | undefined
|
||||
|
||||
function installMatchMedia(matches: boolean) {
|
||||
vi.stubGlobal(
|
||||
'matchMedia',
|
||||
vi.fn(() => ({
|
||||
addEventListener: vi.fn(),
|
||||
matches,
|
||||
media: '(prefers-reduced-motion: reduce)',
|
||||
onchange: null,
|
||||
removeEventListener: vi.fn()
|
||||
}))
|
||||
)
|
||||
}
|
||||
|
||||
function installWindowStateBridge() {
|
||||
const off = vi.fn()
|
||||
|
||||
window.hermesDesktop = {
|
||||
onWindowStateChanged: vi.fn(callback => {
|
||||
windowStateCallback = callback
|
||||
|
||||
return off
|
||||
})
|
||||
} as unknown as typeof window.hermesDesktop
|
||||
|
||||
return off
|
||||
}
|
||||
|
||||
describe('StatusPulse', () => {
|
||||
const played: PlayedAnimation[] = []
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers()
|
||||
vi.spyOn(window.document, 'hasFocus').mockReturnValue(true)
|
||||
installMatchMedia(false)
|
||||
installWindowStateBridge()
|
||||
Object.defineProperty(HTMLElement.prototype, 'animate', {
|
||||
configurable: true,
|
||||
value: (keyframes: Keyframe[], options: KeyframeAnimationOptions) => {
|
||||
const animation = {
|
||||
cancel: vi.fn(),
|
||||
keyframes,
|
||||
options
|
||||
}
|
||||
|
||||
played.push(animation)
|
||||
|
||||
return animation as unknown as Animation
|
||||
},
|
||||
writable: true
|
||||
})
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
played.length = 0
|
||||
windowStateCallback = undefined
|
||||
Reflect.deleteProperty(HTMLElement.prototype, 'animate')
|
||||
delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop
|
||||
vi.useRealTimers()
|
||||
vi.unstubAllGlobals()
|
||||
vi.restoreAllMocks()
|
||||
})
|
||||
|
||||
it('plays a finite ping and sleeps between pulses', () => {
|
||||
render(<StatusPulse kind="ping" opacity={0.7} />)
|
||||
|
||||
expect(played).toHaveLength(1)
|
||||
expect(played[0]?.keyframes).toEqual([
|
||||
{ opacity: 0.7, transform: 'scale(1)' },
|
||||
{ opacity: 0, transform: 'scale(2)' }
|
||||
])
|
||||
expect(played[0]?.options).toMatchObject({ duration: 400, iterations: 1 })
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(4_999))
|
||||
expect(played).toHaveLength(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(1))
|
||||
expect(played).toHaveLength(2)
|
||||
})
|
||||
|
||||
it('cancels scheduled work while minimized and restarts once visible', () => {
|
||||
const offWindowState = installWindowStateBridge()
|
||||
const mounted = render(<StatusPulse kind="opacity" />)
|
||||
|
||||
expect(played).toHaveLength(1)
|
||||
|
||||
act(() => windowStateCallback?.({ isMinimized: true, isVisible: false }))
|
||||
|
||||
expect(played[0]?.cancel).toHaveBeenCalledTimes(1)
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
act(() => vi.advanceTimersByTime(5_000))
|
||||
expect(played).toHaveLength(1)
|
||||
|
||||
act(() => windowStateCallback?.({ isMinimized: false, isVisible: true }))
|
||||
expect(played).toHaveLength(2)
|
||||
|
||||
mounted.unmount()
|
||||
expect(played[1]?.cancel).toHaveBeenCalledTimes(1)
|
||||
expect(offWindowState).toHaveBeenCalledTimes(1)
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
})
|
||||
|
||||
it('stays static when reduced motion is requested', () => {
|
||||
installMatchMedia(true)
|
||||
|
||||
render(<StatusPulse kind="opacity" />)
|
||||
|
||||
expect(played).toHaveLength(0)
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,142 @@
|
||||
import { type ComponentProps, useEffect, useRef } from 'react'
|
||||
|
||||
import { createRendererLoopPauseController } from '@/lib/renderer-loop-pause'
|
||||
|
||||
const PULSE_DURATION_MS = 400
|
||||
const PULSE_PERIOD_MS = 5_000
|
||||
|
||||
// One pause controller + one period timer shared by every StatusPulse
|
||||
// instance. A sidebar can show dozens of pulsing dots at once; per-instance
|
||||
// controllers would mean N×(document/window/bridge) listeners and N
|
||||
// unsynchronized 5s wakes. Ref-counted: the controller and timer exist only
|
||||
// while at least one pulse is mounted, and all pulses play in one aligned
|
||||
// wake so the renderer sleeps between beats.
|
||||
type PulseSubscriber = { play: () => void; cancel: () => void }
|
||||
|
||||
const pulseSubscribers = new Set<PulseSubscriber>()
|
||||
let sharedPauseController: ReturnType<typeof createRendererLoopPauseController> | null = null
|
||||
let sharedTimer = 0
|
||||
|
||||
const stopSharedTimer = () => {
|
||||
if (sharedTimer !== 0) {
|
||||
window.clearTimeout(sharedTimer)
|
||||
sharedTimer = 0
|
||||
}
|
||||
}
|
||||
|
||||
const beat = () => {
|
||||
sharedTimer = 0
|
||||
|
||||
if (sharedPauseController?.isPaused() || pulseSubscribers.size === 0) {
|
||||
return
|
||||
}
|
||||
|
||||
for (const subscriber of pulseSubscribers) {
|
||||
subscriber.play()
|
||||
}
|
||||
|
||||
sharedTimer = window.setTimeout(beat, PULSE_PERIOD_MS)
|
||||
}
|
||||
|
||||
const handleSharedPauseChange = () => {
|
||||
stopSharedTimer()
|
||||
|
||||
if (sharedPauseController?.isPaused()) {
|
||||
// Minimized/hidden: cancel in-flight animations so the compositor can
|
||||
// sleep immediately instead of finishing a pulse nobody sees.
|
||||
for (const subscriber of pulseSubscribers) {
|
||||
subscriber.cancel()
|
||||
}
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
beat()
|
||||
}
|
||||
|
||||
const subscribePulse = (subscriber: PulseSubscriber): (() => void) => {
|
||||
pulseSubscribers.add(subscriber)
|
||||
|
||||
if (!sharedPauseController) {
|
||||
sharedPauseController = createRendererLoopPauseController(handleSharedPauseChange)
|
||||
}
|
||||
|
||||
// First subscriber (or a new one joining mid-sleep): play immediately and
|
||||
// start the beat. Later joiners just wait for the next aligned beat.
|
||||
if (sharedTimer === 0 && !sharedPauseController.isPaused()) {
|
||||
subscriber.play()
|
||||
sharedTimer = window.setTimeout(beat, PULSE_PERIOD_MS)
|
||||
}
|
||||
|
||||
return () => {
|
||||
pulseSubscribers.delete(subscriber)
|
||||
|
||||
if (pulseSubscribers.size === 0) {
|
||||
stopSharedTimer()
|
||||
sharedPauseController?.dispose()
|
||||
sharedPauseController = null
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export interface StatusPulseProps extends Omit<ComponentProps<'span'>, 'children' | 'ref'> {
|
||||
kind: 'opacity' | 'ping'
|
||||
opacity?: number
|
||||
}
|
||||
|
||||
/**
|
||||
* A finite status pulse with a real sleep between plays.
|
||||
*
|
||||
* Continuous CSS animations keep Chromium producing frames and recalculating
|
||||
* styles for an otherwise motionless Desktop window. Drive the same visual
|
||||
* cue directly so React stays out of the loop and the renderer/compositor can
|
||||
* sleep between pulses.
|
||||
*/
|
||||
export function StatusPulse({ kind, opacity = 1, ...props }: StatusPulseProps) {
|
||||
const ref = useRef<HTMLSpanElement>(null)
|
||||
|
||||
useEffect(() => {
|
||||
const element = ref.current
|
||||
|
||||
if (
|
||||
!element ||
|
||||
typeof element.animate !== 'function' ||
|
||||
window.matchMedia?.('(prefers-reduced-motion: reduce)').matches
|
||||
) {
|
||||
return
|
||||
}
|
||||
|
||||
let animation: Animation | null = null
|
||||
|
||||
const play = () => {
|
||||
animation?.cancel()
|
||||
animation = element.animate(
|
||||
kind === 'ping'
|
||||
? [
|
||||
{ opacity, transform: 'scale(1)' },
|
||||
{ opacity: 0, transform: 'scale(2)' }
|
||||
]
|
||||
: [{ opacity: 1 }, { opacity: 0.5 }, { opacity: 1 }],
|
||||
{
|
||||
duration: PULSE_DURATION_MS,
|
||||
easing: kind === 'ping' ? 'cubic-bezier(0, 0, 0.2, 1)' : 'ease-in-out',
|
||||
iterations: 1
|
||||
}
|
||||
)
|
||||
}
|
||||
|
||||
const cancel = () => {
|
||||
animation?.cancel()
|
||||
animation = null
|
||||
}
|
||||
|
||||
const unsubscribe = subscribePulse({ play, cancel })
|
||||
|
||||
return () => {
|
||||
unsubscribe()
|
||||
cancel()
|
||||
}
|
||||
}, [kind, opacity])
|
||||
|
||||
return <span {...props} ref={ref} />
|
||||
}
|
||||
@@ -29,6 +29,7 @@ import './render-counter'
|
||||
// app under REAL sessions instead of a synthetic scenario's toy transcripts.
|
||||
// window.__PERF_LIVE__.on() in the console, then just use the app.
|
||||
import './perf-live'
|
||||
import './right-pane-probe'
|
||||
|
||||
import { watchSessionAtoms } from './watched-atoms'
|
||||
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
export type RightPanePerfEvent =
|
||||
'project-tree-render' | 'project-tree-row-render' | 'terminal-fit-active' | 'terminal-fit-hidden' | 'terminal-measure'
|
||||
|
||||
export interface RightPanePerfSnapshot {
|
||||
counts: Record<RightPanePerfEvent, number>
|
||||
details: Partial<Record<RightPanePerfEvent, Record<string, number>>>
|
||||
rows: Record<string, number>
|
||||
}
|
||||
|
||||
declare global {
|
||||
interface Window {
|
||||
__RIGHT_PANE_PERF__?: {
|
||||
clear: () => void
|
||||
mark: (event: RightPanePerfEvent, detail?: string) => void
|
||||
snapshot: () => RightPanePerfSnapshot
|
||||
start: () => void
|
||||
stop: () => void
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Tiny production-safe callsite; the recorder exists only in dev/perf builds. */
|
||||
export function markRightPanePerf(event: RightPanePerfEvent, detail?: string): void {
|
||||
window.__RIGHT_PANE_PERF__?.mark(event, detail)
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
import type { RightPanePerfEvent, RightPanePerfSnapshot } from './right-pane-events'
|
||||
|
||||
const eventNames: RightPanePerfEvent[] = [
|
||||
'project-tree-render',
|
||||
'project-tree-row-render',
|
||||
'terminal-fit-active',
|
||||
'terminal-fit-hidden',
|
||||
'terminal-measure'
|
||||
]
|
||||
|
||||
const blankCounts = (): Record<RightPanePerfEvent, number> =>
|
||||
Object.fromEntries(eventNames.map(name => [name, 0])) as Record<RightPanePerfEvent, number>
|
||||
|
||||
if (typeof window !== 'undefined' && !window.__RIGHT_PANE_PERF__) {
|
||||
let recording = false
|
||||
let counts = blankCounts()
|
||||
let details: RightPanePerfSnapshot['details'] = {}
|
||||
let rows: Record<string, number> = {}
|
||||
|
||||
window.__RIGHT_PANE_PERF__ = {
|
||||
clear: () => {
|
||||
counts = blankCounts()
|
||||
details = {}
|
||||
rows = {}
|
||||
},
|
||||
mark: (event, detail) => {
|
||||
if (!recording) {
|
||||
return
|
||||
}
|
||||
|
||||
counts[event] += 1
|
||||
|
||||
if (detail) {
|
||||
const eventDetails = details[event] ?? {}
|
||||
eventDetails[detail] = (eventDetails[detail] ?? 0) + 1
|
||||
details[event] = eventDetails
|
||||
|
||||
if (event === 'project-tree-row-render') {
|
||||
rows[detail] = (rows[detail] ?? 0) + 1
|
||||
}
|
||||
}
|
||||
},
|
||||
snapshot: (): RightPanePerfSnapshot => ({
|
||||
counts: { ...counts },
|
||||
details: Object.fromEntries(Object.entries(details).map(([event, eventDetails]) => [event, { ...eventDetails }])),
|
||||
rows: { ...rows }
|
||||
}),
|
||||
start: () => {
|
||||
counts = blankCounts()
|
||||
details = {}
|
||||
rows = {}
|
||||
recording = true
|
||||
},
|
||||
stop: () => {
|
||||
recording = false
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
import { JsonRpcGatewayClient } from '@hermes/shared'
|
||||
|
||||
import { reconnectBackoffDelayMs } from '@/lib/reconnect-backoff'
|
||||
import type {
|
||||
ActionResponse,
|
||||
ActionStatusResponse,
|
||||
@@ -349,8 +350,12 @@ export function pluginSocket(pluginId: string, path: string, onMessage: (data: u
|
||||
socket = null
|
||||
|
||||
if (!disposed) {
|
||||
// Full-jitter exponential backoff: same rationale as the gateway
|
||||
// socket reconnect loops — an immediate-retry loop across many
|
||||
// desktop clients floods the gateway with connection attempts
|
||||
// during a restart.
|
||||
window.setTimeout(() => void connect(), reconnectBackoffDelayMs(attempt, { baseDelayMs: 500, capMs: 30_000 }))
|
||||
attempt += 1
|
||||
window.setTimeout(() => void connect(), Math.min(30_000, 1_000 * 2 ** attempt))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2236,6 +2236,7 @@ export const ar = defineLocale({
|
||||
closeRunningBody:
|
||||
'هذه المحادثة ما زالت تعمل (أو تنتظر إدخالك). إغلاق التبويب يخفيها فقط — ستحتفظ الجلسة بتقدمها ويمكن إعادة فتحها من الشريط الجانبي.',
|
||||
closeRunningConfirm: 'إغلاق التبويب',
|
||||
reload: 'إعادة التحميل',
|
||||
closeOthers: 'إغلاق الأخرى',
|
||||
closeToRight: 'إغلاق ما على اليمين',
|
||||
closeAll: 'إغلاق الكل',
|
||||
|
||||
@@ -2663,6 +2663,7 @@ export const en: Translations = {
|
||||
closeRunningBody:
|
||||
'This chat is still working (or waiting on your input). Closing the tab hides it — the session keeps its progress and can be reopened from the sidebar.',
|
||||
closeRunningConfirm: 'Close tab',
|
||||
reload: 'Reload',
|
||||
closeOthers: 'Close others',
|
||||
closeToRight: 'Close to the right',
|
||||
closeAll: 'Close all',
|
||||
|
||||
@@ -2489,6 +2489,7 @@ export const ja = defineLocale({
|
||||
hideHeader: 'ヘッダーを隠す',
|
||||
minimize: '最小化',
|
||||
restore: '復元',
|
||||
reload: '再読み込み',
|
||||
closeOthers: '他を閉じる',
|
||||
closeToRight: '右側を閉じる',
|
||||
closeAll: 'すべて閉じる',
|
||||
|
||||
@@ -2262,6 +2262,7 @@ export interface Translations {
|
||||
closeRunningTitle: string
|
||||
closeRunningBody: string
|
||||
closeRunningConfirm: string
|
||||
reload: string
|
||||
closeOthers: string
|
||||
closeToRight: string
|
||||
closeAll: string
|
||||
|
||||
@@ -2409,6 +2409,7 @@ export const zhHant = defineLocale({
|
||||
hideHeader: '隱藏標題列',
|
||||
minimize: '最小化',
|
||||
restore: '還原',
|
||||
reload: '重新載入',
|
||||
closeOthers: '關閉其他',
|
||||
closeToRight: '關閉右側',
|
||||
closeAll: '全部關閉',
|
||||
|
||||
@@ -2841,6 +2841,7 @@ export const zh: Translations = {
|
||||
closeRunningTitle: '关闭正在运行的标签?',
|
||||
closeRunningBody: '此对话仍在运行(或正在等待你的输入)。关闭标签只会隐藏它——会话将保留进度,可从侧边栏重新打开。',
|
||||
closeRunningConfirm: '关闭标签',
|
||||
reload: '重新加载',
|
||||
closeOthers: '关闭其他',
|
||||
closeToRight: '关闭右侧',
|
||||
closeAll: '全部关闭',
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { extractEmbeddedImages, extractImageRefs, textWithoutImageRefs } from './embedded-images'
|
||||
import { extractEmbeddedImages, extractImageRefs } from './embedded-images'
|
||||
|
||||
const SAMPLE_PNG_DATA_URL = 'data:image/png;base64,' + 'A'.repeat(120)
|
||||
|
||||
@@ -43,28 +43,6 @@ describe('extractEmbeddedImages', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('textWithoutImageRefs', () => {
|
||||
it('leaves plain text untouched', () => {
|
||||
expect(textWithoutImageRefs('just a question')).toBe('just a question')
|
||||
})
|
||||
|
||||
it('strips a single leading @image directive line', () => {
|
||||
expect(textWithoutImageRefs('@image:/tmp/cat.png\nwhat is this?')).toBe('what is this?')
|
||||
})
|
||||
|
||||
it('strips multiple @image directive lines and trims', () => {
|
||||
const input = '@image:/tmp/a.png\n@image:/tmp/b.png\n describe both '
|
||||
|
||||
expect(textWithoutImageRefs(input)).toBe('describe both')
|
||||
})
|
||||
|
||||
it('does not treat an inline @image mention as a directive line', () => {
|
||||
// Only full-line leading directives are stripped, matching the gateway's
|
||||
// persist-time rewrite. A bare mention mid-prose is preserved.
|
||||
expect(textWithoutImageRefs('see @image:/tmp/cat.png here')).toBe('see @image:/tmp/cat.png here')
|
||||
})
|
||||
})
|
||||
|
||||
describe('extractImageRefs', () => {
|
||||
it('returns the text untouched and no refs when there are no directives', () => {
|
||||
expect(extractImageRefs('a normal prompt')).toEqual({ cleanedText: 'a normal prompt', refs: [] })
|
||||
|
||||
@@ -165,20 +165,15 @@ export function textWithoutEmbeddedImages(text: string): string {
|
||||
// (see tui_gateway/server.py's persist-time rewrite), prepended before the
|
||||
// user's own text. The composer's own optimistic/local turn never carries
|
||||
// this prefix — it keeps the attachment as separate `attachmentRefs`
|
||||
// metadata, not inline text. Comparing raw chatMessageText between the
|
||||
// optimistic turn and the authoritative (persisted) turn therefore always
|
||||
// mismatches whenever an image was attached, which defeats the "is this the
|
||||
// same turn" checks in preserveLocalPendingTurnMessages / appendLiveSessionProjection
|
||||
// and re-appends the optimistic row as if it were a distinct, unconfirmed
|
||||
// turn — a duplicated user bubble. Strip the directive line(s) before any
|
||||
// such equality comparison so both sides reduce to the same visible text.
|
||||
// metadata, not inline text. The turn-equality comparisons in
|
||||
// preserveLocalPendingTurnMessages / appendLiveSessionProjection strip ALL
|
||||
// reference-directive lines (not just images) via
|
||||
// `textWithoutReferenceLines` in components/assistant-ui/reference-kinds.ts;
|
||||
// IMAGE_REF_LINE_RE remains here for extractImageRefs below, which moves the
|
||||
// image directives into attachmentRefs metadata.
|
||||
const IMAGE_REF_LINE_RE = /^@image:[^\n]*\n?/gm
|
||||
|
||||
export function textWithoutImageRefs(text: string): string {
|
||||
return text.replace(IMAGE_REF_LINE_RE, '').trim()
|
||||
}
|
||||
|
||||
// Same directive lines as textWithoutImageRefs, but keeps them instead of
|
||||
// Same directive lines as IMAGE_REF_LINE_RE, but keeps them instead of
|
||||
// discarding — used when converting persisted server messages into
|
||||
// ChatMessage/ThreadMessageLike shape, where `@image:<path>` refs need to
|
||||
// move from inline text into the `attachmentRefs` metadata field (mirroring
|
||||
|
||||
@@ -243,9 +243,13 @@ export function useIncrementalExternalStoreRuntime<T extends ThreadMessage>(
|
||||
): AssistantRuntime {
|
||||
const [runtime] = useState(() => new IncrementalExternalStoreRuntimeCore(store as ExternalStoreAdapter))
|
||||
|
||||
// Re-sync the adapter only when it actually changes — a dep-less effect ran
|
||||
// on EVERY render of the chat surface. `__internal_setAdapter` early-exits
|
||||
// when the store is unchanged, so gating on [runtime, store] is behavior-
|
||||
// preserving while skipping the per-render call entirely.
|
||||
useEffect(() => {
|
||||
runtime.setAdapter(store as ExternalStoreAdapter)
|
||||
})
|
||||
}, [runtime, store])
|
||||
|
||||
const { modelContext } = useRuntimeAdapters() ?? {}
|
||||
|
||||
|
||||
@@ -179,7 +179,9 @@ describe('recoverInFlightTurnJournal', () => {
|
||||
it('overlays the backend text-only projection instead of dropping local tool progress', () => {
|
||||
// Sweeper regression on #44339: a backend `inflight` assistant snapshot
|
||||
// (text only) used to mark the richer local tail "caught up" and delete
|
||||
// locally recorded tool calls.
|
||||
// locally recorded tool calls. After #76444, longer text wins only when it
|
||||
// is a strict extension of the journal answer (flat thinking dumps must
|
||||
// not replace structured answer text).
|
||||
journalEntry([
|
||||
user('u1', 'do the thing'),
|
||||
assistantWithTool('assistant-stream-old', 'local part', { pending: true })
|
||||
@@ -187,7 +189,7 @@ describe('recoverInFlightTurnJournal', () => {
|
||||
|
||||
const base = [
|
||||
user('db-u1', 'do the thing'),
|
||||
assistant('assistant-stream-rt9', 'longer partial text from the backend snapshot', { pending: true })
|
||||
assistant('assistant-stream-rt9', 'local part and more from the backend snapshot', { pending: true })
|
||||
]
|
||||
|
||||
const result = recoverInFlightTurnJournal('stored-1', base, { keepPending: true })
|
||||
@@ -200,13 +202,32 @@ describe('recoverInFlightTurnJournal', () => {
|
||||
// Keeps the BASE projection row id so live deltas keep landing on it.
|
||||
expect(merged.id).toBe('assistant-stream-rt9')
|
||||
expect(result.streamId).toBe('assistant-stream-rt9')
|
||||
// Journal structure survives; the longer backend text wins.
|
||||
// Journal structure survives; strict-extension backend text wins.
|
||||
expect(merged.parts[0]).toMatchObject({ type: 'tool-call', toolName: 'terminal' })
|
||||
expect(merged.parts[1]).toMatchObject({ type: 'text', text: 'longer partial text from the backend snapshot' })
|
||||
expect(merged.parts[1]).toMatchObject({ type: 'text', text: 'local part and more from the backend snapshot' })
|
||||
// Still in flight — the journal must NOT be cleared.
|
||||
expect(readInFlightTurnJournal('stored-1')).not.toBeNull()
|
||||
})
|
||||
|
||||
it('keeps journal answer text when a longer flat dump is not a strict extension (#76444)', () => {
|
||||
journalEntry([user('u1', 'do the thing'), assistantWithTool('assistant-stream-old', 'partial', { pending: true })])
|
||||
|
||||
const base = [
|
||||
user('db-u1', 'do the thing'),
|
||||
assistant(
|
||||
'assistant-stream-rt9',
|
||||
'thinking chatter\nRan terminal\npartial and unrelated dump longer than answer',
|
||||
{ pending: true }
|
||||
)
|
||||
]
|
||||
|
||||
const result = recoverInFlightTurnJournal('stored-1', base, { keepPending: true })
|
||||
const merged = result.messages.at(-1)!
|
||||
|
||||
expect(merged.parts[0]).toMatchObject({ type: 'tool-call', toolName: 'terminal' })
|
||||
expect(merged.parts[1]).toMatchObject({ type: 'text', text: 'partial' })
|
||||
})
|
||||
|
||||
it('keeps the journal text when it is longer than the projection text', () => {
|
||||
journalEntry([
|
||||
user('u1', 'do the thing'),
|
||||
|
||||
@@ -254,6 +254,10 @@ function assistantTextLength(message: ChatMessage): number {
|
||||
* BASE row's id so live deltas keep appending to the row the stream handler
|
||||
* already targets.
|
||||
*/
|
||||
function hasStructuralParts(message: ChatMessage): boolean {
|
||||
return message.parts.some(part => part.type === 'reasoning' || part.type === 'tool-call')
|
||||
}
|
||||
|
||||
function overlayProjectionRow(projection: ChatMessage, journalRow: ChatMessage): ChatMessage {
|
||||
// A projected error (retained failed turn) must survive the overlay.
|
||||
const error = journalRow.error ?? projection.error
|
||||
@@ -271,7 +275,20 @@ function overlayProjectionRow(projection: ChatMessage, journalRow: ChatMessage):
|
||||
|
||||
// Backend text is newer than the journal's last throttled write — swap it
|
||||
// into the journal's first text part, keeping tool calls and reasoning.
|
||||
// When the journal already carries structure, only accept a *strict*
|
||||
// extension of the answer text. A longer flat dump that starts with
|
||||
// thinking chatter must not overwrite / insert as answer text (#76444).
|
||||
const projectionText = chatMessageText(projection)
|
||||
const journalText = chatMessageText(journalRow).trim()
|
||||
|
||||
if (hasStructuralParts(journalRow)) {
|
||||
const next = projectionText.trim()
|
||||
|
||||
if (!journalText || !next.startsWith(journalText)) {
|
||||
return merged
|
||||
}
|
||||
}
|
||||
|
||||
const parts: ChatMessagePart[] = []
|
||||
let textReplaced = false
|
||||
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
import { describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { reconnectBackoffDelayMs } from './reconnect-backoff'
|
||||
|
||||
describe('reconnectBackoffDelayMs', () => {
|
||||
it('increases the delay ceiling across consecutive failed attempts', () => {
|
||||
// Pin Math.random so we can read the ceiling directly through the
|
||||
// returned value instead of statistically sampling it.
|
||||
const randomSpy = vi.spyOn(Math, 'random').mockReturnValue(1)
|
||||
|
||||
try {
|
||||
const delays = [0, 1, 2, 3, 4].map(attempt => reconnectBackoffDelayMs(attempt, { baseDelayMs: 300 }))
|
||||
|
||||
expect(delays).toEqual([300, 600, 1200, 2400, 4800])
|
||||
|
||||
for (let i = 1; i < delays.length; i++) {
|
||||
expect(delays[i]).toBeGreaterThan(delays[i - 1])
|
||||
}
|
||||
} finally {
|
||||
randomSpy.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('caps the delay ceiling instead of growing unbounded', () => {
|
||||
const randomSpy = vi.spyOn(Math, 'random').mockReturnValue(1)
|
||||
|
||||
try {
|
||||
// Attempt 10 would be 300 * 2**10 = 307_200ms uncapped — must clamp.
|
||||
expect(reconnectBackoffDelayMs(10, { baseDelayMs: 300, capMs: 15_000 })).toBe(15_000)
|
||||
expect(reconnectBackoffDelayMs(50, { baseDelayMs: 300, capMs: 15_000 })).toBe(15_000)
|
||||
} finally {
|
||||
randomSpy.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('applies full jitter: delay is uniformly within [0, ceiling)', () => {
|
||||
const randomSpy = vi.spyOn(Math, 'random')
|
||||
|
||||
try {
|
||||
randomSpy.mockReturnValue(0)
|
||||
expect(reconnectBackoffDelayMs(3, { baseDelayMs: 300 })).toBe(0)
|
||||
|
||||
randomSpy.mockReturnValue(0.5)
|
||||
expect(reconnectBackoffDelayMs(3, { baseDelayMs: 300 })).toBe(1200)
|
||||
|
||||
randomSpy.mockReturnValue(0.999)
|
||||
expect(reconnectBackoffDelayMs(3, { baseDelayMs: 300 })).toBeCloseTo(2400 * 0.999, 5)
|
||||
} finally {
|
||||
randomSpy.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('resets to the attempt-0 ceiling after a successful connection (caller passes attempt back to 0)', () => {
|
||||
const randomSpy = vi.spyOn(Math, 'random').mockReturnValue(1)
|
||||
|
||||
try {
|
||||
// Simulates: fail, fail, fail (attempt climbs), succeed (caller resets
|
||||
// its counter to 0), fail again — the very next delay must be back at
|
||||
// the base ceiling, not continuing the climb.
|
||||
reconnectBackoffDelayMs(0, { baseDelayMs: 300 })
|
||||
reconnectBackoffDelayMs(1, { baseDelayMs: 300 })
|
||||
const afterSeveralFailures = reconnectBackoffDelayMs(2, { baseDelayMs: 300 })
|
||||
const afterReset = reconnectBackoffDelayMs(0, { baseDelayMs: 300 })
|
||||
|
||||
expect(afterSeveralFailures).toBe(1200)
|
||||
expect(afterReset).toBe(300)
|
||||
} finally {
|
||||
randomSpy.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('treats negative attempt numbers as attempt 0 rather than throwing or returning a negative delay', () => {
|
||||
const randomSpy = vi.spyOn(Math, 'random').mockReturnValue(1)
|
||||
|
||||
try {
|
||||
expect(reconnectBackoffDelayMs(-5, { baseDelayMs: 300 })).toBe(300)
|
||||
} finally {
|
||||
randomSpy.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('uses sane defaults when no options are passed', () => {
|
||||
const randomSpy = vi.spyOn(Math, 'random').mockReturnValue(1)
|
||||
|
||||
try {
|
||||
expect(reconnectBackoffDelayMs(0)).toBe(300)
|
||||
expect(reconnectBackoffDelayMs(100)).toBe(15_000)
|
||||
} finally {
|
||||
randomSpy.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,45 @@
|
||||
/**
|
||||
* Full-jitter exponential backoff for gateway WebSocket reconnects.
|
||||
*
|
||||
* A bare exponential backoff still lets every renderer in a fleet retry in
|
||||
* lockstep — after a gateway restart (e.g. following an update), N desktop
|
||||
* clients that all disconnected within the same instant all wake up and
|
||||
* redial at the same instant too, which is a reconnect storm by another
|
||||
* name. Full jitter (AWS's "Exponential Backoff And Jitter") spreads that
|
||||
* out: each attempt sleeps a *random* duration between 0 and the exponential
|
||||
* ceiling, so retries desynchronize instead of pulsing together.
|
||||
*
|
||||
* https://aws.amazon.com/blogs/architecture/exponential-backoff-and-jitter/
|
||||
*/
|
||||
|
||||
export interface ReconnectBackoffOptions {
|
||||
/** Ceiling on the exponential delay before jitter is applied, in ms. */
|
||||
capMs?: number
|
||||
/** Delay for the first retry (attempt 0) before jitter is applied, in ms. */
|
||||
baseDelayMs?: number
|
||||
}
|
||||
|
||||
const DEFAULT_BASE_DELAY_MS = 300
|
||||
const DEFAULT_CAP_MS = 15_000
|
||||
|
||||
/**
|
||||
* Delay before reconnect attempt number `attempt` (0-indexed: the first
|
||||
* retry after the initial failure is `attempt = 0`). Returns a value in
|
||||
* `[0, min(capMs, baseDelayMs * 2 ** attempt))` — full jitter, not
|
||||
* "equal jitter" or "decorrelated jitter", so it can occasionally return a
|
||||
* very small delay even at a high attempt count. That's intentional: it's
|
||||
* the variant with the best-documented storm-avoidance behavior and no
|
||||
* accumulated-delay state to track between calls.
|
||||
*/
|
||||
export function reconnectBackoffDelayMs(attempt: number, options: ReconnectBackoffOptions = {}): number {
|
||||
const baseDelayMs = options.baseDelayMs ?? DEFAULT_BASE_DELAY_MS
|
||||
const capMs = options.capMs ?? DEFAULT_CAP_MS
|
||||
const safeAttempt = Math.max(0, attempt)
|
||||
|
||||
// 2 ** attempt overflows to Infinity long before it matters (attempt would
|
||||
// need to be ~1024), and Math.min against a finite cap keeps the ceiling
|
||||
// sane regardless, so no extra clamping is needed here.
|
||||
const ceiling = Math.min(capMs, baseDelayMs * 2 ** safeAttempt)
|
||||
|
||||
return Math.random() * ceiling
|
||||
}
|
||||
@@ -10,6 +10,7 @@ import {
|
||||
refreshAllRepoStatuses,
|
||||
refreshRepoStatus,
|
||||
registerRepoStatusCwd,
|
||||
repoChangeKindForPath,
|
||||
repoStatusForCwd
|
||||
} from './coding-status'
|
||||
import { $currentCwd, $selectedStoredSessionId } from './session'
|
||||
@@ -255,3 +256,32 @@ describe('refreshRepoStatus', () => {
|
||||
release?.()
|
||||
})
|
||||
})
|
||||
|
||||
describe('repoChangeKindForPath', () => {
|
||||
it('does not notify a row when only another path changes', () => {
|
||||
$currentCwd.set('/repo')
|
||||
$repoStatusByCwd.set({ '/repo': { ...sampleStatus, files: [] } })
|
||||
const row = repoChangeKindForPath('/repo/a.ts')
|
||||
const listener = vi.fn()
|
||||
const unsubscribe = row.subscribe(listener)
|
||||
|
||||
$repoStatusByCwd.set({
|
||||
'/repo': {
|
||||
...sampleStatus,
|
||||
files: [{ path: 'b.ts', untracked: true } as HermesRepoStatus['files'][number]]
|
||||
}
|
||||
})
|
||||
expect(listener).toHaveBeenCalledTimes(1)
|
||||
|
||||
$repoStatusByCwd.set({
|
||||
'/repo': {
|
||||
...sampleStatus,
|
||||
files: [{ path: 'a.ts', untracked: true } as HermesRepoStatus['files'][number]]
|
||||
}
|
||||
})
|
||||
expect(listener).toHaveBeenCalledTimes(2)
|
||||
expect(listener.mock.calls.at(-1)?.[0]).toBe('added')
|
||||
|
||||
unsubscribe()
|
||||
})
|
||||
})
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user