diff --git a/.env.example b/.env.example index 02985b54a4..78c0108e1c 100644 --- a/.env.example +++ b/.env.example @@ -533,3 +533,12 @@ IMAGE_TOOLS_DEBUG=false # GOOGLE_CHAT_ALLOW_ALL_USERS=false # Set true to skip the allowlist # GOOGLE_CHAT_HOME_CHANNEL= # Default space (spaces/XXXX) for cron delivery # GOOGLE_CHAT_HOME_CHANNEL_NAME= # Display name for the home channel + +# ============================================================================= +# reddit-reading skill (optional) — app-only credentials, NOT a user login +# ============================================================================= +# The skill works with no credentials via Reddit's public feeds (~1 request/minute). +# For faster access with scores and nested comments, register a free "script" app +# at https://www.reddit.com/prefs/apps and paste its id and secret here. +# REDDIT_CLIENT_ID= +# REDDIT_CLIENT_SECRET= diff --git a/.github/workflows/lint.yml b/.github/workflows/lint.yml index bae4299b1b..6b4023ebe5 100644 --- a/.github/workflows/lint.yml +++ b/.github/workflows/lint.yml @@ -184,3 +184,19 @@ jobs: # reverting one commit, so in-tree code must never depend on them. - name: Forbid in-tree use of plugin-compat pointers run: python scripts/check_compat_pointers.py + + # Advisory: dropped public names / methods / test defs vs the PR base, printed into the log. + # A refactor that silently removes a public symbol breaks plugins that import it; the Sep 2026 + # decomposition opened with 1,703 such drops that reviewers had to find by hand. + # Advisory: it never fails the job. The checkout above is depth-1, so deepen both sides until + # a merge-base exists (the script refuses to report a clean diff without one, by design). + - name: Public-surface diff vs base (advisory) + if: github.event_name == 'pull_request' + continue-on-error: true + run: | + git fetch --no-tags --deepen=200 origin "${{ github.base_ref }}" HEAD + for i in 1 2 3; do + git merge-base "origin/${{ github.base_ref }}" HEAD >/dev/null 2>&1 && break + git fetch --no-tags --deepen=1000 origin "${{ github.base_ref }}" HEAD + done + python scripts/ci/check_public_surface.py --base "origin/${{ github.base_ref }}" --head HEAD diff --git a/agent/agent_init.py b/agent/agent_init.py index 3421e26df8..562dfcc328 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -433,6 +433,17 @@ def _finalize_routing(agent, api_mode, credential_pool): with suppress(Exception): agent._get_transport() + # The Nous agent key lives ~1 h. Without the proactive refresher every agent in the process + # discovers expiry reactively, on its own next request, all in the same minute: with 200 + # in-process subagents that was a 401 storm each hour (620 in one run) and the credential + # pool benched the provider for all of them. The gateway and web server start this thread + # at boot; the CLI process (and everything spawned inside it) never did. Idempotent, + # process-wide, daemon. + if agent.provider == "nous": + with suppress(Exception): + from hermes_cli.nous_auth_keepalive import start_nous_auth_keepalive + start_nous_auth_keepalive() + with suppress(Exception): from hermes_cli.model_normalize import ( _AGGREGATOR_PROVIDERS, normalize_model_for_provider @@ -564,6 +575,10 @@ _SESSION_STATE: Dict[str, Any] = { # prefix, kept separately only to place an early cache marker. "_cached_system_prompt": None, "_cached_system_prompt_static": None, + # ``(cwd, workspace_block)`` pinned on the first build: the git/workspace snapshot is + # probed once per session and replayed on every rebuild, so a moving repo can't push the + # prefix-cache divergence point ahead of the volatile band at a compaction boundary. + "_frozen_workspace_snapshot": None, # Whether close() also closes _session_db. False: a caller-supplied handle is usually the # SHARED launch handle; callers handing over a DEDICATED handle set True. "_owns_session_db": False, @@ -2117,10 +2132,11 @@ def _init_usage_state(agent): _USAGE_STATE: Dict[str, Any] = { "_user_turn_count": 0, "_is_user_initiated_turn": False, # Copilot x-initiator: first call of a user turn = "user" - # Usage anchors (agent/model_metadata.py): last response's exact usage + transcript + # Usage anchors (agent/usage_anchor.py): last response's exact usage + transcript # snapshot; invalidated on compaction/session switch so stale anchors never suppress compression. "_usage_anchor": None, "_turn_base_usage_anchor": None, + "_request_pressure_anchored": False, # whether the last pressure figure came from the anchor # Cumulative token usage for the session "session_prompt_tokens": 0, "session_completion_tokens": 0, diff --git a/agent/anthropic_endpoints.py b/agent/anthropic_endpoints.py index 82b96c5af1..237a562176 100644 --- a/agent/anthropic_endpoints.py +++ b/agent/anthropic_endpoints.py @@ -72,6 +72,25 @@ def _is_kimi_family_endpoint(base_url: str | None, model: str | None = None) -> ) + +_DEEPSEEK_THINKING_MODEL_PREFIXES = ( + "deepseek-r", "deepseek-v4", "deepseek_v4", "deepseek-pro", + "deepseek_pro", "deepseek-flash", "deepseek_flash", +) + + +def _model_name_is_deepseek_thinking(model: str | None) -> bool: + """Known DeepSeek thinking families behind an Anthropic-compatible relay. + + Strip vendor namespaces, but do not treat arbitrary DeepSeek chat/distill + names as evidence of the thinking replay contract. + """ + if not isinstance(model, str): + return False + name = model.strip().lower().rsplit("/", 1)[-1] + return bool(name) and name.startswith(_DEEPSEEK_THINKING_MODEL_PREFIXES) + + def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool: """DeepSeek's ``/anthropic`` route. In thinking mode DeepSeek requires prior-turn ``thinking`` blocks to round-trip while the generic third-party path strips them; its blocks are unsigned, diff --git a/agent/anthropic_message_convert.py b/agent/anthropic_message_convert.py index 2335345ea7..5309ba7fc9 100644 --- a/agent/anthropic_message_convert.py +++ b/agent/anthropic_message_convert.py @@ -12,7 +12,7 @@ from typing import Any, Dict, List, Optional, Tuple from agent.anthropic_endpoints import ( _is_deepseek_anthropic_endpoint, _is_kimi_family_endpoint, _is_nous_portal_endpoint, - _is_third_party_anthropic_endpoint, + _is_third_party_anthropic_endpoint, _model_name_is_deepseek_thinking, ) logger = logging.getLogger(__name__) @@ -565,7 +565,9 @@ def _manage_thinking_signatures(result: List[Dict[str, Any]], base_url: str | No """ is_third_party = _is_third_party_anthropic_endpoint(base_url) and not _is_nous_portal_endpoint(base_url) is_kimi = _is_kimi_family_endpoint(base_url, model) - is_deepseek = _is_deepseek_anthropic_endpoint(base_url) + is_deepseek = _is_deepseek_anthropic_endpoint(base_url) or ( + is_third_party and _model_name_is_deepseek_thinking(model) + ) last_assistant_idx = next((i for i in range(len(result) - 1, -1, -1) if result[i].get("role") == "assistant"), None) for idx, m in _assistant_block_lists(result): if is_kimi: diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index b62809ab4c..f8927a3a36 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -30,6 +30,7 @@ from agent.codex_headers import ( codex_cloudflare_headers as _codex_cloudflare_headers, is_official_codex_base_url as _is_official_codex_base_url, ) +from agent.codex_runtime import _codex_event_has_content # `openai.OpenAI` is imported lazily (~240 ms cold); `OpenAI` below is a proxy # so in-module calls, `auxiliary_client.OpenAI` reads and @@ -377,28 +378,11 @@ def _anthropic_aux_stream_event_hook() -> Callable[[Any], None]: return _on_event -_CODEX_PROGRESS_DELTA_TYPES = frozenset({ - "response.output_text.delta", "response.reasoning_summary_text.delta", "response.text.delta", - "response.audio.delta", "response.function_call_arguments.delta", "response.reasoning_text.delta", -}) - # A dead stream fails at the no-progress window (first token AND between tokens); a live # stream re-arms per event, bounded by _aux_stream_total_ceiling(). _AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS = 60.0 -def _codex_event_has_content(event: Any) -> bool: - """Whether a Codex Responses event carries a non-empty payload.""" - event_type = _field(event, "type") - if event_type in _CODEX_PROGRESS_DELTA_TYPES: - return bool(_field(event, "delta")) - if event_type == "response.output_item.added": - item = _field(event, "item") - return "function_call" in str(_field(item, "type") or "") and any( - bool(_field(item, f)) for f in ("id", "call_id", "name", "arguments")) - return False - - @contextlib.contextmanager def _aux_thread_local_hook(local: threading.local, hook): """Install one thread-local hook, restoring the prior on exit (non-callable = passthrough).""" @@ -591,19 +575,20 @@ def _is_arcee_trinity_thinking(model: Optional[str]) -> bool: return _bare_model(model) == "trinity-large-thinking" -# Codex OAuth hard-caps gpt-5.4/5.5/5.6 at 272K (raw API/OpenRouter expose 1.05M); the default -# 50% trigger would compact at ~136K, so raise to 85% (~231K). +# Codex OAuth hard-caps gpt-5.4/5.5/5.6 and gpt-6 Astra at 272K (raw API/OpenRouter expose 1.05M); +# the default 50% trigger would compact at ~136K, so raise to 85% (~231K). _CODEX_GPT54_GPT55_COMPACTION_THRESHOLD = 0.85 # gpt-5.3-codex-spark: Codex-OAuth-only, native 128K; 70% (~90K) leaves summary headroom. _CODEX_SPARK_COMPACTION_THRESHOLD = 0.70 def _is_codex_gpt54_or_gpt55(model: Optional[str], provider: Optional[str] = None) -> bool: - """True for gpt-5.4/5.5/5.6 (and the Daybreak Sol alias) on the Codex OAuth route only. + """True for gpt-5.4/5.5/5.6, gpt-6 Astra (and the Daybreak Sol alias) on the Codex OAuth route only. Other routes expose a larger window for the same slug and keep the user's threshold. Prefix-matched so ``-pro`` and dated snapshots track every 272K-capped family; ``-900k`` - picker variants are excluded. Name kept for the ``compression.codex_gpt55_autoraise`` key. + picker variants are excluded. Astra is substring-matched (any slug containing ``astra`` + without ``900k``). Name kept for the ``compression.codex_gpt55_autoraise`` key. """ bare = _codex_route_bare_model(model, provider) if bare is None: @@ -611,6 +596,8 @@ def _is_codex_gpt54_or_gpt55(model: Optional[str], provider: Optional[str] = Non from agent.model_metadata import is_codex_context_variant if is_codex_context_variant(bare): return False + if "astra" in bare: + return "900k" not in bare return bare == "gpt-daybreak-blue-latest" or any( bare == fam or bare.startswith(fam + "-") or bare.startswith(fam + ".") for fam in ("gpt-5.4", "gpt-5.5", "gpt-5.6")) @@ -642,7 +629,7 @@ def _compression_threshold_for_model( ) -> Optional[float]: """Per-model/route compression threshold override (fraction of context used), or None. - Arcee Trinity Large Thinking → 0.75 (preserve reasoning context); Codex-route gpt-5.4/5.5/5.6 + Arcee Trinity Large Thinking → 0.75 (preserve reasoning context); Codex-route gpt-5.4/5.5/5.6/Astra → 0.85, gated by ``allow_codex_gpt55_autoraise``; Codex-route gpt-5.3-codex-spark → 0.70, ungated. """ if _is_arcee_trinity_thinking(model): @@ -2957,7 +2944,8 @@ def _contains_any(text: str, needles: Tuple[str, ...]) -> bool: # Billing-body markers (credit exhaustion wrapped in 402/403/404/429 bodies), plus daily/weekly quota -# exhaustion (functionally credit exhaustion; "resource exhausted" is the Vertex/gRPC quota phrasing). +# exhaustion (functionally credit exhaustion; "resource exhausted" is the Vertex/gRPC quota phrasing — +# also serialized by SDK wrappers and NIM as RESOURCE_EXHAUSTED / ResourceExhausted / resource-exhausted). _PAYMENT_KEYWORDS = ( "credits", "insufficient funds", "can only afford", "billing", "payment required", "out of funds", "run out of funds", "balance_depleted", "no usable credits", @@ -2965,6 +2953,7 @@ _PAYMENT_KEYWORDS = ( "requires a subscription", "upgrade for access", "upgrade for higher limits", "reached your session usage limit", "quota exceeded", "quota_exceeded", "too many tokens per day", "daily limit", "tokens per day", "daily quota", "resource exhausted", + "resource_exhausted", "resource-exhausted", "resourceexhausted", "weekly usage limit", "weekly limit", ) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 98f4f25677..49b158bb58 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -323,14 +323,45 @@ def _provider_stream_error_from_text(text: str, finish_reason: Optional[str], *, return None +_IMAGE_PART_TYPES = frozenset({"image_url", "input_image", "image"}) + + +def _image_part_chars(part: Dict[str, Any], image_cost: int) -> int: + """Char-equivalent of one image content part: the per-image cost learned from provider usage + (x4 chars/token), never the base64 payload length. A single native screenshot priced as text + read as ~100K+ tokens and selected the giant-conversation watchdog tiers (#63871, #76411).""" + text = part.get("text") + return image_cost * 4 + (len(text) if isinstance(text, str) else 0) + + +def _payload_chars(value: Any, image_cost: int) -> int: + """``len(str(value))`` with image content parts priced at ``image_cost`` tokens each.""" + if value is None: + return 0 + if isinstance(value, dict): + part_type = value.get("type") + # JSON-Schema nodes may hold a sub-schema (``properties.type``) or a multi-type list + # under the "type" key; only scalar content-part types can ever match (#104793). + if isinstance(part_type, str) and part_type in _IMAGE_PART_TYPES and any(k in value for k in ("image_url", "image", "source", "file_id")): + return _image_part_chars(value, image_cost) + return sum(len(str(k)) + 6 + _payload_chars(v, image_cost) for k, v in value.items()) + if isinstance(value, list): + return sum(_payload_chars(item, image_cost) for item in value) + 2 * len(value) + return len(str(value)) + + def estimate_request_context_tokens(api_payload: Any) -> int: """Cheap char/4 context estimate for the stale-call detectors. Handles both wire shapes so Codex turns don't report ~0 tokens: list -> Chat ``messages``; dict with ``messages`` (+``tools``); dict with ``input`` (Responses API, - +``instructions``/``tools``); any other dict -> sum of its values.""" + +``instructions``/``tools``); any other dict -> sum of its values. Image parts + cost the learned per-image price, not their base64 length.""" + from agent.image_token_cost import current_image_token_cost + + image_cost = current_image_token_cost() def _chars(value: Any) -> int: - return 0 if value is None else len(str(value)) + return _payload_chars(value, image_cost) if isinstance(api_payload, list): return sum(_chars(item) for item in api_payload) // 4 @@ -378,15 +409,24 @@ def _validated_openrouter_provider_sort(raw_sort: Any) -> Optional[str]: def _provider_preferences_for_agent(agent) -> Dict[str, Any]: - """Build the validated provider-routing object shared by request paths.""" - preferences: Dict[str, Any] = {} - for key, value in (("only", agent.providers_allowed), ("ignore", agent.providers_ignored), - ("order", agent.providers_order), ("sort", _validated_openrouter_provider_sort(agent.provider_sort)), - ("require_parameters", True if agent.provider_require_parameters else None), - ("data_collection", agent.provider_data_collection)): - if value: - preferences[key] = value - return preferences + """Build the validated provider-routing object shared by request paths. + + ``provider_routing.models.`` overlays the flat constructor values for the CURRENT + ``agent.model`` (so ``/model`` switches, fallbacks, and delegated children on another + model each get their own pins without any surface re-plumbing the kwargs).""" + flat = {"only": agent.providers_allowed, "ignore": agent.providers_ignored, "order": agent.providers_order, + "sort": agent.provider_sort, "require_parameters": agent.provider_require_parameters, + "data_collection": agent.provider_data_collection} + per_model = {} + with contextlib.suppress(Exception): + from hermes_cli.config import load_config_readonly + from hermes_constants import resolve_per_model_provider_routing + _pr = load_config_readonly().get("provider_routing") + per_model = resolve_per_model_provider_routing(agent.model, (_pr or {}).get("models") if isinstance(_pr, dict) else None) + merged = {**flat, **{k: v for k, v in per_model.items() if k in flat}} + merged["sort"] = _validated_openrouter_provider_sort(merged["sort"]) + merged["require_parameters"] = True if merged["require_parameters"] else None + return {key: value for key, value in merged.items() if value} def _prompt_cache_scope_for_agent(agent) -> "str | None": @@ -446,16 +486,20 @@ def _estimate_chunk_bytes(chunk: Any) -> int: def _codex_wait_notice_recovery(*, stale_timeout: float, ttfb_enabled: bool, ttfb_timeout: float, - last_event_ts: Optional[float], call_start: float, idle_enabled: bool, idle_timeout: float, - elapsed: float) -> str: + last_event_ts: Optional[float], last_progress_ts: Optional[float], + retry_started_ts: Optional[float], call_start: float, idle_enabled: bool, + idle_timeout: float, idle_requires_progress: bool, elapsed: float) -> str: """Describe the earliest enabled Codex watchdog on the call timeline.""" deadlines: list[float] = [] if math.isfinite(stale_timeout): deadlines.append(stale_timeout) - if last_event_ts is None: + if retry_started_ts is not None: + if ttfb_enabled and math.isfinite(ttfb_timeout): + deadlines.append(max(0.0, retry_started_ts - call_start) + ttfb_timeout) + elif last_event_ts is None: if ttfb_enabled and math.isfinite(ttfb_timeout): deadlines.append(ttfb_timeout) - elif idle_enabled and math.isfinite(idle_timeout): + elif (not idle_requires_progress or last_progress_ts is not None) and idle_enabled and math.isfinite(idle_timeout): deadlines.append(max(0.0, last_event_ts - call_start) + idle_timeout) if not deadlines or min(deadlines) <= elapsed: return "" @@ -1012,6 +1056,7 @@ class _NonStreamWatchdogs: ttfb_timeout: float idle_enabled: bool idle_timeout: float + idle_requires_progress: bool def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs: @@ -1019,9 +1064,12 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs The stale detector kills a hung provider early so the retry loop can rotate credentials / fall back. Codex adds two failure modes: accepting the connection - but never emitting an event (no-byte TTFB cutoff; a reconnect succeeds in ~2s) - and stalling after the opening SSE frame (event-idle gap; any SSE event is - activity). Tunables: HERMES_CODEX_TTFB_TIMEOUT_SECONDS, + but never emitting an event (no-event TTFB cutoff; a reconnect succeeds in ~2s) + and stalling after substantive model progress begins (event-idle gap; any parsed SSE + event remains transport activity). Only the implicit official OpenAI Codex policy + for large contexts defers arming until progress; small requests, compatible backends, + and explicit overrides retain the legacy first-event semantics. Tunables: + HERMES_CODEX_TTFB_TIMEOUT_SECONDS, HERMES_CODEX_EVENT_STALE_TIMEOUT_SECONDS (0 disables each), HERMES_CODEX_TTFB_DISABLE_ABOVE_TOKENS / HERMES_CODEX_TTFB_STRICT, HERMES_CODEX_TTFB_MAX_SECONDS, HERMES_CODEX_HARD_TIMEOUT_SECONDS. @@ -1030,13 +1078,14 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs codex = agent.api_mode == "codex_responses" openai_codex_backend = _is_openai_codex_backend(agent) est_tokens = estimate_request_context_tokens(api_kwargs) + codex_floor = 0.0 if codex and openai_codex_backend: # Raise the stale floor for large payloads so healthy gateway-scale # requests aren't aborted mid-prefill. codex_floor = openai_codex_stale_timeout_floor(est_tokens) if codex_floor: stale_timeout = max(stale_timeout, codex_floor) - # Flat hard ceiling (#64507) for a request that emits SOME bytes then wedges. + # Flat hard ceiling (#64507) for a request that emits SOME events then wedges. # Default sits ABOVE the max floor (1200s) — a backstop, never tighter. 0 disables. hard_timeout = env_float("HERMES_CODEX_HARD_TIMEOUT_SECONDS", 1500.0) if hard_timeout > 0: @@ -1046,7 +1095,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs (default for threshold, default in ((100_000, 180.0), (50_000, 120.0), (10_000, 60.0)) if est_tokens > threshold), 12.0) - # No-byte TTFB cutoff. Default 120s: the SDK's own read timeout is 600s, + # No-event TTFB cutoff. Default 120s: the SDK's own read timeout is 600s, # and a tight 12s killed subscription-backed requests mid-prefill. ttfb_enabled = codex ttfb_timeout = env_float("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", 120.0) @@ -1058,22 +1107,29 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs disable_above = env_float("HERMES_CODEX_TTFB_DISABLE_ABOVE_TOKENS", 10_000.0) strict = os.environ.get("HERMES_CODEX_TTFB_STRICT", "").strip().lower() in {"1", "true", "yes", "on"} if not strict and disable_above > 0 and est_tokens >= disable_above and ttfb_timeout < idle_default: - logger.info("Scaling openai-codex no-byte TTFB watchdog from %.0fs to %.0fs " + logger.info("Scaling openai-codex no-event TTFB watchdog from %.0fs to %.0fs " "for large request (context=~%s tokens >= %.0f). " "Set HERMES_CODEX_TTFB_STRICT=1 to keep the smaller cutoff.", ttfb_timeout, idle_default, f"{est_tokens:,}", disable_above) ttfb_timeout = idle_default ttfb_cap = env_float("HERMES_CODEX_TTFB_MAX_SECONDS", 120.0) if ttfb_cap > 0 and ttfb_timeout > ttfb_cap: - logger.info("Capping openai-codex no-byte TTFB timeout from %.0fs to %.0fs " + logger.info("Capping openai-codex no-event TTFB timeout from %.0fs to %.0fs " "(context=~%s tokens). Set HERMES_CODEX_TTFB_MAX_SECONDS to tune.", ttfb_timeout, ttfb_cap, f"{est_tokens:,}") ttfb_timeout = ttfb_cap + # An operator-set idle timeout keeps first-event semantics; only the implicit + # default defers arming until model progress. Sentinel: env_float returns the + # default for unset AND unparseable values, so both count as implicit. + idle_explicit = env_float("HERMES_CODEX_EVENT_STALE_TIMEOUT_SECONDS", -1.0) != -1.0 idle_timeout = env_float("HERMES_CODEX_EVENT_STALE_TIMEOUT_SECONDS", idle_default) return _NonStreamWatchdogs(stale_timeout=stale_timeout, codex=codex, est_tokens=est_tokens, ttfb_enabled=ttfb_enabled, ttfb_timeout=ttfb_timeout, idle_enabled=codex and idle_timeout > 0, - idle_timeout=idle_timeout) + idle_timeout=idle_timeout, + idle_requires_progress=( + codex and openai_codex_backend and codex_floor > 0 and not idle_explicit + )) def _codex_silent_hang_hint(agent, api_kwargs: dict) -> Optional[str]: @@ -1107,6 +1163,18 @@ class _NonStreamRequest: self.codex_token = object() if agent.api_mode == "codex_responses" else None self.codex_retired = False self.wd = _resolve_nonstream_watchdogs(agent, api_kwargs) + self.codex_watchdog_state = ( + SimpleNamespace( + token=self.codex_token, + lock=threading.Lock(), + last_event_ts=None, + last_progress_ts=None, + retry_started_ts=None, + phase_aware=self.wd.idle_requires_progress, + ) + if self.codex_token is not None + else None + ) self.call_start = time.time() self.thread = None @@ -1131,8 +1199,14 @@ class _NonStreamRequest: return self.clients.set_client(client, kind=kind) def _call(self): + watchdog_state_var = watchdog_context_token = None try: self._install_codex_request_token() + if self.codex_watchdog_state is not None: + from agent.codex_runtime import _codex_watchdog_state_var + + watchdog_state_var = _codex_watchdog_state_var + watchdog_context_token = watchdog_state_var.set(self.codex_watchdog_state) self.result["response"] = _dispatch_nonstreaming_api_request( self.agent, self.api_kwargs, make_client=self._make_client) except Exception as e: @@ -1151,6 +1225,8 @@ class _NonStreamRequest: return self.result["error"] = e finally: + if watchdog_state_var is not None: + watchdog_state_var.reset(watchdog_context_token) # Retire first: close_once can raise, and a leaked token would let # a later worker mistake itself for the owning attempt. self._retire_codex_request_token() @@ -1176,13 +1252,23 @@ class _NonStreamRequest: def _model(self) -> str: return self.api_kwargs.get("model", "unknown") + def _codex_watchdog_snapshot(self): + state = self.codex_watchdog_state + if state is None: # non-codex request: no watchdog reads these + return (None, None, None) + with state.lock: + return state.last_event_ts, state.last_progress_ts, state.retry_started_ts + def _emit_wait_notice(self, elapsed: float) -> None: wd = self.wd try: + last_event_ts, last_progress_ts, retry_started_ts = self._codex_watchdog_snapshot() recovery = _codex_wait_notice_recovery(stale_timeout=wd.stale_timeout, ttfb_enabled=wd.ttfb_enabled, ttfb_timeout=wd.ttfb_timeout, - last_event_ts=getattr(self.agent, "_codex_stream_last_event_ts", None), + last_event_ts=last_event_ts, last_progress_ts=last_progress_ts, + retry_started_ts=retry_started_ts, call_start=self.call_start, idle_enabled=wd.idle_enabled, idle_timeout=wd.idle_timeout, + idle_requires_progress=wd.idle_requires_progress, elapsed=elapsed) self.agent._emit_wait_notice( f"⏳ waiting on {self.api_kwargs.get('model', 'the provider')} — " @@ -1191,40 +1277,46 @@ class _NonStreamRequest: logger.debug("wait-notice construction failed", exc_info=True) def _ttfb_kill(self, elapsed: float) -> None: - """No Codex event past the first-byte cutoff — kill so the retry loop + """No parsed Codex event past the first-event cutoff — kill so the retry loop reconnects instead of waiting out the stale timeout.""" agent, wd = self.agent, self.wd silent_hint = _codex_silent_hang_hint(agent, self.api_kwargs) - logger.warning("Codex stream produced no bytes within TTFB cutoff " + logger.warning("Codex stream produced no parsed stream event within TTFB cutoff " "(%.0fs > %.0fs, model=%s). Backend accepted the connection " "but sent no stream events. Killing connection so the retry loop can reconnect.", elapsed, wd.ttfb_timeout, self._model()) agent._buffer_status( - f"⚠️ No first byte from provider in {int(elapsed)}s (codex stream, model: {self._model()}). " + f"⚠️ No first stream event from provider in {int(elapsed)}s (codex stream, model: {self._model()}). " f"Reconnecting." + (f" {silent_hint}" if silent_hint else "")) self._abort_request("codex_ttfb_kill") agent._emit_wait_notice(f"⚠ no response from provider in {int(elapsed)}s — reconnecting...") - agent._touch_activity(f"codex stream killed after {int(elapsed)}s with no first byte") + agent._touch_activity(f"codex stream killed after {int(elapsed)}s with no first stream event") self._await_worker_after_kill( - f"Codex stream produced no bytes within {int(elapsed)}s (TTFB threshold: {int(wd.ttfb_timeout)}s)" + f"Codex stream produced no parsed stream event within {int(elapsed)}s " + f"(TTFB threshold: {int(wd.ttfb_timeout)}s)" + (f". {silent_hint}" if silent_hint else "")) def _idle_kill(self, event_stale_elapsed: float) -> None: - """First byte arrived, then SSE events stopped (keepalive/in_progress - frames refresh the timestamp and don't count).""" + """SSE events stopped after the phase-specific idle arm point. + + Only the implicit official OpenAI Codex policy arms on substantive model + progress; compatible providers and explicit operator timeouts arm on first + parsed event. Once armed, any parsed SSE event refreshes transport activity. + """ agent, wd = self.agent, self.wd - logger.warning("Codex stream produced no SSE events for %.0fs after first byte " + arm_point = "model progress began" if wd.idle_requires_progress else "the first parsed event" + logger.warning("Codex stream produced no SSE events for %.0fs after %s " "(threshold %.0fs, model=%s, context=~%s tokens). Killing " - "connection so the retry loop can reconnect.", event_stale_elapsed, wd.idle_timeout, + "connection so the retry loop can reconnect.", event_stale_elapsed, arm_point, wd.idle_timeout, self._model(), f"{wd.est_tokens:,}") agent._buffer_status( - f"⚠️ Codex stream sent no events for {int(event_stale_elapsed)}s after first byte " + f"⚠️ Codex stream sent no events for {int(event_stale_elapsed)}s after {arm_point} " f"(model: {self._model()}). Reconnecting.") self._abort_request("codex_stream_idle_kill") agent._touch_activity(f"codex stream killed after {int(event_stale_elapsed)}s with no SSE events") self._await_worker_after_kill( f"Codex stream produced no SSE events for {int(event_stale_elapsed)}s " - f"after first byte (threshold: {int(wd.idle_timeout)}s)") + f"after {arm_point} (threshold: {int(wd.idle_timeout)}s)") def _stale_kill(self, elapsed: float) -> None: """No response within the stale timeout: kill and count toward the @@ -1241,8 +1333,9 @@ class _NonStreamRequest: def _interrupt(self, elapsed: float) -> None: agent = self.agent + last_event_ts, _, _ = self._codex_watchdog_snapshot() _record_interrupted_provider_wait(agent, elapsed, - response_started=self.wd.codex and getattr(agent, "_codex_stream_last_event_ts", None) is not None + response_started=self.wd.codex and last_event_ts is not None ) # Mark cancelled BEFORE force-closing so the worker treats the transport # error as a cancel (#6600). Never close the shared client (releasing a @@ -1258,9 +1351,11 @@ class _NonStreamRequest: agent, wd = self.agent, self.wd if wd.codex: # Reset before the worker starts so a marker left over from a previous - # call on this agent can't be misread as first-byte for this one. - agent._codex_stream_last_event_ts = None - agent._codex_stream_last_progress_ts = None + # call on this agent can't be misread as the first event for this one. + with self.codex_watchdog_state.lock: + self.codex_watchdog_state.last_event_ts = None + self.codex_watchdog_state.last_progress_ts = None + self.codex_watchdog_state.retry_started_ts = None agent._touch_activity("waiting for non-streaming API response") self.thread = t = threading.Thread(target=_context_thread_target(self._call), daemon=True) @@ -1271,15 +1366,24 @@ class _NonStreamRequest: poll_count += 1 # Every ~30s: gateway inactivity heartbeat + rewrite the status line # so users see WHAT the wait is (the "infinite thinking" complaint). - elapsed = time.time() - self.call_start + now = time.time() + elapsed = now - self.call_start if poll_count % 100 == 0: # 100 × 0.3s = 30s self._emit_wait_notice(elapsed) - last_event_ts = getattr(agent, "_codex_stream_last_event_ts", None) - if wd.ttfb_enabled and elapsed > wd.ttfb_timeout and last_event_ts is None: + last_event_ts, last_progress_ts, retry_started_ts = self._codex_watchdog_snapshot() + retry_ttfb_elapsed = now - retry_started_ts if retry_started_ts is not None else None + if wd.ttfb_enabled and retry_ttfb_elapsed is not None and retry_ttfb_elapsed > wd.ttfb_timeout: + self._ttfb_kill(retry_ttfb_elapsed) + break + if (retry_started_ts is None and wd.ttfb_enabled + and elapsed > wd.ttfb_timeout and last_event_ts is None): self._ttfb_kill(elapsed) break - if wd.idle_enabled and last_event_ts is not None and (time.time() - last_event_ts) > wd.idle_timeout: - self._idle_kill(time.time() - last_event_ts) + idle_elapsed = now - last_event_ts if last_event_ts is not None else None + if (retry_started_ts is None and wd.idle_enabled and idle_elapsed is not None + and (not wd.idle_requires_progress or last_progress_ts is not None) + and idle_elapsed > wd.idle_timeout): + self._idle_kill(idle_elapsed) break if elapsed > wd.stale_timeout: self._stale_kill(elapsed) @@ -1720,6 +1824,19 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic value = getattr(assistant_message, attr, None) if value: msg[attr] = value + if attr == "codex_reasoning_items": + from agent.codex_responses_adapter import ( + has_replayable_native_compaction_checkpoint, + ) + + note_checkpoint = getattr( + agent.context_compressor, "note_native_compaction_checkpoint", None + ) + if ( + callable(note_checkpoint) + and has_replayable_native_compaction_checkpoint(agent, [msg]) + ): + note_checkpoint() if assistant_tool_calls: msg["tool_calls"] = [_assistant_tool_call_dict(agent, tc, i) for i, tc in enumerate(assistant_tool_calls)] @@ -2138,6 +2255,11 @@ def _iteration_summary_api_messages(agent, messages: list) -> list: agent._copy_reasoning_content_for_api(msg, api_msg) for key in _SUMMARY_FOREIGN_MESSAGE_KEYS: api_msg.pop(key, None) + # Mirror of the transport's role-qualified strip: ``name`` is + # schema-foreign on tool results only (strict providers reject with + # "contains item with unknown key name"); it stays on user/assistant. + if api_msg.get("role") == "tool": + api_msg.pop("name", None) # api_content holds the exact bytes the main loop sent; substituting (not popping) # keeps the summary's prefix identical instead of re-prefilling the largest context. # Strict OpenAI-compatible gateways (Fireworks-backed OpenCode Go, Mistral, Moonshot/Kimi) reject @@ -2924,6 +3046,7 @@ class _StreamingCall: tool_calls = _ToolCallAccumulator() tool_calls_acc = tool_calls.acc finish_reason = model_name = usage_obj = None + response_id = upstream_provider = None # the provider's own id / serving upstream, from the chunks role = "assistant" _diag = self._new_diag() self._writer_token = self._attempt_request_client = self._attempt_stream_response = None @@ -2969,6 +3092,10 @@ class _StreamingCall: continue if hasattr(chunk, "model") and chunk.model: model_name = chunk.model + if response_id is None and isinstance(getattr(chunk, "id", None), str) and chunk.id: + response_id = chunk.id + if upstream_provider is None and isinstance(getattr(chunk, "provider", None), str) and chunk.provider: + upstream_provider = chunk.provider # OpenRouter stamps who served if not chunk.choices: usage, finish_reason = self._choiceless_chunk(chunk, finish_reason) usage_obj = usage or usage_obj @@ -3023,7 +3150,8 @@ class _StreamingCall: if stream.final_response is not None: return self._adopt_final_response(stream.final_response) return self._finish_chat_stream(stream, role, content_parts, reasoning_parts, tool_calls_acc, - finish_reason, model_name, usage_obj, flush_pending=_flush_pending_stream_text) + finish_reason, model_name, usage_obj, flush_pending=_flush_pending_stream_text, + response_id=response_id, upstream_provider=upstream_provider) def _adopt_final_response(self, final_response): """Adapter returned a completed response for ``stream=True``: switch the @@ -3072,7 +3200,7 @@ class _StreamingCall: return mock_tool_calls or None, has_truncated_tool_args def _finish_chat_stream(self, stream, role, content_parts, reasoning_parts, tool_calls_acc, finish_reason, - model_name, usage_obj, *, flush_pending): + model_name, usage_obj, *, flush_pending, response_id=None, upstream_provider=None): """Assemble the non-streaming-shaped response after the chunk loop. A stream ending with no finish_reason is a drop, not a completion: return a partial-stream stub so the loop fails fast instead of executing empty @@ -3107,7 +3235,10 @@ class _StreamingCall: raise provider_stream_error flush_pending() message = SimpleNamespace(role=role, content=full_content, tool_calls=mock_tool_calls, reasoning_content=full_reasoning) - return SimpleNamespace(id="stream-" + str(uuid.uuid4()), model=model_name, usage=usage_obj, + # The provider's id when the chunks carried one (chatcmpl-/gen-...): it is what a provider needs to + # look a request up. Fabricated only when the stream never sent one. + return SimpleNamespace(id=response_id or ("stream-" + str(uuid.uuid4())), model=model_name, usage=usage_obj, + provider=upstream_provider, choices=[SimpleNamespace(index=0, message=message, finish_reason=effective_finish_reason)]) # ── anthropic_messages wire ───────────────────────────────────────── diff --git a/agent/client_lifecycle.py b/agent/client_lifecycle.py index 57384bb1c1..f617870920 100644 --- a/agent/client_lifecycle.py +++ b/agent/client_lifecycle.py @@ -3,6 +3,7 @@ shared primary client, single-slot per-request client caches (owner-thread close credential refresh/rotation, route-derived default headers. Extracted from ``run_agent.py``, MRO unchanged.""" import logging import threading +import time from contextlib import suppress from typing import Any, Optional @@ -524,7 +525,7 @@ class ClientLifecycleMixin: return False return self._adopt_openai_credentials(api_key, base_url, reason=f"{self.provider}_credential_refresh") - def _try_refresh_nous_client_credentials(self, *, force: bool = True) -> bool: + def _try_refresh_nous_client_credentials(self, *, force: bool = True, require_account: str | None = None) -> bool: # Portal serves anthropic/* on the native Messages route, so either client kind may hold the expiring JWT. if self.provider != "nous" or self.api_mode not in ("chat_completions", "anthropic_messages"): return False @@ -542,6 +543,20 @@ class ClientLifecycleMixin: api_key, base_url = creds.get("api_key"), creds.get("base_url") if not _valid_credential_pair(api_key, base_url): return False + if str(api_key).strip() == str(self.api_key or "").strip(): + return False # store holds the same key: nothing to adopt, no client rebuild + if require_account is not None: + try: + from hermes_cli.auth_constants import _decode_jwt_claims + new_account = _decode_jwt_claims(str(api_key)).get("sub") + except Exception: + new_account = None + if str(new_account or "") != require_account: + logger.info( + "Nous pre-expiry adoption skipped: the store's key belongs to a different account " + "than the one in hand; keeping the current credential." + ) + return False if self.api_mode == "anthropic_messages": self.api_key, self.base_url = api_key.strip(), base_url.strip().rstrip("/") self._anthropic_api_key, self._anthropic_base_url = self.api_key, self.base_url @@ -551,6 +566,39 @@ class ClientLifecycleMixin: self._client_kwargs.pop("default_headers", None) return self._adopt_openai_credentials(api_key, base_url, reason="nous_credential_refresh") + # Adopt a fresh key this many seconds before the one in hand expires. Wider than the store's + # own refresh skew (120 s) so the keepalive has normally already minted the replacement. + _NOUS_KEY_ADOPT_SKEW_S = 180 + + def _adopt_nous_key_before_expiry(self) -> bool: + """Swap in a fresh Nous agent key BEFORE the one in hand expires, so the request never 401s. + + The agent key is a JWT; its ``exp`` is read locally (no network). Inside the skew the store + is re-read under the auth-store lock: the keepalive thread normally holds a fresh key already + (adopt, no POST), otherwise ONE refresh runs and every peer adopts its result. Before this, + every agent in a process learned about the hourly expiry from its own 401, all in the same + minute (620 in one 200-subagent run), and the pool benched the sole credential for all of + them. Returns True when a new key was adopted. + + Identity guard: the replacement must belong to the SAME account (``sub`` claim) as the key + in hand. The store holds the logged-in singleton; an agent running on an explicitly supplied + or pool-selected key for a different account must never be silently moved onto it (that + changes who is billed). When either side lacks a ``sub`` nothing is adopted here; the + reactive 401 path is unchanged. + """ + if getattr(self, "provider", "") != "nous" or not getattr(self, "api_key", None): + return False + try: + from hermes_cli.auth_constants import _decode_jwt_claims + claims = _decode_jwt_claims(self.api_key) + except Exception: + return False + exp, account = claims.get("exp"), claims.get("sub") + if not account or not isinstance(exp, (int, float)) or exp - time.time() > self._NOUS_KEY_ADOPT_SKEW_S: + return False + return self._try_refresh_nous_client_credentials(force=False, require_account=str(account)) + + def _resolve_env_credentials(self) -> Optional[tuple]: """Current ``.env``-sourced ``(api_key, base_url, default_base)`` for this provider, or ``None``. diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 5f4395b268..bc079c2d32 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -541,16 +541,10 @@ def classify_responses_route(agent: Any) -> ResponsesRouteFlags: ) -def estimate_native_responses_preflight_tokens( - agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None, -) -> Optional[int]: - """Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted - session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails. - - Automatic preflight previously counted the full durable transcript. On a natively compacted Codex - session that overstates the wire by several times and fires local compression against history the main - request will never send (#96155). - """ +def _native_responses_replay_items( + agent: Any, messages: List[Dict[str, Any]] +) -> Optional[List[Dict[str, Any]]]: + """Build the native-compaction-eligible wire items, or ``None`` when ineligible.""" if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list): return None route = classify_responses_route(agent)._asdict() @@ -565,7 +559,37 @@ def estimate_native_responses_preflight_tokens( native_compaction_eligible=True, ) except Exception: - logger.debug("native Responses preflight conversion failed; falling back to generic estimate", exc_info=True) + logger.debug( + "native Responses replay conversion failed; using the generic fallback", + exc_info=True, + ) + return None + return items + + +def has_replayable_native_compaction_checkpoint( + agent: Any, messages: List[Dict[str, Any]] +) -> bool: + """Whether the current route would replay a persisted native checkpoint.""" + items = _native_responses_replay_items(agent, messages) + if items is None: + return False + from agent.native_compaction import has_compaction_checkpoint + return has_compaction_checkpoint(items) + + +def estimate_native_responses_preflight_tokens( + agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None, +) -> Optional[int]: + """Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted + session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails. + + Automatic preflight previously counted the full durable transcript. On a natively compacted Codex + session that overstates the wire by several times and fires local compression against history the main + request will never send (#96155). + """ + items = _native_responses_replay_items(agent, messages) + if items is None: return None from agent.model_metadata import estimate_request_tokens_rough return estimate_request_tokens_rough(items, system_prompt=system_prompt or "", tools=tools) diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index e5b2820458..f15732f708 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -4,6 +4,7 @@ AIAgent first: ``run_codex_app_server_turn`` drives one ``codex app-server`` sub from __future__ import annotations +import contextvars import json import logging import os @@ -13,8 +14,12 @@ from types import SimpleNamespace from typing import Any, Callable, Dict, List from agent.stream_single_writer import claim_stream_writer, stream_writer_is_current +from agent.usage_anchor import set_usage_anchor logger = logging.getLogger(__name__) +_codex_watchdog_state_var: contextvars.ContextVar[Any | None] = contextvars.ContextVar( + "codex_watchdog_state", default=None +) def _call_guarded(fn: Callable | None, fail_msg: str, *fail_args: Any, args: tuple = (), kwargs: dict | None = None): @@ -77,9 +82,12 @@ def _queue_token_counts(agent, fail_msg: str, *fail_extra: Any, counts: Callable logger.debug(fail_msg, agent.session_id, *fail_extra, exc) -def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]: +def _record_codex_app_server_usage(agent, turn, messages=None) -> dict[str, Any]: """Translate Codex app-server token usage into Hermes accounting. Prompt bucket = uncached + cached - input (the protocol exposes no cache-write tokens); a turn with no usage still counts as one API call.""" + input (the protocol exposes no cache-write tokens); a turn with no usage still counts as one API call. + ``messages`` (the transcript mirror) lets real usage anchor the next preflight: this runtime bypasses + the main loop's capture, and the mirror is never compacted natively, so without an anchor the rough + estimate grows monotonically and hermes-mode fires thread compaction on tiny threads (#100381).""" agent.session_api_calls += 1 usage = getattr(turn, "token_usage_last", None) compressor = getattr(agent, "context_compressor", None) @@ -90,6 +98,8 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]: if compressor is not None and getattr(compressor, "awaiting_real_usage_after_compression", False): # No usage cannot adjudicate the pending compaction; unlatch preflight deferral. compressor.update_from_response({}) + if compressor is not None and callable(getattr(compressor, "note_usage_less_response", None)): + compressor.note_usage_less_response() _queue_token_counts(agent, "Codex app-server api-call persistence failed (session=%s): %s", counts=lambda: billing(billing_mode="subscription_included")) return {} @@ -113,6 +123,12 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]: compressor.context_length = context_window except Exception: logger.debug("codex app-server usage update failed", exc_info=True) + if isinstance(messages, list): + from agent.usage_anchor import capture_usage_anchor, set_usage_anchor + + anchor = capture_usage_anchor(prompt_tokens, canonical_usage.output_tokens, messages) + if anchor is not None: + set_usage_anchor(agent, anchor) for key, value in usage_dict.items(): setattr(agent, f"session_{key}", getattr(agent, f"session_{key}") + value) cost_result = estimate_usage_cost( @@ -157,8 +173,7 @@ def _record_codex_app_server_compaction(agent, turn, *, approx_tokens: int | Non compressor.last_prompt_tokens, compressor.last_completion_tokens = -1, 0 compressor.awaiting_real_usage_after_compression = True # Provider-side context was rewritten; the usage anchor's transcript snapshot no longer matches. - agent._usage_anchor = None - agent._turn_base_usage_anchor = None + set_usage_anchor(agent, None) agent._last_compaction_in_place = False _call_guarded(getattr(agent, "event_callback", None) or None, "event_callback error on codex session:compress", args=("session:compress", { @@ -425,7 +440,7 @@ def _finish_codex_turn(agent, turn, messages: List[Dict[str, Any]], *, original_ # run_conversation() already bumped _turns_since_memory / _user_turn_count; only _iters_since_skill is ours. agent._iters_since_skill = getattr(agent, "_iters_since_skill", 0) + turn.tool_iterations _record_codex_app_server_compaction(agent, turn) - usage_result = _record_codex_app_server_usage(agent, turn) + usage_result = _record_codex_app_server_usage(agent, turn, messages=messages) # Skill nudge check AFTER iters were incremented (same as chat_completions). should_review_skills = (0 < agent._skill_nudge_interval <= agent._iters_since_skill and "skill_manage" in agent.valid_tool_names) @@ -507,6 +522,28 @@ def _event_field(event: Any, name: str, default: Any = None) -> Any: return value if value is not None else default +_CODEX_PROGRESS_DELTA_TYPES = frozenset({ + "response.output_text.delta", "response.reasoning_summary_text.delta", "response.text.delta", + "response.audio.delta", "response.function_call_arguments.delta", "response.reasoning_text.delta", +}) + + +def _codex_event_has_content(event: Any) -> bool: + """Whether a Codex Responses event carries substantive forward progress. + + Lifecycle/keepalive frames and empty structural deltas prove transport + liveness, but do not mean the model has begun producing its response. + """ + event_type = _event_field(event, "type") + if event_type in _CODEX_PROGRESS_DELTA_TYPES: + return bool(_event_field(event, "delta")) + if event_type == "response.output_item.added": + item = _event_field(event, "item") + return "function_call" in str(_event_field(item, "type") or "") and any( + bool(_event_field(item, field)) for field in ("id", "call_id", "name", "arguments")) + return False + + def _raise_stream_error(event: Any) -> None: """Raise ``_StreamErrorEvent`` from a ``type=error`` SSE frame. The spec puts code/message/param at the top level, but the SDK and several proxies nest them under ``error``; read top-level first, then the envelope.""" @@ -832,7 +869,12 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta # Retirement token for THIS request (installed by ``interruptible_api_call``). A watchdog that kills the # connection clears the agent-level token, so a worker still draining frames can tell it was retired. # ``None`` = no watchdog; every check passes. - request_token = getattr(agent, "_active_codex_stream_request_token", None) + watchdog_state = _codex_watchdog_state_var.get() + request_token = ( + watchdog_state.token + if watchdog_state is not None + else getattr(agent, "_active_codex_stream_request_token", None) + ) # Delta-sink claim for the CURRENT physical attempt (None until the stream opens). writer_token = {"value": None} @@ -847,8 +889,17 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta agent._codex_streamed_text_parts.append(text) agent._fire_stream_delta(text) - def _on_event(event: Any) -> None: # TTFB watchdog and activity touch — once per SSE event. - agent._codex_stream_last_event_ts = time.time() + def _on_event(event: Any) -> None: # TTFB/activity touch — once per SSE event. + now = time.time() + has_progress = _codex_event_has_content(event) + if watchdog_state is not None: + with watchdog_state.lock: + if watchdog_state.retry_started_ts is not None: + watchdog_state.retry_started_ts = None + watchdog_state.last_progress_ts = None + watchdog_state.last_event_ts = now + if has_progress: + watchdog_state.last_progress_ts = now agent._touch_activity("receiving stream response") def _interrupt_or_superseded() -> bool: @@ -911,8 +962,15 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta call_role = ("delegated" if getattr(agent, "is_subagent", False) else "fallback" if int(getattr(agent, "_fallback_index", 0) or 0) > 0 else "primary") for attempt in range(max_stream_retries + 1): + if not _request_is_current(): + raise TimeoutError("Codex Responses stream request retired before retry") if agent._interrupt_requested: raise InterruptedError("Agent interrupted before Codex stream retry") + if attempt > 0 and watchdog_state is not None and watchdog_state.phase_aware: + # A physical reconnect has its own no-event TTFB phase. Its first parsed + # event clears this marker and starts a fresh model-progress phase. + with watchdog_state.lock: + watchdog_state.retry_started_ts = time.time() intercepted_events: list = [] writer_token["value"] = event_stream = None try: diff --git a/agent/coding_context.py b/agent/coding_context.py index 5a6b3be08d..178d71eeb9 100644 --- a/agent/coding_context.py +++ b/agent/coding_context.py @@ -5,7 +5,8 @@ frozen :class:`RuntimeMode` built from a :class:`ContextProfile` (pure data). Th system prompt reads ``system_prompt_parts()``; the toolset collapses ONLY under opt-in ``focus`` (never strips a user-enabled toolset). ``agent.coding_context``: ``auto`` (default, prompt-only) / ``focus`` / ``on`` / ``off``. Resolved once, immutable; the -workspace snapshot is never re-probed per turn (cache safety). +workspace snapshot is probed once per session and replayed through +``system_prompt_parts(workspace_block=...)`` on every later build (cache safety). """ from __future__ import annotations @@ -322,13 +323,15 @@ class RuntimeMode: return None return [self.profile.toolset, *_enabled_mcp_servers(config)] - def system_prompt_parts(self, valid_tool_names=None) -> tuple[list[str], list[str], list[str]]: + def system_prompt_parts(self, valid_tool_names=None, workspace_block: Optional[str] = None) -> tuple[list[str], list[str], list[str]]: """Return (prefix, workspace, trailing) posture blocks in the historical flat order — brief, snapshot, operator instructions — so prompt assembly can put a cache boundary before the snapshot without changing persisted bytes. The brief carries the model-family edit-format nudge (one cached string); ``valid_tool_names`` drops the ``todo_list`` sentence when that tool isn't loaded; operator instructions ride their own block so - the brief stays byte-stable.""" + the brief stays byte-stable. ``workspace_block`` replays a snapshot the caller already + pinned at session start (``""`` = no workspace) instead of re-running the git probe; + ``None`` probes.""" if not self.is_coding: return [], [], [] prefix: list[str] = [] @@ -340,7 +343,7 @@ class RuntimeMode: if family is not None: brief = f"{brief}\n{_EDIT_FORMAT_GUIDANCE[family][1]}" prefix.append(brief) - workspace = build_coding_workspace_block(self.cwd) + workspace = build_coding_workspace_block(self.cwd) if workspace_block is None else workspace_block trailing = [f"Operator instructions (from config):\n{self.instructions}"] if self.instructions else [] return prefix, [workspace] if workspace else [], trailing @@ -395,11 +398,12 @@ def coding_selection(*, platform: Optional[str] = None, cwd: Optional[str | Path def coding_system_prompt_parts( *, platform: Optional[str] = None, cwd: Optional[str | Path] = None, config: Optional[dict[str, Any]] = None, - model: Optional[str] = None, valid_tool_names=None, + model: Optional[str] = None, valid_tool_names=None, workspace_block: Optional[str] = None, ) -> tuple[list[str], list[str], list[str]]: - """Return coding prefix, workspace snapshot, and trailing guidance.""" + """Return coding prefix, workspace snapshot, and trailing guidance. ``workspace_block`` + replays the caller's pinned session-start snapshot instead of probing git again.""" mode = resolve_runtime_mode(platform=platform, cwd=cwd, config=config, model=model) - return mode.system_prompt_parts(valid_tool_names=valid_tool_names) + return mode.system_prompt_parts(valid_tool_names=valid_tool_names, workspace_block=workspace_block) def coding_compact_skill_categories(*, platform: Optional[str] = None, cwd: Optional[str | Path] = None, config: Optional[dict[str, Any]] = None) -> frozenset[str]: diff --git a/agent/context_breakdown.py b/agent/context_breakdown.py index 597bfc8218..2626bb612e 100644 --- a/agent/context_breakdown.py +++ b/agent/context_breakdown.py @@ -92,7 +92,8 @@ def _glyph(cat: Dict[str, Any]) -> str: def compute_session_context_breakdown(agent: Any, messages: Optional[List[dict]] = None) -> Dict[str, Any]: """Return a Cursor-style context usage breakdown for one live agent.""" - from agent.model_metadata import anchored_context_tokens, estimate_messages_tokens_rough + from agent.model_metadata import estimate_messages_tokens_rough + from agent.usage_anchor import anchored_context_tokens from agent.system_prompt import build_system_prompt_parts messages = messages or [] diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 33838ca390..5233930417 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -25,7 +25,8 @@ from agent.context_engine import ContextEngine, sanitize_memory_context from agent.error_classifier import FailoverReason, classify_api_error from agent.micro_compaction import MicroCompactionMixin from agent.model_metadata import ( - MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough + MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough, + strip_opaque_replay_items, ) from agent.redact import redact_sensitive_text from agent.turn_context import drop_stale_api_content @@ -953,10 +954,6 @@ def _collect_protected_skill_names(messages: List[Dict[str, Any]], prune_boundar _CHARS_PER_TOKEN = 4 -# Flat per-image token estimate (realistic ceiling; matches Claude Code's constant). -_IMAGE_TOKEN_ESTIMATE = 1600 -# Same figure in char-budget currency. -_IMAGE_CHAR_EQUIVALENT = _IMAGE_TOKEN_ESTIMATE * _CHARS_PER_TOKEN _SUMMARY_FAILURE_COOLDOWN_SECONDS = 600 # Fallback handoff preserves continuity anchors only, not a transcript copy. @@ -1071,14 +1068,18 @@ def _bullets(items: list[str], limit: int = 8) -> str: def _content_length_for_budget(raw_content: Any) -> int: - """Effective char-length of message content for budgeting: text by length plus ``_IMAGE_CHAR_EQUIVALENT`` per image.""" + """Effective char-length of message content for budgeting: text by length plus the learned + per-image price (``agent.image_token_cost``, same figure the trigger estimator uses) per image.""" if isinstance(raw_content, str): return len(raw_content) if not isinstance(raw_content, list): return len(str(raw_content or "")) + from agent.image_token_cost import current_image_token_cost + + image_chars = current_image_token_cost() * _CHARS_PER_TOKEN # Any text-bearing part counts its text; image_url payload size is irrelevant. return sum( - (_IMAGE_CHAR_EQUIVALENT if _is_image_part(p) else len(p.get("text", "") or "")) if isinstance(p, dict) else len(str(p)) + (image_chars if _is_image_part(p) else len(p.get("text", "") or "")) if isinstance(p, dict) else len(str(p)) for p in raw_content ) @@ -1127,12 +1128,15 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) - and always-replayed provider fields. Always-replayed fields are charged because the preflight estimator sees the full shape; a mismatched size class protects blob-heavy rows as "small" and compaction re-fires. ``charge_stale_thinking=False`` skips newest-turn-only thinking keys. Accounting only; never mutates.""" - content = msg.get("content") or "" + # Charge the wire substitute, not both it and the clean display content. + sidecar = msg.get("api_content") + content = sidecar if isinstance(sidecar, str) and sidecar and msg.get("role") in ("user", "assistant") else msg.get("content") or "" text_tokens = estimate_tokens_rough(content) if isinstance(content, str) else _content_length_for_budget(content) // _CHARS_PER_TOKEN tokens = text_tokens + 10 # +10 for role/key overhead tokens += sum(estimate_tokens_rough(str(tc)) for tc in msg.get("tool_calls") or [] if isinstance(tc, dict)) for key in _ALWAYS_REPLAYED_BUDGET_KEYS: - tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN + # Opaque ciphertext is priced only by real usage (same rule as the preflight estimator). + tokens += _serialized_length_for_budget(strip_opaque_replay_items(msg.get(key))) // _CHARS_PER_TOKEN if not charge_stale_thinking: return tokens # Wire ships at most ONE generic thinking key (reasoning_content wins); @@ -1818,10 +1822,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._reset_session_compaction_state() def _reset_real_usage_pairing(self) -> None: - """Forget the real-vs-rough token pairing used by should_defer_preflight_to_real_usage().""" + """Forget the real-usage state read by real_usage_pending().""" self.last_real_prompt_tokens = self.last_compression_rough_tokens = 0 - self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens = 0 - self.awaiting_real_usage_after_compression = False + self.awaiting_real_usage_after_compression = self._provider_omits_usage = False def _reset_session_compaction_state(self) -> None: """Shared per-session reset for /new, /reset and session end.""" @@ -2355,19 +2358,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): """Pair the real prompt count with its rough estimate and judge the armed compaction verdict.""" if self.last_prompt_tokens > 0: self.last_real_prompt_tokens = self.last_prompt_tokens + self._provider_omits_usage = False if self.last_prompt_tokens < self.threshold_tokens: - if self.awaiting_real_usage_after_compression and self.last_compression_rough_tokens > 0: - self.last_rough_tokens_when_real_prompt_fit = self.last_compression_rough_tokens - elif self._pending_request_rough_tokens > 0: - # Pair the real prompt count with the same request's rough estimate so the defer baseline syncs on - # EVERY fitting response, not only after compaction; otherwise a never-compressed session has no - # baseline and preflight fires on the raw rough estimate (overcounts CJK / replay blobs severalfold). - self.last_rough_tokens_when_real_prompt_fit = self._pending_request_rough_tokens # Any real reading below the trigger proves the prompt fits: clear the latch. The fallback streak survives. self._record_ineffective_compression_verdict(0) - else: - self.last_rough_tokens_when_real_prompt_fit = 0 - self._pending_request_rough_tokens = 0 # Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count, # not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it. # Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's @@ -2409,33 +2403,43 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return self.last_prompt_tokens = snapshot - def note_request_rough_estimate(self, rough_tokens: int) -> None: - """Record the rough estimate of the request about to be sent, for pairing with real usage.""" - try: - self._pending_request_rough_tokens = max(0, int(rough_tokens)) - except (TypeError, ValueError): - self._pending_request_rough_tokens = 0 + def note_usage_less_response(self) -> None: + """A completed response carried no usage: until a real reading arrives, this provider cannot + adjudicate context pressure, so rough estimates decide instead of waiting forever (#2153).""" + self._provider_omits_usage = True + + def note_native_compaction_checkpoint(self) -> None: + """Wait for real usage before trusting a newly checkpointed request. + + Native Responses compaction replaces durable history with an opaque + encrypted checkpoint. Its serialized size is unrelated to the token + count billed by the provider, so the first rough estimate after capture + can jump by more than the whole context window. Reuse the one-response + compaction latch and discard any stale local-compression baseline; the + next provider response then pairs its real usage with the rough estimate + for the checkpointed request. + """ + self.awaiting_real_usage_after_compression = True + self.last_compression_rough_tokens = 0 def should_defer_preflight_to_real_usage(self, rough_tokens: int) -> bool: - """Return True when a high rough preflight estimate is known-noisy. - Projects real usage as ``last_real + (rough_now - rough_at_last_real)`` and fires only when the - projection, not the raw estimate, crosses the threshold. Not a strict upper bound for - chars/4-underestimated scripts (Cyrillic, Thai, Arabic); bounded by two backstops: a real - reading at/over threshold clears the baseline, and the overflow handler compacts reactively. - Callers with a smaller (raw-messages) basis can only over-defer; the pre-API pressure check - re-runs with the aligned basis.""" + """True when a whole-context ROUGH estimate over threshold must wait ONE request for the + provider's real usage. Callers skip this for usage-anchored figures (real prompt count + + delta of what was appended since), which never defer. A rough figure defers right after a + local or native compaction (the last real reading is stale — the latch) and on any + transcript the anchor does not cover (first request, rewind/edit-resend, reloaded history): + the next response re-anchors it. It never defers once the provider has proven it omits + usage, or the estimate would be the only signal and compression could never fire (#2153); + the overflow handler compacts reactively in every case.""" if rough_tokens < self.threshold_tokens: return False - # After compaction last_real_prompt_tokens is STALE (above threshold); defer one turn until real usage arrives. if self.awaiting_real_usage_after_compression: return True - if self.last_real_prompt_tokens <= 0 or self.last_real_prompt_tokens >= self.threshold_tokens: + # A real reading already at/over threshold needs no second opinion, and a rough figure past + # the whole window describes a request certain to fail — sending it only buys an overflow error. + if self.last_real_prompt_tokens >= self.threshold_tokens or rough_tokens >= self.context_length: return False - baseline = self.last_rough_tokens_when_real_prompt_fit or self.last_compression_rough_tokens - if baseline <= 0: - return False - # No baseline ratchet here: advancing rough without a matching real reading would defer on stale data. - return self.last_real_prompt_tokens + max(0, rough_tokens - baseline) < self.threshold_tokens + return not self._provider_omits_usage def should_compress(self, prompt_tokens: int = None) -> bool: """True when compression should run now (anti-thrash included; see :meth:`should_compress_info` for the reason).""" diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 18465dc71f..7a82999937 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -32,6 +32,7 @@ from agent.context_engine import automatic_compaction_status_message, sanitize_m from agent.memory_provider import PRE_COMPRESS_CHECKPOINT_API_VERSION from agent.model_metadata import estimate_messages_tokens_rough, estimate_request_tokens_rough from agent.session_activity import ActivityProvenance, normalize_activity_provenance +from agent.usage_anchor import set_usage_anchor logger = logging.getLogger(__name__) @@ -3114,8 +3115,7 @@ def _finish_compaction_boundary( compressor.awaiting_real_usage_after_compression = True # Transcript rewritten: invalidate the usage anchor's base snapshot explicitly # (its structural check would fail closed anyway); estimate until re-anchored. - agent._usage_anchor = None - agent._turn_base_usage_anchor = None + set_usage_anchor(agent, None) # Arm the effectiveness verdict only after a completed rewrite crosses the # boundary so later usage isn't charged to an attempt that changed nothing. if compression_made_progress: @@ -3812,7 +3812,7 @@ def _compress_context_via_codex_app_server( # An empty usage report must consume the pending verdict, not leave deferral # armed until a later turn; minimal test engines may lack update_from_response. if hasattr(agent.context_compressor, "update_from_response"): - _record_codex_app_server_usage(agent, result) + _record_codex_app_server_usage(agent, result, messages=messages) _reset_read_dedup_caches(task_id, skills=False) logger.info( "codex app-server compaction done: session=%s thread=%s turn=%s", _sid, diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 5bd41cc751..2e88c10327 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -1558,9 +1558,9 @@ _PLUGIN_COMPAT_LAZY = { 'PARTIAL_STREAM_STUB_ID': ('hermes_constants', 'PARTIAL_STREAM_STUB_ID'), 'PRE_API_COMPRESSION_STATUS_TEMPLATE': ('agent.conversation_compression', 'PRE_API_COMPRESSION_STATUS_TEMPLATE'), 'adaptive_rate_limit_backoff': ('agent.retry_utils', 'adaptive_rate_limit_backoff'), - 'anchored_context_tokens': ('agent.model_metadata', 'anchored_context_tokens'), + 'anchored_context_tokens': ('agent.usage_anchor', 'anchored_context_tokens'), 'automatic_compaction_status_message': ('agent.context_engine', 'automatic_compaction_status_message'), - 'capture_usage_anchor': ('agent.model_metadata', 'capture_usage_anchor'), + 'capture_usage_anchor': ('agent.usage_anchor', 'capture_usage_anchor'), 'classify_api_error': ('agent.error_classifier', 'classify_api_error'), 'close_interrupted_tool_sequence': ('agent.message_sanitization', 'close_interrupted_tool_sequence'), 'coalesce_tool_call_id': ('agent.message_sanitization', 'coalesce_tool_call_id'), diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 7bf3b8716f..ad4e9ae546 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -113,7 +113,8 @@ _BILLING_ERROR_CODES = frozenset({ # contains an overflow phrase; rate limit is matched first so throttle wins. _RATE_LIMIT_PATTERNS = ( "rate limit", "rate_limit", "too many requests", "throttled", "requests per minute", - "tokens per minute", "requests per day", "try again in", "please retry after", "resource_exhausted", + "tokens per minute", "requests per day", "try again in", "please retry after", + "resource exhausted", "resource_exhausted", "resource-exhausted", "resourceexhausted", "rate increased too quickly", "throttlingexception", "too many concurrent requests", "servicequotaexceededexception", "throttling", ) @@ -486,6 +487,13 @@ def _provider_special_cases(c: _Ctx) -> Optional[Verdict]: # to format_error and a status-less block isn't left retryable (#18028). if any(p in msg for p in _CONTENT_POLICY_BLOCKED_PATTERNS): return _V_CONTENT_BLOCKED + # ChatGPT Codex masks a rejected encrypted-reasoning replay behind the same bare + # ``invalid_prompt: Request blocked.`` it uses for real blocks (#92353). Exact envelope + # + provider only. The verdict keeps format_error's abort-and-fallback hints; the one + # extra thing it buys is turn_recovery's replay strip, which still requires cached + # ``codex_reasoning_items`` — a genuine block with nothing to strip behaves as before. + if _is_codex_masked_replay_rejection(c): + return _v(_R.invalid_encrypted_content, **_ABORT_FALLBACK) # Anthropic thinking-block 400s (signature mismatch after transcript # mutation). Not gated on provider — OpenRouter proxies Anthropic errors. if status == 400 and "thinking" in msg and any(p in msg for p in _THINKING_MUTATION_WORDS): @@ -777,6 +785,22 @@ def _is_server_injected_param_rejection(error_msg: str, provider: str) -> bool: return False +_CODEX_MASKED_REPLAY_MESSAGE = "request blocked." + + +def _is_codex_masked_replay_rejection(c: "_Ctx") -> bool: + """HTTP 400 / status-less ``{code: invalid_prompt, message: "Request blocked."}`` from + ``openai-codex`` — as an SDK error body, a Responses ``error`` SSE frame, or the + ``response.failed`` text ``"invalid_prompt: Request blocked."``.""" + if c.provider_slug != "openai-codex" or c.status_code not in (None, 400): + return False + # The OpenAI SDK unwraps ``body["error"]`` on status errors; stream frames keep the envelope. + body_msg = next((str(m).strip().lower() for m in _body_message_candidates(c.body or {}) if m), "") + return (c.code == "invalid_prompt" and body_msg == _CODEX_MASKED_REPLAY_MESSAGE) or ( + c.msg.strip() == f"invalid_prompt: {_CODEX_MASKED_REPLAY_MESSAGE}" + ) + + def _error_obj(body: Any) -> dict: """``body["error"]`` when it is a dict, else ``{}``.""" err = body.get("error") if isinstance(body, dict) else None diff --git a/agent/image_token_cost.py b/agent/image_token_cost.py new file mode 100644 index 0000000000..1fc5e570ee --- /dev/null +++ b/agent/image_token_cost.py @@ -0,0 +1,135 @@ +"""Per-image token cost learned from the provider's own usage, never from a vendor formula. + +A flat per-image constant is wrong in both directions: a 1920x1080 screenshot costs ~1,100 tokens +on one provider and 4,000+ on a local mmproj model. The provider prices every image exactly on the +request that carries it, so the cost is observable: with a fresh usage anchor (real prompt count of +the previous response), the residual between the next real ``prompt_tokens`` and +``anchor + text-only delta`` is the price of the N images that delta introduced (#70328). + +The learned value is kept per ``model@host`` in ``~/.hermes/cache/image_token_costs.json`` so a new +session starts calibrated, and bound per turn through a ContextVar so every estimator +(preflight trigger, tail-budget walk, gateway hygiene) prices images the same way. +""" + +from __future__ import annotations + +import contextlib +import logging +from contextvars import ContextVar +from typing import Any, Dict, List, Optional + +logger = logging.getLogger(__name__) + +DEFAULT_IMAGE_TOKEN_COST = 1500 +# Observations outside this band are text-estimate noise, not an image price. +_MIN_PLAUSIBLE, _MAX_PLAUSIBLE = 64, 32_768 +_EMA_ALPHA = 0.5 + +_image_cost_var: ContextVar[Optional[int]] = ContextVar("hermes_image_token_cost", default=None) +_LEARNED: Dict[str, int] = {} +_LOADED = False + + +def _cache_path(): + from agent.model_metadata import _cache_file + + return _cache_file("image_token_costs.json") + + +def _key(model: Any, base_url: Any) -> str: + from utils import base_url_hostname + + return f"{model or ''}@{base_url_hostname(base_url or '') or ''}" + + +def _load() -> None: + global _LOADED + if _LOADED: + return + _LOADED = True + from agent.model_metadata import _load_json_dict + + for k, v in _load_json_dict(_cache_path()).items(): + if isinstance(v, int) and _MIN_PLAUSIBLE <= v <= _MAX_PLAUSIBLE: + _LEARNED[k] = v + + +def learned_image_token_cost(model: Any, base_url: Any) -> int: + """Learned per-image cost for ``model@host``, else the flat default.""" + _load() + return _LEARNED.get(_key(model, base_url), DEFAULT_IMAGE_TOKEN_COST) + + +def current_image_token_cost() -> int: + """Per-image cost bound for the running turn (see :func:`image_cost_context`), else the default.""" + bound = _image_cost_var.get() + return bound if bound is not None else DEFAULT_IMAGE_TOKEN_COST + + +@contextlib.contextmanager +def image_cost_context(cost: Optional[int]): + token = _image_cost_var.set(cost) + try: + yield + finally: + _image_cost_var.reset(token) + + +def bind_image_token_cost(agent: Any) -> None: + """Bind the agent's learned per-image cost to the current context for the rest of the turn.""" + _image_cost_var.set(learned_image_token_cost(getattr(agent, "model", None), getattr(agent, "base_url", None))) + + +def count_images(messages: List[Dict[str, Any]]) -> int: + from agent.model_metadata import _count_image_tokens + + return sum(_count_image_tokens(m, 1) for m in messages if isinstance(m, dict)) + + +def calibrate_from_usage(agent: Any, messages: List[Dict[str, Any]], prompt_tokens: Any) -> Optional[int]: + """Learn the per-image cost from the response that just priced ``messages``. + + Requires the PREVIOUS anchor (real count of the prior request) to still match: the residual + ``prompt_tokens - (anchor + text-only delta)`` is then the provider's price for the images the + delta introduced. Returns the new learned cost, or None when this response teaches nothing + (no anchor, no new images, implausible residual).""" + from agent.usage_anchor import anchored_context_tokens + + anchor = getattr(agent, "_usage_anchor", None) + try: + real = int(prompt_tokens or 0) + except (TypeError, ValueError): + return None + if real <= 0 or not isinstance(anchor, dict) or not isinstance(messages, list): + return None + base_count = int(anchor.get("base_count") or 0) + delta = messages[base_count:] + if delta and isinstance(delta[0], dict) and delta[0].get("role") == "assistant": + delta = delta[1:] + n_images = count_images(delta) + if n_images <= 0: + return None + with image_cost_context(0): + text_only = anchored_context_tokens(messages, anchor) + if text_only is None: + return None + per_image = (real - text_only) // n_images + if not _MIN_PLAUSIBLE <= per_image <= _MAX_PLAUSIBLE: + return None + key = _key(getattr(agent, "model", None), getattr(agent, "base_url", None)) + _load() + prior = _LEARNED.get(key) + learned = per_image if prior is None else int(prior + _EMA_ALPHA * (per_image - prior)) + _LEARNED[key] = learned + _image_cost_var.set(learned) + try: + from utils import atomic_json_write + + atomic_json_write(_cache_path(), dict(_LEARNED), indent=0, separators=(",", ":")) + except Exception: + logger.debug("image token cost persist failed", exc_info=True) + logger.info( + "Image token cost calibrated from provider usage: %s images priced %s tokens each (learned %s for %s)", + n_images, f"{per_image:,}", f"{learned:,}", key, + ) + return learned diff --git a/agent/interrupt_control.py b/agent/interrupt_control.py index e06f3ac13a..f99aca9dc1 100644 --- a/agent/interrupt_control.py +++ b/agent/interrupt_control.py @@ -9,6 +9,7 @@ import threading from typing import Optional from agent.interrupt_compat import request_hard_interrupt +from tools.interrupt import request_yield as _request_yield from tools.interrupt import set_interrupt as _set_interrupt # Same logger name as the origin module so log records / caplog filters are unchanged. @@ -245,8 +246,20 @@ class InterruptControlMixin: return False # Never kill a tool to deliver guidance; the steer drain puts it on the final tool result. + # A foreground terminal command would park that delivery until it exits (a 5-minute + # `sleep` poller, a build), so ask the tool workers to YIELD: terminal hands the live + # process to the background registry and returns; tools that don't yield are unaffected. if getattr(self, "_executing_tools", False): - return self.steer(cleaned) + accepted = self.steer(cleaned) + if accepted: + tracker = getattr(self, "_tool_worker_threads", None) + tracker_lock = getattr(self, "_tool_worker_threads_lock", None) + if tracker is not None and tracker_lock is not None: + with tracker_lock: + worker_tids = list(tracker) + for tid in worker_tids: + _request_yield(tid) + return accepted _model_active = getattr(self, "_model_request_active", None) with _ic_lock(self, "_pending_redirect_lock"): diff --git a/agent/memory_manager.py b/agent/memory_manager.py index ce74dc11b4..44ca6f930e 100644 --- a/agent/memory_manager.py +++ b/agent/memory_manager.py @@ -18,6 +18,7 @@ from typing import Any, Callable, Dict, List, Optional from agent.memory_provider import MemoryProvider, PRE_COMPRESS_CHECKPOINT_API_VERSION from agent.skill_commands import extract_user_instruction_from_skill_message +from tools.hook_output_spill import get_spill_config, spill_if_oversized from tools.registry import tool_error logger = logging.getLogger(__name__) @@ -295,6 +296,7 @@ class MemoryManager: def __init__(self, *, external_prefetch_timeout: Optional[float] = None) -> None: self._providers: List[MemoryProvider] = [] self._tool_to_provider: Dict[str, MemoryProvider] = {} + self._external_prefetch_spill_config: Optional[Dict[str, Any]] = None self._has_external: bool = False timeout = external_prefetch_timeout timeout = _EXTERNAL_PREFETCH_TIMEOUT_S if timeout is None else float(timeout) @@ -340,6 +342,7 @@ class MemoryManager: ) return self._has_external = True + self._external_prefetch_spill_config = get_spill_config() self._providers.append(provider) @@ -434,7 +437,15 @@ class MemoryManager: self._external_prefetch_threads.pop(provider.name, None) if "error" in result_box: raise result_box["error"] - return result_box.get("value", "") + result = result_box.get("value", "") + if result and result.strip(): + # Prefetch is stamped into the user turn's api_content and replayed every later turn; + # spill oversized results like plugin hook output so one provider can't inflate the prefix. + result = spill_if_oversized( + result, session_id=session_id, source=f"{provider.name} memory prefetch", + config=self._external_prefetch_spill_config, + ) + return result def describe_recall(self) -> str: """Deterministic recall indicator line (e.g. ``"🧠 Provider — recalled 3 memories"``); ``""`` if none. diff --git a/agent/model_metadata.py b/agent/model_metadata.py index b5742d822d..d24df995f1 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1977,15 +1977,18 @@ def estimate_tokens_rough(text: str) -> int: def estimate_messages_tokens_rough(messages: List[Dict[str, Any]], *, charge_stale_thinking: bool = True) -> int: - """Rough token estimate for a message list (pre-flight only). Images cost a flat ~1500 tokens - each rather than their base64 length. ``charge_stale_thinking=False`` mirrors the tail-budget + """Rough token estimate for a message list (pre-flight only). Images cost the per-image price + learned from provider usage (``agent.image_token_cost``; flat default before calibration) + rather than their base64 length. ``charge_stale_thinking=False`` mirrors the tail-budget walk (``context_compressor._estimate_msg_budget_tokens``): on non-echo routes stale reasoning rides the wire only for the NEWEST assistant turn, so excluding it keeps the compaction TRIGGER in the same size class as the walk — otherwise reasoning-heavy sessions fire preflight forever.""" - _IMAGE_TOKEN_COST = 1500 + from agent.image_token_cost import current_image_token_cost + + image_cost = current_image_token_cost() if not charge_stale_thinking: messages = _strip_stale_thinking_for_estimate(messages) - return sum(_estimate_message_tokens_cached(msg, _IMAGE_TOKEN_COST) for msg in messages) + return sum(_estimate_message_tokens_cached(msg, image_cost) for msg in messages) # Thinking-text keys replayed for at most the newest assistant turn on non-echo routes — must stay @@ -2020,7 +2023,7 @@ def _strip_stale_thinking_for_estimate(messages: List[Dict[str, Any]]) -> List[D # estimate. Because the api_messages build shallow-copies history dicts each iteration, the copies share the # same content strings — so unchanged history messages hit the memo even though the outer dicts are fresh # objects every turn. -_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {} +_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int, int]] = {} # pins, text tokens, image count _MSG_TOKENS_CACHE_MAX = 4096 @@ -2041,19 +2044,23 @@ def _msg_fingerprint(value: Any, pins: list) -> Any: def _estimate_message_tokens_cached(msg: Any, image_cost: int) -> int: - def _compute() -> int: - return _estimate_message_tokens_without_images(msg) + _count_image_tokens(msg, image_cost) + """Text tokens + images x ``image_cost``; the memo holds text and image COUNT so a recalibrated + per-image price re-prices cached rows without invalidating them.""" + def _compute() -> Tuple[int, int]: + return _estimate_message_tokens_without_images(msg), _count_image_tokens(msg, 1) try: pins: list = [] key = _msg_fingerprint(msg, pins) hash(key) except Exception: - return _compute() + text, images = _compute() + return text + images * image_cost cached = _MSG_TOKENS_CACHE.get(key) if cached is not None: - return cached[1] - tokens = _compute() - _MSG_TOKENS_CACHE[key] = (pins, tokens) + return cached[1] + cached[2] * image_cost + text, images = _compute() + tokens = text + images * image_cost + _MSG_TOKENS_CACHE[key] = (pins, text, images) while len(_MSG_TOKENS_CACHE) > _MSG_TOKENS_CACHE_MAX: try: _MSG_TOKENS_CACHE.pop(next(iter(_MSG_TOKENS_CACHE))) @@ -2079,6 +2086,18 @@ def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int: return count * cost_per_image +def strip_opaque_replay_items(items: Any) -> Any: + """``codex_reasoning_items`` with ``encrypted_content`` blanked for local token estimation. + The ciphertext is priced by the provider's own count, never by its bytes (a compaction + checkpoint alone can be 5M chars, #100611); only real usage prices it.""" + if not isinstance(items, list): + return items + return [ + {k: ("" if k == "encrypted_content" else v) for k, v in item.items()} if isinstance(item, dict) else item + for item in items + ] + + def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]: """Shadow of a message holding only what the provider actually receives. * ``api_content`` SUBSTITUTES ``content`` (mirrors ``turn_context.substitute_api_content`` exactly): @@ -2086,7 +2105,11 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]: other shape would UNDERcount — the dangerous direction. * Base64 images become a placeholder; ``_count_image_tokens`` charges them flat. * ``reasoning`` never ships as-is (request builds pop it after optionally promoting it into - ``reasoning_content``); counting both inflated estimates up to +53%.""" + ``reasoning_content``); counting both inflated estimates up to +53%. + * Opaque provider blobs (``encrypted_content`` on codex reasoning / compaction items) are + ciphertext the provider prices by its OWN token count, never by bytes; a native compaction + checkpoint alone can be 5M chars (#100611). They contribute 0 here: only real usage ever + prices them, and the usage anchor carries that price forward.""" sidecar = msg.get("api_content") sidecar_wins = isinstance(sidecar, str) and bool(sidecar) and msg.get("role") in ("user", "assistant") _rc = msg.get("reasoning_content") @@ -2108,6 +2131,10 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]: ] elif k == "content" and isinstance(v, dict) and v.get("_multimodal"): shadow[k] = v.get("text_summary", "") + elif k == "codex_reasoning_items": + shadow[k] = strip_opaque_replay_items(v) + elif k == "encrypted_content": # a Responses reasoning/compaction item passed as a row + shadow[k] = "" else: shadow[k] = v return shadow @@ -2133,54 +2160,6 @@ def estimate_request_tokens_rough( return total -# Usage-anchored accounting: ``usage.prompt_tokens`` is EXACT for everything sent on that request, so -# anchoring shrinks chars/4 estimation to the messages appended since. Fields: prompt_tokens / -# completion_tokens (provider usage at capture); base_count (len(messages) at capture — the reply is -# not yet appended and is covered by completion_tokens, so the delta walk skips it at index base_count); -# base_last_id / base_last_role (identity of the last message; compaction/splices replace it -> full estimation). - - -def capture_usage_anchor(prompt_tokens: Any, completion_tokens: Any, messages: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]: - """Build a usage anchor from provider-reported usage, or None.""" - try: - pt = int(prompt_tokens or 0) - ct = int(completion_tokens or 0) - except (TypeError, ValueError): - return None - if pt <= 0 or not isinstance(messages, list): - return None # no usable usage (some endpoints omit it) — caller keeps its anchor - last = messages[-1] if messages else None - return { - "prompt_tokens": pt, - "completion_tokens": max(0, ct), - "base_count": len(messages), - "base_last_id": id(last) if last is not None else None, - "base_last_role": last.get("role") if isinstance(last, dict) else None, - } - - -def anchored_context_tokens(messages: List[Dict[str, Any]], anchor: Optional[Dict[str, Any]], *, charge_stale_thinking: bool = True) -> Optional[int]: - """Anchored prompt+completion tokens plus a rough estimate of ONLY the messages appended since; - None when the anchor is missing or stale. The anchored response's own reply is skipped (already - in completion_tokens). ``charge_stale_thinking`` is forwarded to the delta estimate.""" - if not isinstance(anchor, dict) or not isinstance(messages, list): - return None - base_count = anchor.get("base_count") or 0 - if base_count <= 0 or len(messages) < base_count: - return None - base_msg = messages[base_count - 1] - base_role = base_msg.get("role") if isinstance(base_msg, dict) else None - if id(base_msg) != anchor.get("base_last_id") or base_role != anchor.get("base_last_role"): - return None - total = int(anchor["prompt_tokens"]) + int(anchor.get("completion_tokens") or 0) - delta = messages[base_count:] - if delta and isinstance(delta[0], dict) and delta[0].get("role") == "assistant": - delta = delta[1:] - if delta: - total += estimate_messages_tokens_rough(delta, charge_stale_thinking=charge_stale_thinking) - return total - - # Keyed by ``id(tools)``; bounded, oldest-first eviction. Repeated ``str(tools)`` on # large schemas stalls GUI event loops under GIL pressure. _TOOLS_TOKENS_CACHE: dict[int, Tuple[int, str, str, int]] = {} diff --git a/agent/native_compaction.py b/agent/native_compaction.py index 0202770c4d..8707cc91af 100644 --- a/agent/native_compaction.py +++ b/agent/native_compaction.py @@ -328,7 +328,12 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool: def has_compaction_checkpoint(items: Any) -> bool: """Does this ``codex_reasoning_items`` sidecar carry a compaction checkpoint? A checkpoint is cumulative context living in exactly one place: rewrite/discard the sidecar only after asking.""" - return isinstance(items, list) and any(_is_compaction_item(item) for item in items) + return isinstance(items, list) and any( + _is_compaction_item(item) + and isinstance(item.get("encrypted_content"), str) + and bool(item["encrypted_content"].strip()) + for item in items + ) def merge_interim_reasoning_items(prior_items: Any, new_items: Any) -> List[Dict[str, Any]]: diff --git a/agent/nous_wire.py b/agent/nous_wire.py new file mode 100644 index 0000000000..baf602abde --- /dev/null +++ b/agent/nous_wire.py @@ -0,0 +1,112 @@ +"""Nous Portal ``anthropic/*`` wire selection when ``nous.anthropic_wire`` is ``auto``. + +Portal serves Claude two ways and Hermes cannot tell which from the request: an OpenRouter +passthrough (today, for every ``anthropic/*`` id) or GMI/Vertex (planned once GMI is back). The +native Messages wire is the better transport, but on the OpenRouter path it re-writes the previous +turn's prompt cache on 14-20% of consecutive calls in concurrent tool loops (measured 2026-09-06; +NousResearch/api#227), so the session must ride chat/completions there. On GMI that is untested, +and until it is measured ``auto`` never promotes to native. + +The upstream IS visible in the first RESPONSE: OpenRouter stamps ``provider`` (chat wire) and +mints ``gen--`` ids; GMI/Vertex responses carry neither. So ``auto`` starts every +session on chat (safe on both upstreams), reads the first response, and switches the session to +native only when the upstream is GMI and native has been cleared for GMI. One decision per +session, at call 1, before there is a cache to lose; later calls never flip. + +``classify_upstream`` is pure and unit-tested; ``maybe_switch_wire_after_first_response`` is +the single hook, called from the usage recorder. +""" +from __future__ import annotations + +import logging +import re +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +# Flip to True only after the 20x6 concurrency probe (evals/postmortem/live_ab) is clean on a +# GMI-served anthropic/* id on the native wire. Until then ``auto`` is chat everywhere. +GMI_NATIVE_WIRE_CLEARED = False + +_OPENROUTER_ID = re.compile(r"^gen-\d{9,}-[A-Za-z0-9_-]{8,}$") + + +def classify_upstream(response: Any) -> Optional[str]: + """``"openrouter"`` / ``"gmi"`` / ``None`` (unknown) from a Portal response object. + + Works on both wires: the OpenAI SDK object exposes ``.provider`` (OpenRouter's upstream name, + e.g. ``"Anthropic"``, ``"Amazon Bedrock"``) and an OpenRouter-minted ``.id``; the Anthropic SDK + object has ``.id`` only. GMI/Vertex responses have Anthropic-native ``msg_…`` ids and no + ``provider``. Anything else is unknown, and unknown never triggers a switch. + """ + if response is None: + return None + if isinstance(getattr(response, "provider", None), str) and getattr(response, "provider"): + return "openrouter" + rid = getattr(response, "id", None) + if isinstance(rid, str): + if _OPENROUTER_ID.match(rid): + return "openrouter" + if rid.startswith("msg_"): + return "gmi" + return None + + +def wire_for_upstream(upstream: Optional[str]) -> str: + """The api_mode ``auto`` wants once the upstream is known. Chat unless GMI and cleared.""" + if upstream == "gmi" and GMI_NATIVE_WIRE_CLEARED: + return "anthropic_messages" + return "chat_completions" + + +def maybe_switch_wire_after_first_response(agent: Any, response: Any, api_call_count: int) -> bool: + """Decide the session's wire from its first response; the switch itself is applied at the + start of the next iteration (``apply_pending_wire_switch``), never while a response is being + consumed. Returns True when a switch was scheduled. + + Only for provider=nous, anthropic/* models, ``nous.anthropic_wire: auto``, and only on the + session's first API call. + """ + if api_call_count != 1 or getattr(agent, "_nous_wire_decided", False): + return False + if (getattr(agent, "provider", "") or "").lower() != "nous": + return False + model = str(getattr(agent, "model", "") or "") + if not model.lower().startswith("anthropic/"): + return False + try: + from hermes_cli.providers import _nous_anthropic_wire + if _nous_anthropic_wire() != "auto": + return False + except Exception: + return False + agent._nous_wire_decided = True # one decision per session, whatever it is + upstream = classify_upstream(response) + want = wire_for_upstream(upstream) + if want == getattr(agent, "api_mode", None): + logger.debug("nous wire auto: upstream=%s, staying on %s", upstream, want) + return False + agent._nous_wire_pending = (want, upstream) + return True + + +def apply_pending_wire_switch(agent: Any) -> bool: + """At iteration start, with no response in flight: perform the switch scheduled by + ``maybe_switch_wire_after_first_response``. Reuses ``switch_model`` (same model/provider, new + api_mode) so client rebuild, cache policy and ``_primary_runtime`` stay consistent. A failure + is logged and the session stays on its current wire.""" + pending = getattr(agent, "_nous_wire_pending", None) + if not pending: + return False + agent._nous_wire_pending = None + want, upstream = pending + try: + from agent.agent_runtime_helpers import switch_model + switch_model(agent, agent.model, "nous", api_key=getattr(agent, "api_key", "") or "", + base_url=getattr(agent, "base_url", "") or "", api_mode=want) + except Exception as exc: # never let wire selection break a turn + logger.warning("nous wire auto: switch to %s failed (%s); staying on %s", want, exc, agent.api_mode) + return False + logger.info("nous wire auto: upstream=%s -> %s for the rest of session %s", upstream, want, + getattr(agent, "session_id", "?")) + return True diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 9e00dd0dc2..2f36082498 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -96,7 +96,15 @@ def _scan_context_content(content: str, filename: str) -> str: def _find_git_root(start: Path) -> Optional[Path]: """Nearest ancestor (or *start* itself) containing ``.git``, else None.""" current = start.resolve() - return next((p for p in (current, *current.parents) if (p / ".git").exists()), None) + # A parent the process may not stat (locked-down /home on shared hosts) is "no .git here", not a crash. + return next((p for p in (current, *current.parents) if _exists_or_denied(p / ".git")), None) + + +def _exists_or_denied(path: Path) -> bool: + try: + return path.exists() + except OSError: + return False def _find_hermes_md(cwd: Path) -> Optional[Path]: @@ -845,13 +853,6 @@ def _tenv_read(name: str, default: str = "") -> str: _BACKEND_IMAGE_KEYS = {b: f"{b}_image" for b in ("docker", "singularity", "modal", "daytona")} # (config key, default) pairs forwarded to _create_environment's container_config. -_CONTAINER_CONFIG_DEFAULTS = ( - ("container_cpu", 1), ("container_memory", 5120), ("container_disk", 51200), ("container_persistent", True), - ("modal_mode", "auto"), ("docker_volumes", []), ("docker_mount_cwd_to_workspace", False), - ("docker_forward_env", []), ("docker_env", {}), ("docker_run_as_host_user", False), ("docker_extra_args", []), - ("docker_shm_size", "1g"), ("docker_persist_across_processes", True), ("docker_shared_container_key", ""), - ("docker_orphan_reaper", True), -) # Single-line POSIX probe; `2>/dev/null` keeps a missing binary from polluting output. _BACKEND_PROBE_CMD = ( "printf 'os=%s\\nkernel=%s\\nhome=%s\\ncwd=%s\\nuser=%s\\n' \"$(uname -s 2>/dev/null || echo unknown)\" " @@ -862,31 +863,33 @@ _BACKEND_PROBE_CMD = ( def _run_backend_probe(env_type: str, terminal_tool) -> str: """Execute the probe command inside a freshly built backend; "" when it yields nothing.""" - from tools.terminal_tool_backends import _create_environment, _ssh_config_from_config + from tools.terminal_tool_backends import _container_config_from_config, _create_environment, _ssh_config_from_config from tools.terminal_tool_lifecycle import _cleanup_env config = terminal_tool._get_env_config() - # Mirrors tools/terminal_tool.py's live-command assembly (`_create_environment` is the factory). + # Same container_config shaper as the live terminal path: a private copy of the key table here + # drifted (no docker_network) and gave the probe a bridge-networked container under lockdown. env = _create_environment( env_type=env_type, image=config.get(_BACKEND_IMAGE_KEYS[env_type], "") if env_type in _BACKEND_IMAGE_KEYS else "", cwd=config.get("cwd", ""), timeout=config.get("timeout", 180), ssh_config=_ssh_config_from_config(config) if env_type == "ssh" else None, - container_config=({k: config.get(k, d) for k, d in _CONTAINER_CONFIG_DEFAULTS} + container_config=(_container_config_from_config(config) if terminal_tool._is_container_backend(env_type) else None), task_id="prompt-backend-probe", host_cwd=config.get("host_cwd"), + # Only ssh honors this: an isolated ControlMaster socket and no remote dir setup / file sync / + # snapshot. A normal SSHEnvironment would upload the whole ~/.hermes tree just to run `uname`, + # and its later __del__ would sync_back() and close the master shared with the agent's own env. + probe_only=True, ) try: result = env.execute(_BACKEND_PROBE_CMD, timeout=4) finally: # One-shot `uname`; without teardown the backend leaves a second idle sandbox # (task_id="prompt-backend-probe") running for the whole process next to the agent's own. - # ssh is left alone: no task-scoped sandbox, and its cleanup() closes a ControlMaster socket - # (keyed by user@host:port) shared with the agent's real environment; ControlPersist expires it. - if env_type != "ssh": - try: - _cleanup_env(env, force_remove=True) - except Exception: - logger.debug("Backend probe cleanup failed", exc_info=True) + try: + _cleanup_env(env, force_remove=True) + except Exception: + logger.debug("Backend probe cleanup failed", exc_info=True) if result.get("returncode") != 0: logger.debug("Backend probe returned non-zero: %r", result) return "" @@ -1451,6 +1454,11 @@ def load_soul_md(context_length: Optional[int] = None, home_override: "Path | No return None try: content = (_read_text_with_timeout(soul_path) or "").strip() + if content: + # Plugin-era desktop builds appended a frozen Bot Mode roster to SOUL.md; the server + # now injects the live section in Bot Chat only, so the copy is dead weight everywhere. + from tools.bot_mode_probe import strip_legacy_protocol + content = strip_legacy_protocol(content).strip() if not content: return None return _truncate_content(_scan_context_content(content, "SOUL.md"), "SOUL.md", context_length=context_length, diff --git a/agent/session_activity.py b/agent/session_activity.py index ddd04c3ab7..1ab19ef978 100644 --- a/agent/session_activity.py +++ b/agent/session_activity.py @@ -4,6 +4,7 @@ only (notification, timeout, kill and retry policy live elsewhere). Provenance i from __future__ import annotations +import sys import time from contextlib import suppress from enum import Enum @@ -45,6 +46,21 @@ def normalize_activity_provenance(provenance: Optional[ActivityProvenance | str] return ActivityProvenance.UNKNOWN +def format_iteration_progress(api_call_count: Any, max_iterations: Any) -> str: + """``iteration N/M`` for user-facing status lines, or ``iteration N`` when the cap is unbounded. + + ``AIAgent.max_iterations`` defaults to ``sys.maxsize`` (unlimited), so printing the pair verbatim + shows ``iteration 3/9223372036854775807`` in busy acks, heartbeats and timeout diagnostics (#102806). + """ + try: + cap = int(max_iterations) + except (TypeError, ValueError): + cap = sys.maxsize + if cap >= sys.maxsize: + return f"iteration {api_call_count}" + return f"iteration {api_call_count}/{cap}" + + def reset_session_activity_persist_window(agent: Any) -> None: """Clear the persist rate-limit so the next stamp writes through (terminal compression labels must not stick on mid-compress text).""" with suppress(Exception): diff --git a/agent/system_prompt.py b/agent/system_prompt.py index c3e51f0050..9bd52f03fc 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -538,12 +538,28 @@ def _alibaba_identity_part(agent: Any) -> List[str]: def _coding_parts(agent: Any) -> Tuple[List[str], List[str], List[str]]: """``(prefix, workspace, trailing)`` coding-posture blocks; all empty - without tools or when probing fails (it must never block prompt build).""" + without tools or when probing fails (it must never block prompt build). + + The workspace block is a live git probe that leads the context tier, ahead of the whole + volatile band; re-probing at the compaction rebuild re-emits different bytes for any + repo that moved and defeats the keep-prompt fast path. So the bytes are pinned per + session on the agent, keyed by the resolved cwd (a gateway serves many cwds), and + replayed on rebuilds; ``reset_session_state`` drops the pin at a session boundary. + """ try: from agent.coding_context import coding_system_prompt_parts - if agent.valid_tool_names: - return coding_system_prompt_parts(platform=agent.platform, cwd=resolve_context_cwd(), - model=agent.model, valid_tool_names=agent.valid_tool_names) + if not agent.valid_tool_names: + return [], [], [] + cwd = resolve_context_cwd() + cwd_key = str(cwd) if cwd is not None else "" + pinned = getattr(agent, "_frozen_workspace_snapshot", None) + # "" is a real pinned value (no workspace here) — only a cwd mismatch re-probes. + replay = pinned[1] if pinned is not None and pinned[0] == cwd_key else None + parts = coding_system_prompt_parts(platform=agent.platform, cwd=cwd, model=agent.model, + valid_tool_names=agent.valid_tool_names, workspace_block=replay) + if replay is None: + agent._frozen_workspace_snapshot = (cwd_key, parts[1][0] if parts[1] else "") + return parts except Exception: pass return [], [], [] diff --git a/agent/tool_executor.py b/agent/tool_executor.py index a7fbd13384..799199c10f 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -769,6 +769,15 @@ def _resolve_sequential_tool_timeout() -> float | None: return resolve_timeout("tools.sequential_call", default=_resolve_concurrent_tool_timeout()) +# Tools whose call blocks on a long-running operation that supervises its own liveness: no generic +# sequential deadline. ``delegate_task`` in a nested orchestrator blocks for the whole batch by design +# (children carry heartbeats, the stale monitor, and ``delegation.child_timeout_seconds``); under the +# 420 s deadline every real batch "timed out" while its children ran on as orphans, and the orchestrator +# spent the following hours polling transcripts (measured: 332 timeouts, ~$4k of orchestrator turns in +# one run). +_SEQUENTIAL_DEADLINE_EXEMPT_TOOLS = frozenset({"delegate_task"}) + + def _abandoned_sequential_result(agent, ref: _ToolCallRef, message: str, result_cls, **outcome) -> _ManagedToolResult: """Emit the terminal post_tool_call for a worker the sequential runner gave up on (timeout / interrupt) and wrap ``message`` in its marker ``result_cls``.""" @@ -815,7 +824,7 @@ def _run_sequential_tool_execution_middleware( """Run one sequential call on a worker thread under the concurrent executor's deadline. Interactive tools (``clarify``) own their wait via ``agent.clarify_timeout``; the generic deadline would report ``tool_timeout`` while the prompt is still live.""" - timeout_s = _resolve_sequential_tool_timeout() + timeout_s = None if function_name in _SEQUENTIAL_DEADLINE_EXEMPT_TOOLS else _resolve_sequential_tool_timeout() ref = _ToolCallRef(function_name, function_args, effective_task_id, tool_call_id, middleware_trace) kwargs = dict(ref.middleware_kwargs(), execute=execute, scope_block=scope_block, display_index=display_index) if function_name in _NEVER_PARALLEL_TOOLS: diff --git a/agent/trace_upload.py b/agent/trace_upload.py index 84868bacb0..31b3af5b84 100644 --- a/agent/trace_upload.py +++ b/agent/trace_upload.py @@ -210,17 +210,14 @@ def _do_upload(jsonl: str, *, token: str, session_id: str, dataset_name: str = D def load_session_messages(session_id: str, db_path=None) -> Tuple[List[Dict[str, Any]], Dict[str, Any]]: """``(messages, meta)`` from SQLite; ``meta`` is ``{}`` when the session row is missing (a live, untitled session may still have messages).""" - from hermes_state import SessionDB - db = SessionDB(db_path=db_path) if db_path else SessionDB() + from hermes_state_registry import acquire, release_or_close + db = acquire(db_path or None) try: resolved = db.resolve_session_id(session_id) or session_id meta = db.get_session(resolved) or {} return db.get_messages_as_conversation(resolved), meta finally: - try: - db.close() - except Exception: - logger.debug("Failed to close trace-upload SessionDB", exc_info=True) + release_or_close(db) def upload_session_trace( diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 723174f87d..5ead023e5b 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -146,8 +146,10 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> return thinking_config if effort not in {"minimal", "low", "medium", "high", "xhigh", "max", "ultra"}: effort = "medium" - # Gemini 3 Flash documents low/medium/high; Gemini 3 Pro only low/high. - if normalized_model.startswith(("gemini-3", "gemini-3.1")): + # Gemini 3 Flash documents low/medium/high thinking levels; Gemini 3 Pro + # is stricter (low/high). Clamp Hermes' wider effort set to what each + # family accepts so we never forward an undocumented level verbatim. + if normalized_model.startswith("gemini-3"): if "flash" in normalized_model: thinking_config["thinkingLevel"] = ( "low" if effort in {"minimal", "low"} else "high" if effort in _HIGH_EFFORTS else "medium" @@ -296,12 +298,19 @@ def _sanitize_message(msg: Any, strip_extra_content: bool) -> dict | None: """Sanitized copy of ``msg``, or None when nothing needs stripping. Drops persistence sidecars, ``_``-prefixed scaffolding markers, tool-call ``call_id`` / - ``response_item_id`` (and ``extra_content`` unless Gemini), and an assistant - ``tool_calls: []`` / ``null`` (strict providers reject both). + ``response_item_id`` (and ``extra_content`` unless Gemini), an assistant + ``tool_calls: []`` / ``null`` (strict providers reject both), and ``name`` + on tool results (schema-valid only on user/assistant messages; strict + providers reject it with ``contains item with unknown key name``). """ if not isinstance(msg, dict): return None strip_keys = [k for k in msg if k in _STRIP_MSG_KEYS or (isinstance(k, str) and k.startswith("_"))] + # ``name`` is schema-valid on user/assistant messages, so the removal is + # role-qualified: only tool results carry it illegally (strict providers + # reject with "contains item with unknown key name"). + if msg.get("role") == "tool" and "name" in msg: + strip_keys.append("name") out_msg = {k: v for k, v in msg.items() if k not in strip_keys} tool_calls = msg.get("tool_calls") copied_tool_calls = None diff --git a/agent/turn_context.py b/agent/turn_context.py index 82bba061a8..6dfad44e56 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -22,9 +22,9 @@ from agent.iteration_budget import IterationBudget from agent.memory_manager import build_memory_context_block from agent.memory_provider import is_trivial_prompt from agent.message_metadata import append_message, stamp_message_timestamp -from agent.model_metadata import ( - anchored_context_tokens, estimate_messages_tokens_rough, estimate_request_tokens_rough -) +from agent.model_metadata import estimate_messages_tokens_rough, estimate_request_tokens_rough +from agent.image_token_cost import bind_image_token_cost +from agent.usage_anchor import anchored_context_tokens, restore_usage_anchor logger = logging.getLogger(__name__) @@ -40,6 +40,7 @@ def _preflight_request_tokens( """Token estimate for automatic preflight compression: a valid provider usage anchor, else the checkpoint-pruned native wire payload, else the generic estimator.""" anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None)) + agent._request_pressure_anchored = anchored is not None if anchored is not None: return anchored tools = getattr(agent, "tools", None) or None @@ -525,13 +526,40 @@ def _stage_turn_user_message( def _hydrate_from_history(agent: Any, conversation_history: Optional[List[Any]]) -> None: - """Hydrate the todo store and per-session nudge counters from persisted history.""" + """Hydrate process-local state from persisted history on the first resumed turn.""" if not conversation_history: return if not agent._todo_store.has_items(): agent._hydrate_todo_store(conversation_history) - # Hydrate per-session nudge counters from persisted history. + # A live native checkpoint arms this latch while its response is captured. A + # restarted agent must recover the same one-response deferral before turn-start + # compression can rewrite the restored opaque checkpoint. Reuse the adapter's + # exact route/issuer/replay filtering and tolerate plugin compressors without the + # optional hook. if agent._user_turn_count == 0: + # A fresh process has no in-memory anchor; the persisted one is honored only while the + # restored transcript still carries the priced prefix (see agent/usage_anchor.py). + restore_usage_anchor(agent, conversation_history) + note_checkpoint = getattr( + getattr(agent, "context_compressor", None), + "note_native_compaction_checkpoint", + None, + ) + if callable(note_checkpoint): + try: + from agent.codex_responses_adapter import ( + has_replayable_native_compaction_checkpoint, + ) + + if has_replayable_native_compaction_checkpoint( + agent, conversation_history + ): + note_checkpoint() + except Exception: + logger.debug( + "restored native checkpoint hydration skipped", exc_info=True + ) + # Hydrate per-session nudge counters from persisted history. prior_user_turns = sum(1 for m in conversation_history if m.get("role") == "user") if prior_user_turns > 0: agent._user_turn_count = prior_user_turns @@ -680,7 +708,7 @@ def _memory_turn_start_and_prefetch(agent: Any, original_user_message: Any) -> s ext_prefetch_cache = "" with suppress(Exception): if not is_trivial_prompt(_query): - ext_prefetch_cache = agent._memory_manager.prefetch_all(_query) or "" + ext_prefetch_cache = agent._memory_manager.prefetch_all(_query, session_id=agent.session_id) or "" # Deterministic recall indicator via _emit_status so the model can't silently # drop injected memory. if ext_prefetch_cache: @@ -807,6 +835,8 @@ def build_turn_context( persist_user_platform_id, persist_user_display_kind, persist_user_display_metadata, ) _hydrate_from_history(agent, conversation_history) + # Every estimator this turn prices images at the cost learned from this model's real usage. + bind_image_token_cost(agent) # Append the user message now that close persistence is safe. append_message(messages, user_msg) current_turn_user_idx = len(messages) - 1 diff --git a/agent/turn_context_compaction.py b/agent/turn_context_compaction.py index a9f8a9b703..590946c6b9 100644 --- a/agent/turn_context_compaction.py +++ b/agent/turn_context_compaction.py @@ -156,6 +156,11 @@ def _idle_compaction( if _idle_gap < _idle_after: return _compressor = agent.context_compressor + # A live or restored native checkpoint must reach its issuer once so real usage, + # rather than an opaque ciphertext estimate, decides whether local compression is + # still needed. Threshold and post-tool preflight honor the same latch. + if bool(getattr(_compressor, "awaiting_real_usage_after_compression", False)): + return # Route-aware pressure: on compacted native-Codex sessions the durable figure # overstates the wire, so reuse the preflight estimator. _idle_tokens = _tc._preflight_request_tokens( @@ -253,7 +258,8 @@ def _preflight_compression( # snapshot may arm the interrupted-turn rollback. if isinstance(_snapshot_val, int) and not isinstance(_snapshot_val, bool): agent._turn_preflight_display_snapshot = _snapshot_val - _preflight_deferred = getattr( + # An anchored figure is real usage + delta: never deferred. + _preflight_deferred = not getattr(agent, "_request_pressure_anchored", False) and getattr( _compressor, "should_defer_preflight_to_real_usage", lambda _tokens: False )(_preflight_tokens) _codex_native_auto = _codex_native_auto_compaction(agent) @@ -273,8 +279,8 @@ def _preflight_compression( _compress_block_reason = None if _preflight_deferred: logger.info( - "Skipping preflight compression: rough estimate ~%s >= %s, " - "but last real provider prompt was %s after compression", + "Skipping preflight compression: rough estimate ~%s >= %s is not anchored on " + "real usage (last real provider prompt %s); deferring to the next response", f"{_preflight_tokens:,}", f"{_compressor.threshold_tokens:,}", f"{_compressor.last_real_prompt_tokens:,}", ) diff --git a/agent/turn_iteration_prep.py b/agent/turn_iteration_prep.py index 1b4917758c..ff25f98ea7 100644 --- a/agent/turn_iteration_prep.py +++ b/agent/turn_iteration_prep.py @@ -39,6 +39,12 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera _INTERRUPT_SCAFFOLD_MARKER, _maybe_inject_run_budget_wrapup ) + # nous.anthropic_wire=auto: a wire switch decided from the previous response lands here, + # before this iteration's request is built and with nothing in flight. + if getattr(agent, "_nous_wire_pending", None): + from agent.nous_wire import apply_pending_wire_switch + apply_pending_wire_switch(agent) + # Fire step_callback for gateway hooks (agent:step event). if agent.step_callback is not None: try: @@ -50,6 +56,15 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera if agent._skill_nudge_interval > 0 and "skill_manage" in agent.valid_tool_names: agent._iters_since_skill += 1 + # Nous agent keys live ~1 h and a single turn can run for hours: adopt the keepalive's fresh + # key before the one in hand expires (local JWT exp read; no network unless inside the skew) + # instead of letting this iteration's request 401. With many agents sharing the hour that + # 401 was a storm, and the pool benched the sole credential for all of them. + try: + agent._adopt_nous_key_before_expiry() + except Exception: + logger.debug("Nous key pre-expiry adoption failed", exc_info=True) + # Drain a /steer sent during the last API call into the newest tool message so # it lands THIS iteration. Never put in a user message (breaks alternation). _pre_api_steer = agent._drain_pending_steer() diff --git a/agent/turn_preflight.py b/agent/turn_preflight.py index cc2a5ff011..557a02661a 100644 --- a/agent/turn_preflight.py +++ b/agent/turn_preflight.py @@ -263,32 +263,23 @@ def compress_after_tool_results( ) _compressor = agent.context_compressor - # Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens - # is the actual context size the provider reported plus the assistant turn — a tight lower bound for the - # next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves - # ample headroom; if tool results push past it, the next API call will report the real total and trigger - # compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage - # data), fall back to rough estimate to avoid missing compression. Without this, a session can grow - # unbounded after disconnects because should_compress(0) never fires. (#2153) - if _compressor.last_prompt_tokens > 0: + # Real usage decides: the anchor is the provider's last prompt count plus a rough delta for + # ONLY the tool results appended since (the raw last_prompt_tokens ignores them). Right after + # a compaction (-1 sentinel) there is no real count yet: never treat the schema-heavy rough + # figure as pressure. The whole-request rough estimate is the last resort (usage-less + # provider, post-disconnect, gateway restart), kept route-aware (#96995/#97602). + from agent.usage_anchor import anchored_context_tokens + + _anchored = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None)) + if _anchored is not None: + _real_tokens = _anchored + elif _compressor.last_prompt_tokens > 0: # Only prompt_tokens: thinking models inflate completion_tokens with - # reasoning that uses no context → premature compression. - # Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026) + # reasoning that uses no context → premature compression. (#12026) _real_tokens = _compressor.last_prompt_tokens elif _compressor.last_prompt_tokens == -1: - # Compression just ran, no API prompt count yet: don't treat a rough - # schema-heavy post-compression estimate as real context pressure. _real_tokens = 0 else: - # Include tool schemas (20-30K tokens the messages-only estimate misses) and - # stay route-aware: on a compacted native-Codex session the generic - # durable-history figure would false-trigger. - # Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate - # misses, which can skip compression past the configured threshold (#14695). Route-aware - # (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure - # overstates the wire and would false-trigger compression here exactly like the pre-API guard — this - # fallback runs precisely when no provider usage is available (post-disconnect / gateway restart), - # the unanchored case from #97602's repro. _real_tokens = _midturn_request_pressure_tokens( agent, messages, active_system_prompt or "", estimate_request_tokens_rough(messages, tools=agent.tools or None), @@ -297,6 +288,9 @@ def compress_after_tool_results( if ( agent.compression_enabled and compression_attempts < max_compression_attempts + and not bool( + getattr(_compressor, "awaiting_real_usage_after_compression", False) + ) and _compressor.should_compress(_real_tokens) ): compression_attempts += 1 diff --git a/agent/turn_preflight_gate.py b/agent/turn_preflight_gate.py index 9982cd1afd..b01224107c 100644 --- a/agent/turn_preflight_gate.py +++ b/agent/turn_preflight_gate.py @@ -90,8 +90,11 @@ def run_preflight_gate( return run_preflight_compression( agent, v, compressor=_compressor, request_pressure_tokens=request_pressure_tokens, provider_overflow_preflight=_provider_overflow_preflight, - defer_preflight=getattr( - _compressor, "should_defer_preflight_to_real_usage", lambda _t: False + # An anchored figure is real usage + delta: never deferred. Only a whole-context rough + # estimate waits for the provider's count. + defer_preflight=( + (lambda _t: False) if getattr(agent, "_request_pressure_anchored", False) + else getattr(_compressor, "should_defer_preflight_to_real_usage", lambda _t: False) ), moa_prepared_request=_moa_prepared_request, system_message=system_message, user_message=user_message, max_compression_attempts=max_compression_attempts, diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index 4dd2f1b5a1..be2a46378f 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -892,7 +892,8 @@ def log_api_error_attempt( if agent._is_openrouter_url() and "support tool use" in error_msg: _blines(agent, f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings.") - if agent.providers_allowed: + from agent.chat_completion_helpers import _provider_preferences_for_agent + if _provider_preferences_for_agent(agent).get("only"): _blines( agent, " Your provider_routing.only restriction is filtering out tool-capable providers.", @@ -973,30 +974,44 @@ def compute_error_backoff( agent: Any, api_error: Exception, *, retry_count: int, max_retries: int, is_rate_limited: bool, is_zai_coding_overload: bool, base_url: Any, model: Any, ) -> float: - """Pick the wait before the next API retry and announce it. Retry-After wins for rate - limits (capped at 600s: Anthropic Tier 1 buckets reset in ~171s, so a 120s cap re-tripped - the limit); otherwise jittered backoff, replaced by the adaptive policy for 429s / Z.AI - overloads. Normal retries are buffered; long Z.AI Coding waits surface immediately.""" + """Pick the wait before the next API retry and announce it. Retry-After wins for + rate limits and any other retryable error (capped at 600s: Anthropic Tier 1 buckets + reset in ~171s, so a 120s cap re-tripped the limit); otherwise jittered backoff, + replaced by the adaptive policy for 429s / Z.AI overloads. Normal retries are + buffered; long Z.AI Coding waits surface immediately.""" # Imported lazily so tests that patch ``agent.retry_utils.jittered_backoff`` / # ``adaptive_rate_limit_backoff`` (incl. the run_agent conftest fast-backoff fixture) intercept. - from agent.retry_utils import adaptive_rate_limit_backoff, jittered_backoff + from agent.retry_utils import adaptive_rate_limit_backoff, jittered_backoff, parse_retry_after_seconds - _retry_after = None - _resp_headers = getattr(getattr(api_error, "response", None), "headers", None) if is_rate_limited else None - if _resp_headers and hasattr(_resp_headers, "get"): - _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") - if _ra_raw: - try: - # Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap - # caused us to retry before the actual reset window and re-trip the limit. 600s covers all - # realistic provider reset windows while still rejecting pathological values. (#26293) - _retry_after = min(float(_ra_raw), 600) - except (TypeError, ValueError): - pass - wait_time = _retry_after if _retry_after else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) + # Respect Retry-After on every retryable provider error, not just 429s. Retryable + # 5xx responses (e.g. Cloudflare 520/524) also carry the header or a structured + # ``retry_after`` problem-detail body field; ignoring either turns an origin + # outage into a retry storm. + _retry_after = parse_retry_after_seconds( + getattr(getattr(api_error, "response", None), "headers", None) + ) + if _retry_after is None: + _error_body = getattr(api_error, "body", None) + if isinstance(_error_body, dict): + # Some providers nest it as error.retry_after (the same unwrap + # extract_api_error_context uses), others put it at the top level. + _nested = _error_body.get("error") + _payload = _nested if isinstance(_nested, dict) else _error_body + _retry_after = parse_retry_after_seconds(_payload.get("retry_after")) + if _retry_after is not None: + # Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap + # caused us to retry before the actual reset window and re-trip the limit. 600s covers all + # realistic provider reset windows while still rejecting pathological values. (#26293) + _retry_after = min(_retry_after, 600) + if _retry_after <= 0: + # A zero/expired cooldown (retry-after: 0, or an HTTP-date in the + # past, which the parser clamps to 0.0) carries no usable wait — + # treat it as absent so we never hot-loop the provider. + _retry_after = None + wait_time = _retry_after if _retry_after is not None else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) _backoff_policy = None _adaptive = is_rate_limited or is_zai_coding_overload - if _adaptive and not _retry_after: + if _adaptive and _retry_after is None: wait_time, _backoff_policy = adaptive_rate_limit_backoff( retry_count, base_url=str(base_url), model=model, error=api_error, default_wait=wait_time, ) @@ -1009,7 +1024,16 @@ def compute_error_backoff( else: agent._buffer_status(_rate_limit_status) else: - agent._buffer_status(f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})...") + _retry_status = ( + f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})..." + ) + if _retry_after is not None and _retry_after > 60: + # A 5xx Retry-After can now reach the 600s cap; buffering that wait + # would leave the user silent for minutes, so surface long provider + # cooldowns immediately (mirrors the zai_coding_overload_long path). + agent._emit_status(_retry_status) + else: + agent._buffer_status(_retry_status) logger.warning( "Retrying API call in %ss (attempt %s/%s) %s policy=%s error=%s", wait_time, retry_count, max_retries, agent._client_log_context(), diff --git a/agent/turn_request_assembly.py b/agent/turn_request_assembly.py index 8eb9beadb9..c4d98ad211 100644 --- a/agent/turn_request_assembly.py +++ b/agent/turn_request_assembly.py @@ -14,7 +14,7 @@ import logging from typing import Any from agent.message_sanitization import _sanitize_messages_surrogates -from agent.model_metadata import anchored_context_tokens +from agent.usage_anchor import anchored_context_tokens from agent.prompt_caching import build_prompt_cache_plan, effective_cache_ttl from agent.turn_context import build_api_messages @@ -245,6 +245,7 @@ def assemble_api_request( # Usage-anchored override: real prompt_tokens (incl. system + tool schemas) + # delta estimate replaces the whole-history heuristic when the anchor is fresh. _anchored_pressure = anchored_context_tokens(messages, getattr(agent, "_usage_anchor", None)) + agent._request_pressure_anchored = _anchored_pressure is not None if _anchored_pressure is not None: request_pressure_tokens = _anchored_pressure else: @@ -254,11 +255,6 @@ def assemble_api_request( request_pressure_tokens = _pressure_with_real_floor( agent.context_compressor, request_pressure_tokens ) - # Stash the rough estimate so update_from_response() can pair it with the real - # count (should_defer_preflight_to_real_usage). getattr: test doubles lack it. - _note_rough = getattr(agent.context_compressor, "note_request_rough_estimate", None) - if callable(_note_rough): - _note_rough(request_pressure_tokens) return AssembledRequest( "fallthrough", api_messages, tools_for_api, _moa_prepared_request, pending_moa_prepared_request, approx_tokens, request_pressure_tokens, approx_tokens * 4, diff --git a/agent/turn_usage.py b/agent/turn_usage.py index 131e2cd348..f3c43a690c 100644 --- a/agent/turn_usage.py +++ b/agent/turn_usage.py @@ -15,7 +15,8 @@ from contextlib import suppress from dataclasses import dataclass from typing import Any, Dict, List -from agent.model_metadata import capture_usage_anchor +from agent.image_token_cost import calibrate_from_usage +from agent.usage_anchor import capture_usage_anchor, set_usage_anchor from agent.usage_pricing import estimate_usage_cost, normalize_usage logger = logging.getLogger("agent.conversation_loop") @@ -81,6 +82,9 @@ def record_response_usage( # pending verdict so later readings aren't charged to it and # preflight deferral isn't latched indefinitely. compressor.update_from_response({}) + _note_usage_less = getattr(compressor, "note_usage_less_response", None) + if callable(_note_usage_less): + _note_usage_less() logger.info( "API call #%d: model=%s provider=%s in=? out=? total=? latency=%.1fs usage=unavailable", agent.session_api_calls, agent.model, agent.provider or "unknown", api_duration, @@ -116,13 +120,14 @@ def record_response_usage( # transcript (main-loop ONLY; MoA uses pre-fold aggregator usage). The display meter # anchors on the turn's FIRST response: later same-turn responses inflate # prompt_tokens with replayed thinking. Display-only; compression math uses real usage. + # The provider just priced this request exactly: if the delta since the previous anchor + # introduced images, the residual is their real per-image cost (learned before re-anchoring). + calibrate_from_usage(agent, messages, aggregator_usage.prompt_tokens) _new_anchor = capture_usage_anchor( aggregator_usage.prompt_tokens, aggregator_usage.output_tokens, messages ) if _new_anchor is not None: - agent._usage_anchor = _new_anchor - if api_call_count == 1: - agent._turn_base_usage_anchor = _new_anchor + set_usage_anchor(agent, _new_anchor, turn_base=api_call_count == 1) _compression_threshold = int(getattr(compressor, "threshold_tokens", 0) or 0) if _loop_mod()._should_rearm_compression_budget( compression_attempts, completed_compaction_pending=_completed_compaction_pending, @@ -143,6 +148,9 @@ def record_response_usage( # Stash canonical usage for on_turn_complete(); keep the latest call's. agent._last_turn_usage = dict(usage_dict) + # The parent's CURRENT prompt size for headroom math (delegate summary budgets): the + # aggregator's own prompt, never the MoA-folded total (advisor prompts are not in this context). + agent._last_prompt_size_tokens = int(aggregator_usage.prompt_tokens or 0) # Persist only provider-confirmed context lengths, not probe tiers. if getattr(compressor, "_context_probed", False): @@ -175,12 +183,28 @@ def record_response_usage( _cache_pct = "" if canonical_usage.cache_read_tokens and prompt_tokens: _cache_pct = f" cache={canonical_usage.cache_read_tokens}/{prompt_tokens} ({100*canonical_usage.cache_read_tokens/prompt_tokens:.0f}%)" + # write= is the money (cache writes cost 50x a read); id= is what a provider needs to look the + # request up; upstream= is who actually served it when the route reports that (OpenRouter's + # `provider`). Diagnosing the 1,393-agent run's cache misses took a DB join and a live probe + # because none of the three were on this line. + if canonical_usage.cache_write_tokens: + _cache_pct += f" write={canonical_usage.cache_write_tokens}" + _rid = getattr(response, "id", None) + _ident = f" id={_rid}" if isinstance(_rid, str) and _rid else "" + _upstream = getattr(response, "provider", None) + if isinstance(_upstream, str) and _upstream: + _ident += f" upstream={_upstream}" logger.info( - "API call #%d: model=%s provider=%s in=%d out=%d total=%d latency=%.1fs%s", + "API call #%d: model=%s provider=%s in=%d out=%d total=%d latency=%.1fs%s%s", agent.session_api_calls, agent.model, agent.provider or "unknown", prompt_tokens, completion_tokens, total_tokens, - api_duration, _cache_pct, + api_duration, _cache_pct, _ident, ) + # nous.anthropic_wire=auto: the session's wire is decided once, from this first response. + if agent.session_api_calls == 1 and (agent.provider or "") == "nous": + with suppress(Exception): + from agent.nous_wire import maybe_switch_wire_after_first_response + maybe_switch_wire_after_first_response(agent, response, agent.session_api_calls) # MoA: agent.model/provider are the virtual preset/"moa" with no pricing entry, silently # dropping aggregator spend. Price at the REAL model/provider from the aggregator slot. diff --git a/agent/usage_anchor.py b/agent/usage_anchor.py new file mode 100644 index 0000000000..d93930c720 --- /dev/null +++ b/agent/usage_anchor.py @@ -0,0 +1,165 @@ +"""Usage-anchored token accounting: the provider's real ``usage.prompt_tokens`` is the only +authoritative context size; the local ``bytes/4`` estimate covers ONLY messages appended since. + +An anchor = provider usage at capture + a snapshot of the transcript position it priced: +``base_count`` (len(messages) at capture; the reply is not yet appended and is covered by +``completion_tokens``, so the delta walk skips an assistant row at that index), ``base_last_role`` +and ``base_last_fp`` (content fingerprint of the last priced message; compaction, splices and +rewinds replace it → anchor fails closed → full estimation until the next real reading). + +The fingerprint (not ``id()``) is the identity: the gateway re-reads the transcript from the DB +every turn and a resumed session runs in a fresh process, so object identity is never stable +across the surfaces where the estimate mattered most (#99421, #104462). The anchor also persists +on the session row (``model_config._usage_anchor``) so a restarted process can restore it; a +restored anchor is honored only while the durable transcript still matches its fingerprint. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +from typing import Any, Dict, List, Optional + +logger = logging.getLogger(__name__) + +USAGE_ANCHOR_MODEL_CONFIG_KEY = "_usage_anchor" + +# Identity of a priced message = the provider-visible fields that round-trip the session DB +# byte-for-byte. Display/persistence metadata (timestamps, row ids, display kinds) is rewritten +# on reload and would only ever fail the match closed. +_FINGERPRINT_KEYS = ("role", "content", "api_content", "tool_call_id", "tool_calls") + + +def message_fingerprint(msg: Any) -> Optional[str]: + """Stable digest of one transcript message over its provider-visible, persisted fields.""" + if not isinstance(msg, dict): + return None + payload = {k: msg.get(k) for k in _FINGERPRINT_KEYS if msg.get(k) is not None} + try: + raw = json.dumps(payload, sort_keys=True, default=str, ensure_ascii=True, separators=(",", ":")) + except (TypeError, ValueError): + raw = repr(sorted(payload.items())) + return hashlib.sha256(raw.encode("utf-8", "replace")).hexdigest() + + +def capture_usage_anchor(prompt_tokens: Any, completion_tokens: Any, messages: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]: + """Build a usage anchor from provider-reported usage, or None when usage is unusable.""" + try: + pt = int(prompt_tokens or 0) + ct = int(completion_tokens or 0) + except (TypeError, ValueError): + return None + if pt <= 0 or not isinstance(messages, list) or not messages: + return None # some endpoints omit usage — caller keeps its anchor + last = messages[-1] + return { + "prompt_tokens": pt, + "completion_tokens": max(0, ct), + "base_count": len(messages), + "base_last_role": last.get("role") if isinstance(last, dict) else None, + "base_last_fp": message_fingerprint(last), + } + + +def _anchor_matches(messages: List[Dict[str, Any]], anchor: Dict[str, Any]) -> bool: + try: + base_count = int(anchor.get("base_count") or 0) + except (TypeError, ValueError): + return False + if base_count <= 0 or len(messages) < base_count: + return False + base_msg = messages[base_count - 1] + if not isinstance(base_msg, dict) or base_msg.get("role") != anchor.get("base_last_role"): + return False + fp = anchor.get("base_last_fp") + return isinstance(fp, str) and bool(fp) and message_fingerprint(base_msg) == fp + + +def anchored_context_tokens(messages: List[Dict[str, Any]], anchor: Optional[Dict[str, Any]], *, charge_stale_thinking: bool = True) -> Optional[int]: + """Anchored prompt+completion tokens plus a rough estimate of ONLY the messages appended since; + None when the anchor is missing or stale. The anchored response's own reply is skipped (already + in completion_tokens). ``charge_stale_thinking`` is forwarded to the delta estimate.""" + if not isinstance(anchor, dict) or not isinstance(messages, list) or not _anchor_matches(messages, anchor): + return None + from agent.model_metadata import estimate_messages_tokens_rough + + total = int(anchor["prompt_tokens"]) + int(anchor.get("completion_tokens") or 0) + delta = messages[int(anchor["base_count"]):] + if delta and isinstance(delta[0], dict) and delta[0].get("role") == "assistant": + delta = delta[1:] + if delta: + total += estimate_messages_tokens_rough(delta, charge_stale_thinking=charge_stale_thinking) + return total + + +def _serialize(anchor: Any) -> Optional[Dict[str, Any]]: + if not isinstance(anchor, dict): + return None + try: + pt, ct, base_count = (int(anchor.get(k) or 0) for k in ("prompt_tokens", "completion_tokens", "base_count")) + except (TypeError, ValueError): + return None + fp, role = anchor.get("base_last_fp"), anchor.get("base_last_role") + if pt <= 0 or base_count <= 0 or not isinstance(fp, str) or not fp: + return None + return {"prompt_tokens": pt, "completion_tokens": max(0, ct), "base_count": base_count, + "base_last_role": role if isinstance(role, str) else None, "base_last_fp": fp} + + +def persist_usage_anchor(agent: Any, anchor: Optional[Dict[str, Any]]) -> None: + """Write (or clear, ``None``) the session row's anchor blob. Best-effort: the row may not exist yet.""" + if getattr(agent, "_persist_disabled", False): + return + session_id = getattr(agent, "session_id", None) + patcher = getattr(getattr(agent, "_session_db", None), "patch_session_model_config", None) + if not session_id or not callable(patcher): + return + try: + patcher(session_id, {USAGE_ANCHOR_MODEL_CONFIG_KEY: _serialize(anchor)}) + except Exception: + logger.debug("usage anchor persist failed", exc_info=True) + + +def set_usage_anchor(agent: Any, anchor: Optional[Dict[str, Any]], *, turn_base: bool = False) -> None: + """Install ``anchor`` on the agent (``None`` clears) and mirror it to the session row.""" + agent._usage_anchor = anchor + if turn_base or anchor is None: + agent._turn_base_usage_anchor = anchor + persist_usage_anchor(agent, anchor) + + +def restore_usage_anchor(agent: Any, conversation_history: Optional[List[Dict[str, Any]]]) -> None: + """On a resumed session, adopt the persisted anchor when ``conversation_history`` still carries + the priced prefix; otherwise clear the stale blob so it can never suppress compression.""" + if getattr(agent, "_usage_anchor", None) is not None or getattr(agent, "_persist_disabled", False): + return + session_id = getattr(agent, "session_id", None) + getter = getattr(getattr(agent, "_session_db", None), "get_session_model_config_value", None) + if not session_id or not callable(getter) or not isinstance(conversation_history, list): + return + try: + anchor = _serialize(getter(session_id, USAGE_ANCHOR_MODEL_CONFIG_KEY, None)) + except Exception: + logger.debug("usage anchor load failed", exc_info=True) + return + if anchor is None: + return + if _anchor_matches(conversation_history, anchor): + agent._usage_anchor = anchor + else: + persist_usage_anchor(agent, None) + + +def persisted_anchor_tokens(session_db: Any, session_id: Any, messages: Any) -> Optional[int]: + """Anchored token figure from the session row's persisted anchor, for callers without a live + agent (gateway hygiene); None when absent, unreadable, or stale against ``messages``.""" + getter = getattr(session_db, "get_session_model_config_value", None) + if not session_id or not callable(getter) or not isinstance(messages, list): + return None + try: + anchor = _serialize(getter(session_id, USAGE_ANCHOR_MODEL_CONFIG_KEY, None)) + except Exception: + logger.debug("usage anchor load failed", exc_info=True) + return None + return anchored_context_tokens(messages, anchor) if anchor else None diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 0f73ce36f9..9830e2e28b 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -191,6 +191,9 @@ _SNAPSHOTS: tuple[tuple[str, Optional[str], str, dict], ...] = ( ("deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"): ("0.14", "0.28", "0.0028"), "deepseek-v4-pro": ("0.435", "0.87", "0.003625"), }), + ("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-09-02", { + ("gemini-3.8-flash", "gemini-3.7-flash"): ("0.75", "3.75", "0.075"), + }), ("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-07-28", { "gemini-3.6-flash": ("1.50", "7.50", "0.15"), "gemini-3.5-flash-lite": ("0.30", "2.50", "0.03"), }), diff --git a/agent/verify/runner.py b/agent/verify/runner.py index eb6f746cfe..4016c1638b 100644 --- a/agent/verify/runner.py +++ b/agent/verify/runner.py @@ -182,6 +182,30 @@ def _run_start_phase( return ReadinessResult(url, ready, status, time.monotonic() - started, error, _tail(output)) +def _compose_live_state_reason(root: Path) -> str | None: + """Why ``docker compose build``/``up`` must not run at *root*, or ``None`` to proceed. + + Read-only ``docker compose ps`` probe. Only a missing docker binary proceeds -- the + build phase would fail the same way, so there is nothing to protect. A hung daemon + or a non-zero probe refuses: containers may be live and unobservable, which is + exactly the #103567 loss window. + """ + try: + result = subprocess.run( + ["docker", "compose", "ps", "--status", "running", "--format", "{{.Name}}"], + cwd=root, capture_output=True, text=True, timeout=15, stdin=subprocess.DEVNULL, + ) + except FileNotFoundError: + return None + except subprocess.TimeoutExpired: + return "docker compose ps timed out after 15s; live containers cannot be ruled out" + if result.returncode != 0: + detail = (result.stderr or result.stdout or "").strip().splitlines() + return f"docker compose ps failed (exit {result.returncode}): {detail[-1] if detail else 'no output'}" + names = [line for line in result.stdout.splitlines() if line.strip()] + return f"this compose project already has running container(s): {', '.join(names)}" if names else None + + def run_verify( root: Path, recipe: Recipe, phases: tuple[str, ...] | list[str] | None = None, phase_timeout: float = DEFAULT_PHASE_TIMEOUT, ready_timeout: float = DEFAULT_READY_TIMEOUT, @@ -189,11 +213,34 @@ def run_verify( on_output: Callable[[str], None] | None = None, ) -> VerifyResult: """Run the selected command phases sequentially, then (unless ``skip_start`` or a - phase failed) boot ``recipe.start``, poll readiness, and tear the process group down.""" + phase failed) boot ``recipe.start``, poll readiness, and tear the process group down. + + A ``compose`` recipe refuses outright when the project already has running + containers: ``docker compose build`` + ``up`` replaces them on an image-hash + change, destroying any container-local state they carry -- this has caused a + real state-loss incident (#103567). The check is best-effort and read-only + (``docker compose ps``); when it cannot run at all, verify proceeds rather than + blocking on an unrelated environment gap, matching every other recipe kind's + behavior when its own tooling is unavailable.""" root = Path(root) selected = tuple(phases) if phases else PHASE_ORDER + ("start",) result = VerifyResult(recipe_name=recipe.name) + mutating = ("build" in selected) or ("start" in selected and not skip_start) + if recipe.kind == "compose" and mutating: + reason = _compose_live_state_reason(root) + if reason: + result.phases.append(PhaseResult( + phase="build", command=recipe.build[0] if recipe.build else "docker compose build", + exit_code=1, duration=0.0, output_tail=( + f"Refusing to run: {reason}. `docker compose build` + `up` would replace " + "live containers on an image-hash change, destroying any container-local " + "state they carry. If you intend to rebuild this live deployment, run " + "`docker compose build`/`up` yourself." + ), + )) + return result + for phase in PHASE_ORDER: if phase not in selected: continue diff --git a/apps/desktop/electron/backend-dial-claim.test.ts b/apps/desktop/electron/backend-dial-claim.test.ts index 6aa7ce1353..30c7e8195c 100644 --- a/apps/desktop/electron/backend-dial-claim.test.ts +++ b/apps/desktop/electron/backend-dial-claim.test.ts @@ -126,10 +126,10 @@ describe('main.ts wiring for #90812', () => { it('routes the profile-scoped dial IPC through the single-owner claim', () => { const handlerStart = mainSource.indexOf("ipcMain.handle('hermes:connection', ") expect(handlerStart).toBeGreaterThan(-1) - const body = mainSource.slice(handlerStart, handlerStart + 900) + const body = mainSource.slice(handlerStart, handlerStart + 1200) expect(body).toContain('backendDialClaims.run(') - expect(body).toContain('ensureBackend(profile)') + expect(body).toContain('ensureBackend(profile, { spawnPriority })') }) it('routes the registry-scoped dial IPC through the claim keyed by backendScopeKey(connectionId, profile)', () => { @@ -137,8 +137,9 @@ describe('main.ts wiring for #90812', () => { expect(handlerStart).toBeGreaterThan(-1) const body = mainSource.slice(handlerStart, handlerStart + 1_200) - expect(body).toContain('backendDialClaims.run(backendScopeKey(id, profile)') - expect(body).toContain('ensureRegistryBackend(id, profile)') + expect(body).toContain('const scopeKey = backendScopeKey(id, profile)') + expect(body).toContain('backendDialClaims.run(scopeKey, ') + expect(body).toContain("ensureRegistryBackend(id, profile, '', { spawnPriority })") }) // The four IPC/probe surfaces below call ensureRegistryBackend()/ensureBackend() diff --git a/apps/desktop/electron/connection-registry.test.ts b/apps/desktop/electron/connection-registry.test.ts index 261671b816..d92428af69 100644 --- a/apps/desktop/electron/connection-registry.test.ts +++ b/apps/desktop/electron/connection-registry.test.ts @@ -1054,6 +1054,33 @@ test('token only persists on token-auth remotes; oauth/cloud drop it', () => { assert.equal(cloud.token, undefined) }) +test('an ssh entry keeps its session token through a label rename', () => { + // #103795, second half: saveRegistryConnection resolves the surviving + // envelope (resolvePersistedRemoteToken keeps the stored one when the + // editor sends no new value) and hands it to normalizeConnectionInput — + // whose ssh branch used to drop it, so renaming a connection wiped the live + // backend's reuse credential and re-armed the reap-and-respawn loop. + const stored = { + host: 'spark1', + id: 'spark', + kind: 'ssh' as const, + label: 'Spark', + port: 2222, + token: { enc: 'ssh-session-token' }, + user: 'tek' + } + + const merged = mergeConnectionInput( + { id: 'spark', kind: 'ssh', label: 'Spark (office)', token: stored.token }, + stored + ) + + const renamed = normalizeConnectionInput(merged, emptyRegistry()) + + assert.equal(renamed.label, 'Spark (office)') + assert.deepEqual(renamed.token, { enc: 'ssh-session-token' }) +}) + // --- mergeConnectionInput (edit inheritance) --- test('merge preserves fields the editor does not carry (org, ssh extras)', () => { @@ -1325,6 +1352,46 @@ test('normalizeRegistry falls back to Primary when the last-used source is missi assert.equal(registry.lastUsed, 'homelab') }) +test('normalizeRegistry keeps the persisted ssh session token across a cold read', () => { + // #103795: persistSshConnectionToken() writes the adopted per-serve token + // onto the ssh entry, but normalization rebuilt the entry from the DIAL + // fields alone and dropped it. The token then lived only in the mtime-keyed + // in-process cache, so the next launch dialed with an empty reuseToken, + // failed remote-lifecycle's `Boolean(reuseToken)` reuse gate, reaped a + // healthy owned backend and respawned it on a new port — while the renderer + // kept dialing the old token and got 403 forever. + const saved = { + version: REGISTRY_VERSION, + primary: 'spark', + connections: [ + { id: LOCAL_CONNECTION_ID, kind: 'local', label: 'This device' }, + { + id: 'spark', + kind: 'ssh', + label: 'Spark', + host: 'spark1', + user: 'tek', + port: 2222, + token: { enc: 'ssh-session-token' } + } + ] + } + + const registry = normalizeRegistry(saved) + const spark = registry.connections.find(connection => connection.id === 'spark') + + assert.deepEqual(spark?.token, { enc: 'ssh-session-token' }) + assert.equal(spark?.host, 'spark1') + + // Write → read → normalize again: the token must survive every cold read, + // not just the first. + const reread = normalizeRegistry(JSON.parse(JSON.stringify(registry))) + + assert.deepEqual(reread.connections.find(connection => connection.id === 'spark')?.token, { + enc: 'ssh-session-token' + }) +}) + // --- v1 → v2 migration --- test('migrate: v1 local-only config → local-only registry', () => { diff --git a/apps/desktop/electron/connection-registry.ts b/apps/desktop/electron/connection-registry.ts index 2875f53c66..a6a22c3d90 100644 --- a/apps/desktop/electron/connection-registry.ts +++ b/apps/desktop/electron/connection-registry.ts @@ -906,7 +906,16 @@ export function normalizeConnectionInput(input: ConnectionInput, registry: Conne throw new Error(`A connection to this SSH host already exists ("${sshDupe.label}").`) } - return { id, kind: 'ssh', label, ...sshFields } + const entry: RegistryConnection = { id, kind: 'ssh', label, ...sshFields } + + // Carry the adopted session-token envelope across edits (mirrors the remote + // branch): dropping it made a label rename wipe the backend's reuse + // credential and force the reap-and-respawn loop of #103795. + if (input.token !== undefined) { + entry.token = input.token + } + + return entry } if (kind === 'remote' || kind === 'cloud') { @@ -1191,6 +1200,14 @@ export function normalizeRegistry(raw: unknown): ConnectionRegistry { const { mode: _mode, ...sshFields } = ssh Object.assign(clean, sshFields) + + // normalizeSshConfig describes only the dial, so the token + // persistSshConnectionToken() adopted must be carried explicitly (as the + // remote/cloud branch does). Losing it on a cold read fails the + // remote-lifecycle reuse gate and reaps a healthy backend (#103795). + if (entry.token !== undefined) { + clean.token = entry.token + } } connections.push(clean) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 5116cdb7bc..ecbc3ebb71 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -287,7 +287,9 @@ import { import { selectPoolEvictions } from './pool-eviction' import { clampPoolLimits, parsePoolLimits, POOL_LIMITS_DEFAULTS } from './pool-limits' import { + isBackgroundSlotWaitTimeout, LocalBackendSpawnCoordinator, + type LocalBackendSpawnPriority, type LocalBackendSpawnRequest, releaseLocalBackendSlotAfterExit } from './pool-spawn-coordinator' @@ -1492,6 +1494,70 @@ const localBackendSpawnCoordinator = new LocalBackendSpawnCoordinator(poolLimits // the queued ticket fails before the renderer does and the user sees why. const POOL_SLOT_WAIT_MS = 30_000 +function spawnPriorityFrom(value: unknown): LocalBackendSpawnPriority { + return value === 'foreground' ? 'foreground' : 'background' +} + +// Foreground intent for a dial whose pool entry does not exist yet: a user +// click that joins an in-flight backendDialClaims claim never re-enters +// ensureBackend(), and the claim owner may still be awaiting poolStopper / +// registry resolution before backendPool.set(). The local spawn takes the mark +// right before its slot request; the IPC handler that set it clears it once +// the claim settles, so a dial that never reaches a slot request (primary +// route, remote scope, a guard rejection) cannot leave it for a later +// hydration spawn of the same key to pick up. +const pendingForegroundSpawns = new Set() + +function takeForegroundSpawn(...poolKeys: string[]): boolean { + let marked = false + + for (const poolKey of poolKeys) { + marked = pendingForegroundSpawns.delete(poolKey) || marked + } + + return marked +} + +// Upgrade a pooled entry (running, spawning, or queued for a slot) to +// foreground so a queued slot wait can take the reserved foreground slot. +function promotePoolEntry(entry: any): void { + entry.spawnPriority = 'foreground' + entry.localBackendSpawnRequest?.promote?.('foreground') +} + +// Land a spawn failure in desktop.log. A background slot-wait timeout is +// routine under a saturated pool (the next hydration pass retries), so it is +// logged as such instead of as a backend-start failure. +function logPoolSpawnFailure(label: string, error: unknown): void { + if (isBackgroundSlotWaitTimeout(error)) { + rememberLog(`Profile backend ${label} slot wait timed out (background); will retry on the next hydration`) + } else { + rememberLog( + `Hermes backend for profile ${label} failed to start: ${error instanceof Error ? error.message : String(error)}` + ) + } +} + +// Apply foreground intent to the dial claim for `scopeKey`: an entry already +// in the pool is promoted directly, otherwise the intent is marked for the +// spawn the claim owner is about to start. Returns the cleanup that clears a +// mark the dial never consumed. +function applySpawnPriority(scopeKey: string, spawnPriority: LocalBackendSpawnPriority): () => void { + if (spawnPriority !== 'foreground') { + return () => undefined + } + + const existing = backendPool.get(scopeKey) + + if (existing) { + promotePoolEntry(existing) + } else { + pendingForegroundSpawns.add(scopeKey) + } + + return () => void pendingForegroundSpawns.delete(scopeKey) +} + function poolMaxBackends() { return poolLimits.maxBackends } @@ -11382,8 +11448,9 @@ function profileRouteOptions(profile, request?) { // Resolve a backend connection for the given profile, per the routing table in // resolveProfileBackendRoute(). An empty / unknown profile resolves to the // primary, so legacy callers are unchanged. -async function ensureBackend(profile) { +async function ensureBackend(profile, opts: { spawnPriority?: LocalBackendSpawnPriority } = {}) { const key = profile && String(profile).trim() ? String(profile).trim() : primaryProfileKey() + const spawnPriority = spawnPriorityFrom(opts.spawnPriority) profileDeletionGate.assertCanStart(key) @@ -11415,6 +11482,11 @@ async function ensureBackend(profile) { if (existing) { existing.lastActiveAt = Date.now() + + if (spawnPriority === 'foreground') { + promotePoolEntry(existing) + } + const connection = await existing.connectionPromise setWslBridgeProfileState(key, connection.mode !== 'remote') @@ -11432,16 +11504,15 @@ async function ensureBackend(profile) { remoteBaseUrl: null, releaseLocalBackendSlot: null, localBackendSlotKey: null, - localBackendSpawnRequest: null + localBackendSpawnRequest: null, + spawnPriority } entry.connectionPromise = spawnPoolBackend(key, entry).catch(async error => { // Land the failure in desktop.log: without this a spawn that dies before // its child exists (guard rejection, runtime resolution) leaves no trace // beyond renderer-side rejections users never see in a bundle. - rememberLog( - `Hermes backend for profile "${key}" failed to start: ${error instanceof Error ? error.message : String(error)}` - ) + logPoolSpawnFailure(`"${key}"`, error) await teardownFailedLocalBackend(key, entry) throw error @@ -11462,7 +11533,13 @@ async function ensureBackend(profile) { // a genuinely-local child when the v1 mode says remote; non-local connections // pool under the composite key from backendScopeKey() and reuse the same pool // entry lifecycle (LRU, idle reaper, touch) as per-profile local backends. -async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrelation = '') { +async function ensureRegistryBackend( + connectionId, + profile, + managedUpdateCorrelation = '', + opts: { spawnPriority?: LocalBackendSpawnPriority } = {} +) { + const spawnPriority = spawnPriorityFrom(opts.spawnPriority) const registry = readDesktopConnectionsRegistry() const id = String(connectionId || '').trim() || registry.primary const source = registry.connections.find(c => c.id === id) @@ -11519,7 +11596,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela const primary = await reuseMatchingPrimarySshBackend({ connectionId: id, effectiveFingerprint: resolveRegistryEffectiveFingerprint, - ensurePrimary: () => ensureBackend(profile), + ensurePrimary: () => ensureBackend(profile, { spawnPriority }), profile, registry, source @@ -11570,7 +11647,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela }) if (localRoute.delegate) { - return ensureBackend(profile) + return ensureBackend(profile, { spawnPriority }) } const stoppingLocal = poolStopper.inFlight(localRoute.poolKey) @@ -11584,6 +11661,10 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela if (existingLocal) { existingLocal.lastActiveAt = Date.now() + if (spawnPriority === 'foreground') { + promotePoolEntry(existingLocal) + } + return existingLocal.connectionPromise } @@ -11598,7 +11679,8 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela remoteBaseUrl: null, releaseLocalBackendSlot: null, localBackendSlotKey: null, - localBackendSpawnRequest: null + localBackendSpawnRequest: null, + spawnPriority } localEntry.connectionPromise = spawnPoolBackend(profileKey, localEntry, { @@ -11607,9 +11689,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela }).catch(async error => { // Same trace rule as the v1 pool path: a forced-local child whose spawn // rejects before the child exists must still land in desktop.log. - rememberLog( - `Hermes backend for profile "${profileKey}" (forced-local) failed to start: ${error instanceof Error ? error.message : String(error)}` - ) + logPoolSpawnFailure(`"${profileKey}" (forced-local)`, error) await teardownFailedLocalBackend(localRoute.poolKey, localEntry) throw error @@ -12447,11 +12527,23 @@ async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; po // pool-idle window (10 min) would hold the pool key hostage and every // later click on the profile would join that stale wait. Failing here // surfaces the "all N slots busy" reason instead of a generic boot timeout. - const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { timeoutMs: POOL_SLOT_WAIT_MS }) + // The caller stamped entry.spawnPriority from its own request; a foreground + // dial that joined the claim before this entry existed left a mark instead. + if (takeForegroundSpawn(poolKey, profile)) { + entry.spawnPriority = 'foreground' + } + + const spawnPriority: LocalBackendSpawnPriority = spawnPriorityFrom(entry.spawnPriority) + + const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { + timeoutMs: POOL_SLOT_WAIT_MS, + priority: spawnPriority + }) + entry.localBackendSlotKey = poolKey entry.localBackendSpawnRequest = spawnRequest - if (localBackendSpawnCoordinator.activeCount >= poolMaxBackends()) { + if (spawnRequest.queued) { rememberLog( `Profile backend "${profile}" waiting for a free local slot (${localBackendSpawnCoordinator.activeCount}/${poolMaxBackends()} busy, ${localBackendSpawnCoordinator.queuedCount} queued)` ) @@ -14711,13 +14803,27 @@ function createWindow() { }) } -ipcMain.handle('hermes:connection', async (_event, profile) => { +ipcMain.handle('hermes:connection', async (_event, profile, extra) => { // Coalesce concurrent renderer dials for one profile scope (#90812): the // renderer-side reconnect lock is per-window, so two windows waking at once // both land here. The claim key mirrors ensureBackend()'s own profile // normalization so every spelling of the primary coalesces onto one dial. const profileKey = profile && String(profile).trim() ? String(profile).trim() : primaryProfileKey() - const connection = await backendDialClaims.run(backendScopeKey(null, profileKey), () => ensureBackend(profile)) + // A user click may join an in-flight hydration claim; the foreground intent + // is applied to that claim so its slot wait can take the reserved slot. + const spawnPriority = spawnPriorityFrom(extra?.priority) + + const scopeKey = backendScopeKey(null, profileKey) + const clearSpawnPriority = applySpawnPriority(scopeKey, spawnPriority) + + let connection + + try { + connection = await backendDialClaims.run(scopeKey, () => ensureBackend(profile, { spawnPriority })) + } finally { + clearSpawnPriority() + } + const connectionId = resolvedConnectionId(readDesktopConnectionsRegistry(), connection) return connectionId ? { ...connection, connectionId } : connection @@ -14728,13 +14834,24 @@ ipcMain.handle('hermes:connection', async (_event, profile) => { // forces a genuinely-local child when the v1 global mode is remote (the // registry 'local' entry always means this machine). ipcMain.handle('hermes:connection:for', async (_event, payload) => { - const { connectionId, profile } = payload && typeof payload === 'object' ? (payload as any) : ({} as any) + const { connectionId, profile, priority } = payload && typeof payload === 'object' ? (payload as any) : ({} as any) const registry = readDesktopConnectionsRegistry() const id = String(connectionId || '').trim() || registry.primary + const spawnPriority = spawnPriorityFrom(priority) + // Same single-owner claim as 'hermes:connection', keyed by the composite // (connectionId, profile) scope (#90812): concurrent registry dials for one // scope share the first spawn instead of bootstrapping duplicate remotes. - const connection = await backendDialClaims.run(backendScopeKey(id, profile), () => ensureRegistryBackend(id, profile)) + const scopeKey = backendScopeKey(id, profile) + const clearSpawnPriority = applySpawnPriority(scopeKey, spawnPriority) + + let connection + + try { + connection = await backendDialClaims.run(scopeKey, () => ensureRegistryBackend(id, profile, '', { spawnPriority })) + } finally { + clearSpawnPriority() + } return { ...connection, connectionId: id, registryScoped: true } }) diff --git a/apps/desktop/electron/pool-spawn-coordinator.test.ts b/apps/desktop/electron/pool-spawn-coordinator.test.ts index 73b2291fe1..b7cf4ee30e 100644 --- a/apps/desktop/electron/pool-spawn-coordinator.test.ts +++ b/apps/desktop/electron/pool-spawn-coordinator.test.ts @@ -6,7 +6,11 @@ import { fileURLToPath } from 'node:url' import { test } from 'vitest' -import { LocalBackendSpawnCoordinator, releaseLocalBackendSlotAfterExit } from './pool-spawn-coordinator' +import { + LocalBackendSlotWaitTimeoutError, + LocalBackendSpawnCoordinator, + releaseLocalBackendSlotAfterExit +} from './pool-spawn-coordinator' const deferred = () => { let resolve!: () => void @@ -306,6 +310,197 @@ test('setLimit rejects a non-positive or fractional cap', () => { assert.equal(coordinator.limit, 2) }) +test('cap 3: two background leases leave a reserved slot for foreground', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bg1 = await coordinator.request('bg-1', { priority: 'background' }).acquired + const bg2 = await coordinator.request('bg-2', { priority: 'background' }).acquired + assert.equal(coordinator.activeCount, 2) + assert.equal(coordinator.queuedCount, 0) + + let fgGranted = false + + const fgPromise = coordinator.request('fg', { priority: 'foreground' }).acquired.then(release => { + fgGranted = true + + return release + }) + + await flush() + assert.equal(fgGranted, true) + assert.equal(coordinator.activeCount, 3) + + const releaseFg = await fgPromise + bg1() + bg2() + releaseFg() + assert.equal(coordinator.activeCount, 0) +}) + +test('untagged acquire still fills the cap (foreground default)', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const releases = await Promise.all(['a', 'b', 'c'].map(key => coordinator.acquire(key))) + assert.equal(coordinator.activeCount, 3) + assert.equal(coordinator.queuedCount, 0) + + for (const release of releases) { + release() + } + + assert.equal(coordinator.activeCount, 0) +}) + +test('foreground is granted the reserved slot ahead of a background hydration queue', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + + const bgRunning = await Promise.all( + ['bg-run-1', 'bg-run-2'].map(key => coordinator.request(key, { priority: 'background' }).acquired) + ) + + const queued = Array.from({ length: 20 }, (_, index) => + coordinator.request(`bg-wait-${index}`, { priority: 'background', timeoutMs: 5_000 }) + ) + + await flush() + assert.equal(coordinator.activeCount, 2) + assert.equal(coordinator.queuedCount, 20) + + const started = Date.now() + const releaseFg = await coordinator.request('user-click', { priority: 'foreground', timeoutMs: 100 }).acquired + assert.ok(Date.now() - started < 80, 'foreground must not wait behind the background queue') + assert.equal(coordinator.activeCount, 3) + + for (const request of queued) { + request.cancel() + } + + releaseFg() + + for (const release of bgRunning) { + release() + } + + await Promise.all( + queued.map(request => + request.acquired.then( + () => undefined, + () => undefined + ) + ) + ) + assert.equal(coordinator.activeCount, 0) + assert.equal(coordinator.queuedCount, 0) +}) + +test('drain prefers a foreground waiter over an earlier background waiter', async () => { + const coordinator = new LocalBackendSpawnCoordinator(1) + const releaseHolder = await coordinator.acquire('holder') + const background = coordinator.request('background', { priority: 'background' }) + const foreground = coordinator.request('foreground', { priority: 'foreground' }) + await flush() + assert.equal(coordinator.queuedCount, 2) + + let backgroundEntered = false + let foregroundEntered = false + + const backgroundGrant = background.acquired.then(release => { + backgroundEntered = true + + return release + }) + + const foregroundGrant = foreground.acquired.then(release => { + foregroundEntered = true + + return release + }) + + releaseHolder() + await flush() + assert.equal(foregroundEntered, true) + assert.equal(backgroundEntered, false) + assert.equal(coordinator.activeCount, 1) + + const releaseForeground = await foregroundGrant + releaseForeground() + const releaseBackground = await backgroundGrant + assert.equal(backgroundEntered, true) + releaseBackground() + assert.equal(coordinator.activeCount, 0) +}) + +test('background slot-wait timeout is distinguishable; foreground keeps a user-facing message', async () => { + const coordinator = new LocalBackendSpawnCoordinator(1) + const releaseFirst = await coordinator.acquire('first') + + const background = coordinator.request('bg', { priority: 'background', timeoutMs: 10 }) + await assert.rejects(background.acquired, error => { + assert.ok(error instanceof LocalBackendSlotWaitTimeoutError) + assert.equal(error.name, 'LocalBackendSlotWaitTimeoutError') + assert.equal(error.priority, 'background') + assert.equal(error.silent, true) + assert.match(error.message, /timed out while waiting for a free slot/) + assert.match(error.message, /\(background\)/) + + return true + }) + + const foreground = coordinator.request('fg', { priority: 'foreground', timeoutMs: 10 }) + await assert.rejects(foreground.acquired, error => { + assert.ok(error instanceof Error) + assert.match(error.message, /timed out while waiting for a free slot/) + assert.doesNotMatch(error.message, /\(background\)/) + assert.notEqual(error.name, 'LocalBackendSlotWaitTimeoutError') + + return true + }) + + releaseFirst() + assert.equal(coordinator.activeCount, 0) +}) + +test('request() reports whether the caller actually waited behind the queue', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bg1 = coordinator.request('bg-1', { priority: 'background' }) + const bg2 = coordinator.request('bg-2', { priority: 'background' }) + const bgWait = coordinator.request('bg-3', { priority: 'background' }) + assert.equal(bg1.queued, false) + assert.equal(bg2.queued, false) + assert.equal(bgWait.queued, true) + + // The reserved slot is free: a foreground request is granted immediately + // even though a background waiter is queued. + const fg = coordinator.request('fg', { priority: 'foreground' }) + assert.equal(fg.queued, false) + assert.equal(coordinator.activeCount, 3) + + bgWait.cancel() + await bgWait.acquired.catch(() => undefined) + ;(await fg.acquired)() + ;(await bg1.acquired)() + ;(await bg2.acquired)() + assert.equal(coordinator.activeCount, 0) +}) + +test('promoting a queued background waiter lets it take the reserved foreground slot', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bg1 = await coordinator.request('bg-1', { priority: 'background' }).acquired + const bg2 = await coordinator.request('bg-2', { priority: 'background' }).acquired + const queued = coordinator.request('same-bot', { priority: 'background' }) + await flush() + assert.equal(coordinator.activeCount, 2) + assert.equal(coordinator.queuedCount, 1) + + assert.equal(queued.promote('foreground'), true) + const releasePromoted = await queued.acquired + assert.equal(coordinator.activeCount, 3) + assert.equal(coordinator.queuedCount, 0) + + releasePromoted() + bg1() + bg2() + assert.equal(coordinator.activeCount, 0) +}) + // ── main.ts wiring ────────────────────────────────────────────────────────── // The coordinator is only as good as the timeout main.ts hands it. A queued // ticket that outlives the renderer's backend-boot budget holds the pool key @@ -330,7 +525,10 @@ test('setLimit rejects a non-positive or fractional cap', () => { assert.ok(Number.isFinite(slotWait) && slotWait > 0, 'POOL_SLOT_WAIT_MS must be a literal in main.ts') assert.ok(Number.isFinite(bootBudget), 'BACKEND_BOOT_WAIT_TIMEOUT_MS must be a literal') assert.ok(slotWait < bootBudget, `slot wait ${slotWait}ms must be below the boot budget ${bootBudget}ms`) - assert.match(mainSource, /localBackendSpawnCoordinator\.request\(poolKey, \{ timeoutMs: POOL_SLOT_WAIT_MS \}\)/) + assert.match( + mainSource, + /localBackendSpawnCoordinator\.request\(poolKey, \{\s*timeoutMs: POOL_SLOT_WAIT_MS,\s*priority: spawnPriority\s*\}\)/ + ) assert.doesNotMatch(mainSource, /request\(poolKey, \{ timeoutMs: POOL_IDLE_MS \}\)/) }) diff --git a/apps/desktop/electron/pool-spawn-coordinator.ts b/apps/desktop/electron/pool-spawn-coordinator.ts index 8e565ed133..5f80af332d 100644 --- a/apps/desktop/electron/pool-spawn-coordinator.ts +++ b/apps/desktop/electron/pool-spawn-coordinator.ts @@ -1,17 +1,47 @@ export type ReleaseLocalBackendSlot = () => void +export type LocalBackendSpawnPriority = 'foreground' | 'background' + export type LocalBackendSpawnRequest = { acquired: Promise cancel: () => boolean + promote: (priority: LocalBackendSpawnPriority) => boolean + /** False when the slot was granted without waiting behind the queue. */ + queued: boolean } type Waiter = { key: string + priority: LocalBackendSpawnPriority resolve: (release: ReleaseLocalBackendSlot) => void reject: (error: Error) => void timer: ReturnType | null } +const SLOT_WAIT_TIMEOUT_MESSAGE = (key: string) => + `Local backend start for "${key}" timed out while waiting for a free slot.` + +/** + * Slot-wait timeout. Background hydrations set `silent` so call sites can fail + * quiet instead of toasting a user-visible backend-start failure. + */ +export class LocalBackendSlotWaitTimeoutError extends Error { + readonly priority: LocalBackendSpawnPriority + readonly silent: boolean + + constructor(key: string, priority: LocalBackendSpawnPriority) { + const suffix = priority === 'background' ? ' (background)' : '' + super(`${SLOT_WAIT_TIMEOUT_MESSAGE(key)}${suffix}`) + this.name = 'LocalBackendSlotWaitTimeoutError' + this.priority = priority + this.silent = priority === 'background' + } +} + +export function isBackgroundSlotWaitTimeout(error: unknown): boolean { + return error instanceof LocalBackendSlotWaitTimeoutError && error.silent +} + export async function releaseLocalBackendSlotAfterExit( release: ReleaseLocalBackendSlot, waitForExit: () => Promise @@ -25,10 +55,15 @@ export async function releaseLocalBackendSlotAfterExit( * * A lease is acquired immediately before local start work and is held until * the child exits or the start fails. Remote descriptors never call request(). + * + * When the cap is at least 2, one slot is reserved for foreground (user-open) + * requests so background roster hydration cannot occupy the whole pool. + * Untagged acquire() is foreground, so existing cap tests still fill `limit`. */ export class LocalBackendSpawnCoordinator { #limit: number - #active = 0 + #activeForeground = 0 + #activeBackground = 0 #queue: Waiter[] = [] constructor(limit: number) { @@ -40,7 +75,7 @@ export class LocalBackendSpawnCoordinator { } get activeCount(): number { - return this.#active + return this.#activeForeground + this.#activeBackground } get limit(): number { @@ -66,39 +101,47 @@ export class LocalBackendSpawnCoordinator { return this.#queue.length } - request(key: string, options: { timeoutMs?: number } = {}): LocalBackendSpawnRequest { + request( + key: string, + options: { timeoutMs?: number; priority?: LocalBackendSpawnPriority } = {} + ): LocalBackendSpawnRequest { if (options.timeoutMs !== undefined && (!Number.isFinite(options.timeoutMs) || options.timeoutMs < 1)) { throw new RangeError('Local backend spawn timeout must be a positive number.') } - if (this.#active < this.#limit) { + const priority: LocalBackendSpawnPriority = options.priority === 'background' ? 'background' : 'foreground' + + if (this.#queue.length === 0 && this.#canGrant(priority)) { return { - acquired: Promise.resolve(this.#grant()), - cancel: () => false + acquired: Promise.resolve(this.#grant(priority)), + cancel: () => false, + promote: () => false, + queued: false } } let waiter!: Waiter const acquired = new Promise((resolve, reject) => { - waiter = { key, resolve, reject, timer: null } + waiter = { key, priority, resolve, reject, timer: null } this.#queue.push(waiter) if (options.timeoutMs !== undefined) { waiter.timer = setTimeout(() => { - this.#rejectWaiter( - waiter, - new Error(`Local backend start for "${key}" timed out while waiting for a free slot.`) - ) + this.#rejectWaiter(waiter, this.#timeoutError(waiter)) }, options.timeoutMs) waiter.timer.unref?.() } }) + this.#drain() + return { acquired, cancel: () => - this.#rejectWaiter(waiter, new Error(`Local backend start for "${key}" was cancelled while queued.`)) + this.#rejectWaiter(waiter, new Error(`Local backend start for "${key}" was cancelled while queued.`)), + promote: (nextPriority: LocalBackendSpawnPriority) => this.#promoteWaiter(waiter, nextPriority), + queued: this.#queue.includes(waiter) } } @@ -106,6 +149,45 @@ export class LocalBackendSpawnCoordinator { return this.request(key).acquired } + #timeoutError(waiter: Waiter): Error { + if (waiter.priority === 'background') { + return new LocalBackendSlotWaitTimeoutError(waiter.key, 'background') + } + + return new Error(SLOT_WAIT_TIMEOUT_MESSAGE(waiter.key)) + } + + #backgroundLimit(): number { + return this.#limit >= 2 ? this.#limit - 1 : this.#limit + } + + #canGrant(priority: LocalBackendSpawnPriority): boolean { + if (this.activeCount >= this.#limit) { + return false + } + + if (priority === 'background' && this.#activeBackground >= this.#backgroundLimit()) { + return false + } + + return true + } + + #promoteWaiter(waiter: Waiter, priority: LocalBackendSpawnPriority): boolean { + if (!this.#queue.includes(waiter)) { + return false + } + + if (waiter.priority === priority) { + return false + } + + waiter.priority = priority + this.#drain() + + return true + } + #rejectWaiter(waiter: Waiter, error: Error): boolean { const index = this.#queue.indexOf(waiter) @@ -127,8 +209,13 @@ export class LocalBackendSpawnCoordinator { } } - #grant(): ReleaseLocalBackendSlot { - this.#active += 1 + #grant(priority: LocalBackendSpawnPriority): ReleaseLocalBackendSlot { + if (priority === 'background') { + this.#activeBackground += 1 + } else { + this.#activeForeground += 1 + } + let released = false return () => { @@ -137,17 +224,49 @@ export class LocalBackendSpawnCoordinator { } released = true - this.#active -= 1 + + if (priority === 'background') { + this.#activeBackground -= 1 + } else { + this.#activeForeground -= 1 + } + this.#drain() } } - /** Hand free slots to queued waiters while under the (possibly lowered) cap. */ + #takeWaiter(priority: LocalBackendSpawnPriority): Waiter | undefined { + const index = this.#queue.findIndex(waiter => waiter.priority === priority) + + if (index === -1) { + return undefined + } + + return this.#queue.splice(index, 1)[0] + } + + /** Hand free slots to queued waiters. Foreground waiters always go first. */ #drain(): void { - while (this.#active < this.#limit && this.#queue.length > 0) { - const next = this.#queue.shift()! + while (this.#canGrant('foreground')) { + const next = this.#takeWaiter('foreground') + + if (!next) { + break + } + this.#clearTimer(next) - next.resolve(this.#grant()) + next.resolve(this.#grant('foreground')) + } + + while (this.#canGrant('background')) { + const next = this.#takeWaiter('background') + + if (!next) { + break + } + + this.#clearTimer(next) + next.resolve(this.#grant('background')) } } } diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index fd9668752b..ca8bb45a28 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -18,7 +18,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { // Launch-flag fact: the app was started with --local, so the renderer may // show the local-models surfaces. Static for the window's lifetime. localModelsEnabled: launchFlags?.localModels === true, - getConnection: profile => ipcRenderer.invoke('hermes:connection', profile), + getConnection: (profile, opts) => ipcRenderer.invoke('hermes:connection', profile, opts), // Registry-scoped backend resolution: { connectionId, profile } → descriptor. getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload), getProfileRoutes: profiles => ipcRenderer.invoke('hermes:plugin-profile-routes', profiles), diff --git a/apps/desktop/electron/remote-lifecycle.test.ts b/apps/desktop/electron/remote-lifecycle.test.ts index dffc716193..408cd9e27a 100644 --- a/apps/desktop/electron/remote-lifecycle.test.ts +++ b/apps/desktop/electron/remote-lifecycle.test.ts @@ -1849,3 +1849,66 @@ test('cleanupStale keeps the lockfile when even SIGKILL cannot confirm the pid d // The record must survive so the next connect's reap pass retries. assert.ok(!ssh.calls.some(c => /rm -f .*backend\.lock\.json/.test(c))) }) +test.skipIf(process.platform === 'win32')( + 'buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh parse)', + async () => { + // expandRemotePath() output is pre-quoted; a second shq() ships literal quote + // characters to the remote python. Parse the composed command with a real sh, + // as the remote login shell does, and require every path to come out clean. + const cmd = buildSpawnCommand('/x/hermes', 'work', { + hermesHome: '~/.hermes', + logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), + ownershipId: OWNERSHIP_ID, + reservationNonce: SPAWN_NONCE, + spawnNonce: SPAWN_NONCE, + tokenFilePath: spawnTokenPath(OWNERSHIP_ID, SPAWN_NONCE), + lockMetadata: { ownershipId: OWNERSHIP_ID, spawnNonce: SPAWN_NONCE } + }) + + // Capture the argv a remote shell would hand to python3, via a shim on PATH. + const root = await mkdtemp(path.join(os.tmpdir(), 'hermes-argv-shim-')) + + try { + const shimDir = path.join(root, 'shim') + const fakeHome = path.join(root, 'home') + await mkdir(shimDir) + await mkdir(fakeHome) + const argvFile = path.join(shimDir, 'argv') + await writeFile(path.join(shimDir, 'python3'), `#!/bin/sh\nprintf '%s\\0' "$@" > '${argvFile}'\n`, { + mode: 0o755 + }) + await exec(cmd, { + env: { ...process.env, PATH: `${shimDir}:${process.env.PATH}`, HOME: fakeHome } + }) + const argv = (await readFile(argvFile, 'utf8')).split('\0') + + // argv: ['-c', , , ] + assert.equal( + argv[2], + `${fakeHome}/.hermes/.hermes-update-in-progress.mutex`, + 'mutex path must reach python fully expanded, with no quote characters' + ) + + // The payload assigns reservation/lock/owner_file before its mkdir loop. + // Evaluate only that prefix the way the remote sh does; never the loop itself. + const payload = argv[3] + const loopStart = payload.indexOf('i=0;') + assert.ok(loopStart > 0, 'payload prefix sentinel missing') + + const { stdout } = await exec( + `${payload.slice(0, loopStart)} printf '%s\\n' "$reservation" "$lock" "$owner_file"`, + { + env: { ...process.env, HOME: fakeHome } + } + ) + + const [reservation, lock, ownerFile] = stdout.split('\n') + const base = `${fakeHome}/.hermes/desktop-ssh/${OWNERSHIP_ID}` + assert.equal(reservation, `${base}/.connect.lock`) + assert.equal(lock, `${base}/backend.lock.json`) + assert.equal(ownerFile, `${base}/.connect.lock/owner`) + } finally { + await rm(root, { recursive: true, force: true }) + } + } +) diff --git a/apps/desktop/electron/remote-lifecycle.ts b/apps/desktop/electron/remote-lifecycle.ts index d096705c95..40108383cd 100644 --- a/apps/desktop/electron/remote-lifecycle.ts +++ b/apps/desktop/electron/remote-lifecycle.ts @@ -895,7 +895,9 @@ finally: // the marker check, spawns the backend, and publishes its initial lockfile. // Python keeps the descriptor close-on-exec by default and passes it explicitly // only to the intended outer shell; each detached child closes it before -// execing Hermes. +// execing Hermes. mutexPath is expandRemotePath() output — a complete shell +// word ("$HOME"'/…' or '/abs/…') embedded raw so $HOME expands remotely; a +// second shq() would hand python the quote characters as part of the path. function withRemoteUpdateMutex(command, mutexPath) { const script = ` import fcntl,os,subprocess,sys @@ -913,7 +915,7 @@ finally: sys.exit(result.returncode if result is not None else 1) `.trim() - return `python3 -c ${shq(script)} ${shq(mutexPath)} ${shq(command)}` + return `python3 -c ${shq(script)} ${mutexPath} ${shq(command)}` } /** diff --git a/apps/desktop/scripts/desktop-update-ui.test.mjs b/apps/desktop/scripts/desktop-update-ui.test.mjs new file mode 100644 index 0000000000..bc69320a4a --- /dev/null +++ b/apps/desktop/scripts/desktop-update-ui.test.mjs @@ -0,0 +1,100 @@ +import assert from 'node:assert/strict' +import fs from 'node:fs' +import { JSDOM } from 'jsdom' +import { afterEach, test, vi } from 'vitest' + +// Execute the shipped page, including its inline script, rather than matching +// source strings or testing a second implementation of the progress client. +const html = fs.readFileSync(new URL('../../../scripts/desktop-update/ui.html', import.meta.url), 'utf8') +const windows = [] + +function openPage(fetch) { + vi.useFakeTimers() + const dom = new JSDOM(html, { + url: 'http://127.0.0.1:12345/', + runScripts: 'dangerously', + beforeParse(window) { + window.fetch = fetch + window.AbortController = AbortController + window.setTimeout = setTimeout + window.clearTimeout = clearTimeout + window.requestAnimationFrame = () => 1 + window.cancelAnimationFrame = () => {} + } + }) + windows.push(dom.window) + return dom.window.document +} + +afterEach(() => { + windows.splice(0).forEach(window => window.close()) + vi.useRealTimers() +}) + +test.each(['done', 'manual', 'error'])('renders %s before acknowledging terminal delivery', async status => { + let document + const requests = [] + const receipt = '550e8400-e29b-41d4-a716-446655440000' + const fetch = vi.fn(async (url, options) => { + requests.push(url) + if (url.startsWith('/ack/')) { + assert.equal(options.method, 'POST') + assert.equal(document.body.className, status === 'error' ? 'error' : 'done') + assert.notEqual(document.getElementById('title').textContent, 'Updating Hermes') + return { ok: true } + } + return { ok: true, json: async () => ({ status, receipt, message: 'The updater result' }) } + }) + document = openPage(fetch) + await vi.advanceTimersByTimeAsync(1000) + assert.equal(document.body.className, status === 'error' ? 'error' : 'done') + assert.notEqual(document.getElementById('title').textContent, 'Updating Hermes') + assert.deepEqual(requests, ['/progress', `/ack/${receipt}`]) +}) + +test.each(['disconnect', 'hung', 'hung-body', 'http', 'invalid'])('bounds %s progress failures without inventing an update outcome', async failure => { + let attempts = 0 + const fetch = vi.fn((_url, options) => { + attempts++ + if (attempts === 1) { + return Promise.resolve({ ok: true, json: async () => ({ status: 'running', message: 'Installing dependencies' }) }) + } + if (failure === 'hung' || failure === 'hung-body') { + const pending = () => new Promise((_resolve, reject) => { + options.signal?.addEventListener('abort', () => reject(new Error('timeout')), { once: true }) + }) + return failure === 'hung' ? pending() : Promise.resolve({ ok: true, json: pending }) + } + if (failure === 'http') return Promise.resolve({ ok: false }) + if (failure === 'invalid') return Promise.resolve({ ok: true, json: async () => ({}) }) + return Promise.reject(new Error('connection refused')) + }) + const document = openPage(fetch) + await vi.advanceTimersByTimeAsync(20_000) + assert.equal(document.body.className, 'disconnected') + assert.equal(document.getElementById('title').textContent, 'Update status unavailable') + assert.match(document.getElementById('line').textContent, /Check Hermes/) + assert.ok(attempts <= 4, `unbounded retry loop: ${attempts}`) +}) + +test.each(['legacy', 'transient', 'ack-failure'])('preserves terminal truth with %s servers', async mode => { + let attempts = 0 + const fetch = vi.fn(async url => { + if (url.startsWith('/ack/')) throw new Error('server already stopped') + if (++attempts === 1 && mode === 'transient') throw new Error('temporary disconnect') + return { ok: true, json: async () => ({ status: 'done', ...(mode === 'ack-failure' ? { receipt: 'test-receipt' } : {}) }) } + }) + const document = openPage(fetch) + await vi.advanceTimersByTimeAsync(20_000) + assert.equal(document.body.className, 'done') + assert.equal(document.getElementById('title').textContent, 'Update complete') +}) + +test('continues displaying a healthy long update while progress remains reachable', async () => { + const fetch = vi.fn(async () => ({ ok: true, json: async () => ({ status: 'running', message: 'Building Desktop' }) })) + const document = openPage(fetch) + await vi.advanceTimersByTimeAsync(60_000) + assert.equal(document.body.className, '') + assert.equal(document.getElementById('title').textContent, 'Updating Hermes') + assert.equal(document.getElementById('line').textContent, 'Building Desktop') +}) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts new file mode 100644 index 0000000000..1c5fb5f0b7 --- /dev/null +++ b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts @@ -0,0 +1,200 @@ +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' + +import { $composerActionsBySession } from '@/store/composer-actions' +import { $previewStatusBySession } from '@/store/preview-status' +import { $sessionControlBySession, type SessionControlEntry } from '@/store/session-control' +import { $todosBySession } from '@/store/todos' + +import { useSessionStatusPresence } from './use-status-presence' + +const SID = 'presence-session-1' + +const mockEntry = (overrides?: Partial): SessionControlEntry => ({ + capability: 'supported', + error: null, + loading: false, + pendingAction: null, + snapshot: { + goal: null, + heartbeat: null, + loop: null, + revision: 'rev-1', + updated_at: 1000 + }, + ...overrides +}) + +describe('useSessionStatusPresence', () => { + beforeEach(() => { + $todosBySession.set({}) + $composerActionsBySession.set({}) + $previewStatusBySession.set({}) + $sessionControlBySession.set({}) + }) + + afterEach(() => { + cleanup() + $todosBySession.set({}) + $composerActionsBySession.set({}) + $previewStatusBySession.set({}) + $sessionControlBySession.set({}) + }) + + it('returns false when session is null or empty', () => { + const { result } = renderHook(() => useSessionStatusPresence(null)) + expect(result.current).toBe(false) + + const { result: emptyResult } = renderHook(() => useSessionStatusPresence(SID)) + expect(emptyResult.current).toBe(false) + }) + + it('returns true when legacy status items exist', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $todosBySession.set({ + [SID]: [{ content: 'task 1', id: '1', status: 'in_progress' }] + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns true when structured goal exists in session control', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: { + contract: { + boundaries: '', + constraints: '', + outcome: 'test outcome', + stop_when: '', + verification: '' + }, + gates: [], + max_turns: 10, + status: 'active', + subgoals: [], + title: 'Structured Goal', + turns_used: 1 + }, + heartbeat: null, + loop: null, + revision: 'rev-2', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns true when structured loop exists in session control', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: null, + heartbeat: null, + loop: { + awaiting_response: false, + created_at: 1000, + current_delay: 60, + deferred_by_goal: false, + interval_seconds: 60, + last_fired_at: 1000, + max_ticks: 10, + mode: 'interval', + next_due_at: 2000, + prompt: 'Run loop', + status: 'active', + ticks_fired: 0, + times: 5, + until: '' + }, + revision: 'rev-3', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns true when structured heartbeat exists in session control', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: null, + heartbeat: { + created_at: 1000, + fire_count: 1, + interval_seconds: 300, + last_fired_at: 1000, + prompt: 'Heartbeat check', + status: 'active' + }, + loop: null, + revision: 'rev-4', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns false when session control entry has empty snapshot (null goal, loop, heartbeat)', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: null, + heartbeat: null, + loop: null, + revision: 'rev-empty', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(false) + }) + + it('returns true when session control entry has only an error', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + error: 'Gateway connection failed', + snapshot: null + }) + }) + }) + + expect(result.current).toBe(true) + }) +}) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts index b4655ffd50..b2c059ba35 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts @@ -3,8 +3,9 @@ import { useSyncExternalStore } from 'react' import { $composerActionsBySession } from '@/store/composer-actions' import { $statusItemsBySession } from '@/store/composer-status' import { $previewStatusBySession } from '@/store/preview-status' +import { $sessionControlBySession } from '@/store/session-control' -/** Structural view of the three per-session feeds — they hold different item +/** Structural view of the per-session feeds — they hold different item * types, and all this hook needs from each is "does this key have rows". */ interface PresenceFeed { get(): Record @@ -14,7 +15,7 @@ interface PresenceFeed { const FEEDS: PresenceFeed[] = [$statusItemsBySession, $composerActionsBySession, $previewStatusBySession] const subscribe = (onChange: () => void) => { - const offs = FEEDS.map(feed => feed.listen(onChange)) + const offs = [...FEEDS.map(feed => feed.listen(onChange)), $sessionControlBySession.listen(onChange)] return () => { for (const off of offs) { @@ -24,8 +25,9 @@ const subscribe = (onChange: () => void) => { } /** - * Whether a session has any status items, micro actions, or previews, as a - * coarse *edge*: the boolean only flips when the stack appears/disappears. + * Whether a session has any status items, micro actions, previews, or + * structured session controls (goal, loop, heartbeat), as a coarse *edge*: + * the boolean only flips when the stack appears/disappears. * ChatBar uses it to toggle a styling data-attr — subscribing to the whole * `$statusItemsBySession` (a `computed` that rebuilds the entire map) / * `$previewStatusBySession` maps re-rendered the ~1.4k ChatBar on every @@ -39,6 +41,16 @@ export function useSessionStatusPresence(sessionId: string | null): boolean { return false } - return FEEDS.some(feed => (feed.get()[sessionId]?.length ?? 0) > 0) + if (FEEDS.some(feed => (feed.get()[sessionId]?.length ?? 0) > 0)) { + return true + } + + const control = $sessionControlBySession.get()[sessionId] + + return Boolean( + control?.error || + (control?.snapshot && + (control.snapshot.goal !== null || control.snapshot.loop !== null || control.snapshot.heartbeat !== null)) + ) }) } diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index b01381e41d..371bf32b07 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -1195,6 +1195,7 @@ export function ChatBar({ grows upward over the thread and the dock's own measurement covers it. Collapses to nothing when every status is empty. */} 0 ? ( group.type === 'todo' && group.items.some(item => item.todoStatus === 'in_progress' && item.state === 'running') interface ComposerStatusStackProps { + onSubmit?: (value: string, options?: SubmitTextOptions) => Promise | boolean /** The queue, built by the composer (it owns the queue's callbacks). Rendered * as the last group so it stays fused to the composer like before. */ queue: ReactNode @@ -85,7 +89,7 @@ interface ComposerStatusStackProps { * every session-scoped status — subagents, background tasks, queue — grouped by * type and separated by light dividers. Collapses to nothing when empty. */ -export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackProps) { +export function ComposerStatusStack({ onSubmit, queue, sessionId }: ComposerStatusStackProps) { const { t } = useI18n() const navigate = useNavigate() // Subscribe to THIS session's slice only. Both maps churn on other @@ -96,10 +100,22 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro // items actually changed. const items = useSessionSlice($statusItemsBySession, sessionId) const previews = useSessionSlice($previewStatusBySession, sessionId) + const controlEntry = useSessionValue($sessionControlBySession, sessionId) + const scrolledUp = useStore($threadScrolledUp) const billing = useStore($billingBlock) - const groups = useMemo(() => groupStatusItems(items), [items]) + const isStructuredSupported = controlEntry?.capability === 'supported' + + const groups = useMemo(() => { + const raw = groupStatusItems(items) + + if (isStructuredSupported) { + return raw.filter(g => g.type !== 'goal') + } + + return raw + }, [items, isStructuredSupported]) // Seed from the registry on session open; event-driven refreshes (terminal / // process tool completions) live in use-message-stream. This must NOT reset @@ -112,7 +128,7 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro useEffect(() => { if (sessionId) { void refreshBackgroundProcesses(sessionId) - void refreshSessionGoal(sessionId) + void refreshSessionControl(sessionId) } }, [sessionId]) @@ -160,6 +176,21 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro sections.push({ key: 'billing', node: }) } + const hasControlContent = Boolean( + controlEntry && + (controlEntry.error || + controlEntry.snapshot?.goal || + controlEntry.snapshot?.loop || + controlEntry.snapshot?.heartbeat) + ) + + if (sessionId && controlEntry && hasControlContent) { + sections.push({ + key: 'session-control', + node: + }) + } + for (const group of groups) { sections.push({ key: group.type, diff --git a/apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx b/apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx new file mode 100644 index 0000000000..59d55dbd4a --- /dev/null +++ b/apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx @@ -0,0 +1,639 @@ +import { memo, useCallback, useState } from 'react' + +import { queueKickoffIfSessionBusy } from '@/app/session/hooks/use-prompt-actions/queue-if-busy' +import type { SubmitTextOptions } from '@/app/session/hooks/use-prompt-actions/utils' +import { StatusSection } from '@/components/chat/status-section' +import { Button } from '@/components/ui/button' +import { Codicon } from '@/components/ui/codicon' +import { ConfirmDialog } from '@/components/ui/confirm-dialog' +import { + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger +} from '@/components/ui/context-menu' +import { + Dialog, + DialogContent, + DialogDescription, + DialogFooter, + DialogHeader, + DialogTitle +} from '@/components/ui/dialog' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Tip } from '@/components/ui/tooltip' +import { useI18n } from '@/i18n' +import { + runSessionControlAction, + type SessionControlAction, + type SessionControlActionArgs, + type SessionControlGoal +} from '@/store/session-control' + +import type { ConfirmState } from './session-control-utils' + +interface GoalSectionProps { + goal: SessionControlGoal + sessionId: string + pendingAction: SessionControlAction | null + onSubmit?: (value: string, options?: SubmitTextOptions) => Promise | boolean + onFeedback: (error: string | null, success: string | null) => void +} + +export const SessionControlGoalSection = memo(function SessionControlGoalSection({ + goal, + sessionId, + pendingAction, + onSubmit, + onFeedback +}: GoalSectionProps) { + const { t } = useI18n() + const s = t.statusStack + const ctrl = s.control + + const [detailsOpen, setDetailsOpen] = useState(false) + const [addCriterionOpen, setAddCriterionOpen] = useState(false) + const [addCriterionError, setAddCriterionError] = useState(null) + const [newCriterionText, setNewCriterionText] = useState('') + const [confirmState, setConfirmState] = useState(null) + const [menuOpen, setMenuOpen] = useState(false) + + const isBusy = Boolean(pendingAction) + + const handleAction = useCallback( + async ( + action: SessionControlAction, + args?: SessionControlActionArgs, + onFailure?: (message: string) => void + ): Promise => { + onFeedback(null, null) + + try { + const dispatch = await runSessionControlAction(sessionId, action, args) + + if (dispatch.type === 'send') { + if (!dispatch.message || !onSubmit) { + onFeedback(ctrl.continuationFailed, null) + onFailure?.(ctrl.continuationFailed) + + return false + } + + // The backend has already resumed the goal; if a turn is running the + // kickoff must queue (same as a typed `/goal resume`, slash.ts), not + // read as a failure. + const queued = queueKickoffIfSessionBusy({ + displayText: dispatch.display ?? undefined, + sessionId, + text: dispatch.message + }) + + if (queued !== 'idle') { + const copy = queued === 'queued' ? ctrl.continuationQueued : ctrl.continuationBusy + onFeedback(queued === 'queued' ? null : copy, queued === 'queued' ? copy : null) + + return queued === 'queued' + } + + const submitted = await onSubmit(dispatch.message, { + displayKind: 'hidden', + sessionId + }) + + if (!submitted) { + onFeedback(ctrl.continuationFailed, null) + onFailure?.(ctrl.continuationFailed) + + return false + } + } + + onFeedback(null, ctrl.actionSucceeded) + + return true + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + const failure = ctrl.actionFailed(msg) + onFeedback(failure, null) + onFailure?.(failure) + + return false + } + }, + [sessionId, onSubmit, onFeedback, ctrl] + ) + + const copyCriterionText = useCallback( + async (text: string) => { + try { + await navigator.clipboard.writeText(text) + onFeedback(null, ctrl.copySuccess) + } catch { + onFeedback(ctrl.copyFailure, null) + } + }, + [onFeedback, ctrl] + ) + + const visibleState: 'waiting' | 'active' | 'paused' | 'done' = goal.wait_barrier + ? 'waiting' + : goal.status === 'paused' + ? 'paused' + : goal.status === 'done' + ? 'done' + : 'active' + + const iconClass = + goal.last_verdict === 'blocked' + ? 'text-red-500' + : visibleState === 'done' + ? 'text-muted-foreground/70' + : visibleState === 'active' + ? 'text-emerald-500' + : 'text-amber-500' + + const stateLabel = + goal.last_verdict === 'blocked' + ? s.goalBlocked + : visibleState === 'waiting' + ? s.goalWaiting + : visibleState === 'paused' + ? s.goalPaused + : visibleState === 'done' + ? s.goalDone + : s.goalActive + + const headerLabel = + visibleState === 'done' + ? `${stateLabel} · ${ctrl.goalDoneTurns(goal.turns_used)}` + : goal.max_turns > 0 + ? `${stateLabel} · ${ctrl.goalActiveTurns(goal.turns_used, goal.max_turns)}` + : goal.turns_used > 0 + ? `${stateLabel} · ${ctrl.goalTurn(goal.turns_used)}` + : stateLabel + + const confirmClearGoal = () => { + setConfirmState({ + title: ctrl.clearGoalConfirmTitle, + description: ctrl.clearGoalConfirmBody, + destructive: true, + confirmLabel: ctrl.clearGoal, + onConfirm: async () => { + await handleAction('goal.clear') + } + }) + } + + const confirmRemoveCriterion = (index: number) => { + setConfirmState({ + title: ctrl.removeCriterionConfirmTitle(index), + description: ctrl.removeCriterionConfirmBody(index), + destructive: true, + confirmLabel: ctrl.removeCriterion(index), + onConfirm: async () => { + await handleAction('subgoal.remove', { index }) + } + }) + } + + const confirmClearCriteria = () => { + setConfirmState({ + title: ctrl.clearCriteriaConfirmTitle, + description: ctrl.clearCriteriaConfirmBody, + destructive: true, + confirmLabel: ctrl.clearCriteria, + onConfirm: async () => { + await handleAction('subgoal.clear') + } + }) + } + + const openAddCriterion = () => { + setAddCriterionError(null) + setAddCriterionOpen(true) + } + + const renderMenuItems = (isContext = false) => { + const Item = isContext ? ContextMenuItem : DropdownMenuItem + const Sep = isContext ? ContextMenuSeparator : DropdownMenuSeparator + + return ( + <> + {hasDetails && ( + setDetailsOpen(true)}> + + {ctrl.viewDetails} + + )} + {visibleState !== 'done' && ( + + + {ctrl.addCriterion} + + )} + {(hasDetails || visibleState !== 'done') && } + {visibleState === 'active' && ( + void handleAction('goal.pause')}> + + {ctrl.pauseGoal} + + )} + {visibleState === 'paused' && ( + void handleAction('goal.resume')}> + + {ctrl.resumeGoal} + + )} + {visibleState === 'waiting' && ( + <> + void handleAction('goal.unwait')}> + + {ctrl.resumeNow} + + void handleAction('goal.pause')}> + + {ctrl.pauseGoal} + + + )} + + + {ctrl.clearGoal} + + + ) + } + + const hasDetails = Boolean( + goal.contract.outcome || + goal.contract.verification || + goal.contract.constraints || + goal.contract.boundaries || + goal.contract.stop_when || + goal.wait_barrier || + goal.gates.length > 0 + ) + + return ( + <> + + +
+ + + + + + + + + + {renderMenuItems(false)} + + + } + defaultCollapsed={false} + icon={} + label={headerLabel} + > +
+ {/* Full goal title */} +
{goal.title}
+ + {/* Optional reasons */} + {!detailsOpen && goal.wait_barrier && ( +
+ {goal.wait_barrier.reason + ? `${ctrl.waitBarrierTitle}: ${goal.wait_barrier.reason}` + : ctrl.waitBarrierTitle} +
+ )} + {!goal.wait_barrier && goal.paused_reason && ( +
{goal.paused_reason}
+ )} + {!goal.wait_barrier && !goal.paused_reason && goal.last_reason && ( +
{goal.last_reason}
+ )} + + {/* View details button */} + {hasDetails && ( +
+ +
+ )} + + {/* Criteria subsection */} +
+
+
+ + {ctrl.criteriaHeader(goal.subgoals.length)} +
+
+ + {goal.subgoals.length > 0 && ( + + )} +
+
+ + {goal.subgoals.length > 0 ? ( +
+ {goal.subgoals.map((subgoal, idx) => { + const index = idx + 1 + + return ( +
+
+ {index}. + {subgoal} +
+
+ + + + + + +
+
+ ) + })} +
+ ) : null} +
+
+
+
+
+ {renderMenuItems(true)} +
+ + {/* Add Criterion Dialog */} + { + if (!open) { + setAddCriterionOpen(false) + setAddCriterionError(null) + } + }} + open={addCriterionOpen} + > + +
{ + e.preventDefault() + const text = newCriterionText.trim() + + if (!text || isBusy) { + return + } + + setAddCriterionError(null) + const ok = await handleAction('subgoal.add', { text }, setAddCriterionError) + + if (ok) { + setNewCriterionText('') + setAddCriterionOpen(false) + } + }} + > + + {ctrl.addCriterionDialogTitle} + {ctrl.addCriterionPlaceholder} + +
+