review-fix(comments): restore lost #NNNN rationale comments across non-test source (mechanical sweep, condensed, code unchanged)

For each issue anchor present in BASE 63279301bc non-test .py and absent on HEAD, the BASE comment/docstring block was re-attached at the HEAD location of the code it explained (matched by the distinctive code line / enclosing def). Sentences already covered by an existing HEAD comment were deduped; the issue number always survives. Insert-only: no code lines changed.
This commit is contained in:
Teknium
2026-09-03 09:44:26 -07:00
parent ad62a15e94
commit e83816a4d1
586 changed files with 13883 additions and 829 deletions
+2
View File
@@ -236,6 +236,8 @@ class SlashCommandsMixin:
original_count = len(state.history) original_count = len(state.history)
# Include system prompt + tool schemas so the figure reflects real request pressure. # Include system prompt + tool schemas so the figure reflects real request pressure.
# See #6217.
# See #6217.
_sys_prompt = getattr(agent, "_cached_system_prompt", "") or "" _sys_prompt = getattr(agent, "_cached_system_prompt", "") or ""
_tools = getattr(agent, "tools", None) or None _tools = getattr(agent, "tools", None) or None
approx_tokens = _estimate_tokens(state.history, agent, _sys_prompt, _tools) approx_tokens = _estimate_tokens(state.history, agent, _sys_prompt, _tools)
+3
View File
@@ -103,6 +103,9 @@ def _decode_text_bytes(data: bytes, mime_type: str | None) -> str | None:
return data.decode(encoding) return data.decode(encoding)
except UnicodeDecodeError: except UnicodeDecodeError:
continue continue
# Binary (ELF/Mach-O/PE), not a shell script: feeding its decoded bytes back into the guard tokenizes
# machine code into bogus NUL-bearing paths and crashes the scanner (#77703). Mirror
# lifecycle_guard._read_referenced_script and treat it as nothing to scan.
return data.decode("utf-8", errors="replace") return data.decode("utf-8", errors="replace")
+3
View File
@@ -185,6 +185,9 @@ def main(argv: list[str] | None = None) -> None:
# MCP discovery from config.yaml runs in a background daemon thread so the ACP server is # MCP discovery from config.yaml runs in a background daemon thread so the ACP server is
# responsive immediately (blocking here cost 2-5 s); per-session MCP servers registered via # responsive immediately (blocking here cost 2-5 s); per-session MCP servers registered via
# asyncio.to_thread are unaffected. Metadata-only hosts can opt out of the global startup. # asyncio.to_thread are unaffected. Metadata-only hosts can opt out of the global startup.
# Previously this blocked asyncio.run() for 2-5 s. (ACP also registers per-session MCP servers
# dynamically via asyncio.to_thread inside the event loop; that path is unaffected.) Moved from
# model_tools.py module scope to avoid freezing the gateway's loop on lazy import (#16856).
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1": if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
try: try:
from hermes_cli.mcp_startup import start_background_mcp_discovery from hermes_cli.mcp_startup import start_background_mcp_discovery
+1
View File
@@ -37,6 +37,7 @@ def _build_permission_options(
# A gate that re-asks every time (allow_session=False, e.g. protected # A gate that re-asks every time (allow_session=False, e.g. protected
# agent-instruction writes) collapses to the same two options as a Smart # agent-instruction writes) collapses to the same two options as a Smart
# DENY override — offering a scope Hermes discards would re-prompt every write. # DENY override — offering a scope Hermes discards would re-prompt every write.
# See #81887.
once_only = smart_denied or not allow_session once_only = smart_denied or not allow_session
options = [PermissionOption(option_id="allow_once", kind="allow_once", name="Allow once")] options = [PermissionOption(option_id="allow_once", kind="allow_once", name="Allow once")]
if not once_only: if not once_only:
+8
View File
@@ -554,6 +554,14 @@ class HermesACPAgent(SlashCommandsMixin, acp.Agent):
Best-effort: a corrupt message must not turn the load into an error.""" Best-effort: a corrupt message must not turn the load into an error."""
if replay_verb: if replay_verb:
try: try:
# Per ACP spec, `session/load` must stream the prior conversation back to the client via
# `session/update` notifications BEFORE responding, so the client receives the full
# transcript within the load request's lifetime. Awaiting the replay here matches Codex /
# Claude Code / OpenCode / Pi and the Zed client (which registers the session-update routing
# entry before awaiting the loadSession RPC specifically so in-call history replay updates
# can find the thread). Deferring this via `loop.call_soon` (as we did briefly in May 2026)
# broke every spec-compliant ACP client that measures notifications synchronously against
# the load response — see #12285 follow-up.
await self._replay_session_history(state) await self._replay_session_history(state)
except Exception: except Exception:
logger.warning( logger.warning(
+13
View File
@@ -44,6 +44,11 @@ def _normalize_cwd_for_compare(cwd: str | None) -> str:
# ``/private/tmp``) that otherwise drop a workspace's own sessions; it is lexical # ``/private/tmp``) that otherwise drop a workspace's own sessions; it is lexical
# for missing paths (e.g. WSL-translated drives). # for missing paths (e.g. WSL-translated drives).
try: try:
# Resolve symlink aliases so equivalent spellings of the same directory compare equal — macOS
# reports editor workspaces as ``/var/...`` while sessions get stored under ``/private/var/...``
# (and ``/tmp`` vs ``/private/tmp``), which made ACP history filters silently drop a workspace's own
# sessions. WSL-translated Windows drives — keep the previous normpath behavior. Ported from
# PrimeIntellect-ai/prime-agent#628.
return os.path.realpath(expanded) return os.path.realpath(expanded)
except OSError: except OSError:
return os.path.normpath(expanded) return os.path.normpath(expanded)
@@ -340,6 +345,14 @@ class SessionManager:
# incrementally (append_message) and keeps pre-compaction turns as archived # incrementally (append_message) and keeps pre-compaction turns as archived
# active=0 rows; replace_messages() would DELETE those (and, after a compression # active=0 rows; replace_messages() would DELETE those (and, after a compression
# id rotation, clobber the ended parent transcript). Skip it in that case. # id rotation, clobber the ended parent transcript). Skip it in that case.
# Calling replace_messages() here would then be a redundant double-write that DELETEs exactly
# those archived rows (and, after a compression-driven id rotation where agent.session_id no
# longer equals state.session_id, clobbers the ended parent transcript) — silent data loss for
# any ACP conversation long enough to compress. Only fall back to the destructive atomic replace
# when the agent is NOT persisting itself to this DB (e.g. a test agent factory, or a fresh
# create/fork whose copied history the agent has not flushed yet). That path still rolls back on
# a mid-rewrite failure so the previously persisted conversation survives (salvaged from
# #13675).
agent = state.agent agent = state.agent
if getattr(agent, "_session_db", None) is db and getattr(agent, "_session_db_created", False): if getattr(agent, "_session_db", None) is db and getattr(agent, "_session_db_created", False):
return return
+5
View File
@@ -306,6 +306,11 @@ def _format_read_file_result(tool_name: str, data: Args, a: Args) -> Optional[st
@_structured() @_structured()
def _format_search_files_result(tool_name: str, data: Args, args: Args) -> Optional[str]: def _format_search_files_result(tool_name: str, data: Args, args: Args) -> Optional[str]:
files, matches = data.get("files"), data.get("matches") files, matches = data.get("files"), data.get("matches")
# Surface file/image attachments as compact text markers. The thread-context fetch is text-only, so
# without this the agent has no idea prior messages carried images/files at all (#69185, #32315): "@bot
# what do you think of the chart above?" reads as a question about nothing. Markers keep context bounded
# — the agent can ask for a re-share (or the caller may separately deliver the thread root's image, see
# _collect_thread_root_images).
if isinstance(files, list): if isinstance(files, list):
shown = min(len(files), 20) shown = min(len(files), 20)
lines = ["File search results", f"Found {_plural(data.get('total_count', len(files)), 'file')}; showing {shown}.", ""] lines = ["File search results", f"Found {_plural(data.get('total_count', len(files)), 'file')}; showing {shown}.", ""]
+5
View File
@@ -304,6 +304,11 @@ def _resolve_codex_usage_credentials(
# and hand back a DIFFERENT pool account's usage; such errors must propagate to the fail-open outer guard. # and hand back a DIFFERENT pool account's usage; such errors must propagate to the fail-open outer guard.
# account_id is best-effort: a partial singleton store must not sink a usable credential. # account_id is best-effort: a partial singleton store must not sink a usable credential.
try: try:
# Tier 2: the native runtime resolver. It ALREADY falls back to the credential pool when the
# singleton is empty (see ``resolve_codex_runtime_credentials`` — issue #32992), so in a pool-only
# setup this returns a usable ``source="credential_pool"`` token. A refresh/network error must
# propagate — the outer ``fetch_account_usage`` guard fails open (shows nothing this turn) rather
# than reporting the wrong account.
creds = resolve_codex_runtime_credentials(refresh_if_expiring=True) creds = resolve_codex_runtime_credentials(refresh_if_expiring=True)
account_id: Optional[str] = None account_id: Optional[str] = None
try: try:
+7
View File
@@ -33,6 +33,8 @@ class ActivityTrackingMixin:
``_touch_activity`` stamps under it and the liveness watchdog samples/commits under it, so a stall ``_touch_activity`` stamps under it and the liveness watchdog samples/commits under it, so a stall
observation can never abort a turn that resumed in between. observation can never abort a turn that resumed in between.
Created lazily so ``AIAgent.__new__``-based test doubles keep working. See #95663.
""" """
return _activity_lock(self) return _activity_lock(self)
@@ -48,6 +50,9 @@ class ActivityTrackingMixin:
projection. ``provenance`` names special writers (compression); ``force_persist`` bypasses the projection. ``provenance`` names special writers (compression); ``force_persist`` bypasses the
SessionDB rate limit. Module-level lock helper, not ``self._liveness_activity_lock()``: doubles bind SessionDB rate limit. Module-level lock helper, not ``self._liveness_activity_lock()``: doubles bind
only ``_touch_activity`` (tests/run_agent/test_session_activity_persist.py). only ``_touch_activity`` (tests/run_agent/test_session_activity_persist.py).
Bridge is rate-limited (60s) and best-effort — it never raises into the agent loop. See #31752.
See #72016, #72039.
""" """
from agent.session_activity import ( from agent.session_activity import (
bound_activity_description, normalize_activity_provenance, bound_activity_description, normalize_activity_provenance,
@@ -117,6 +122,8 @@ class ActivityTrackingMixin:
Keeps ``_last_activity_ts`` so idle/watchdog clocks stay continuous across turns; clears description + Keeps ``_last_activity_ts`` so idle/watchdog clocks stay continuous across turns; clears description +
provenance so idle agents / SessionDB listings stop advertising the last mid-turn stamp. provenance so idle agents / SessionDB listings stop advertising the last mid-turn stamp.
See #15654, #72039.
""" """
self._last_activity_desc = "" self._last_activity_desc = ""
self._last_activity_provenance = ActivityProvenance.UNKNOWN self._last_activity_provenance = ActivityProvenance.UNKNOWN
+61
View File
@@ -443,6 +443,11 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url):
# Covers api.meta.ai → codex_responses (prompt caching: 0% on chat vs 93-99%). # Covers api.meta.ai → codex_responses (prompt caching: 0% on chat vs 93-99%).
# URL-driven, not provider-name-driven: `providers.meta` may point anywhere. # URL-driven, not provider-name-driven: `providers.meta` may point anywhere.
try: try:
# Note: provider="meta" without an api.meta.ai base_url (or with a non-api.meta.ai base_url)
# intentionally falls through to chat_completions here. The wire protocol for Meta is URL-driven
# BY DESIGN, not provider-name-driven, because user config `providers.meta` may point at any
# OpenAI-compatible endpoint, and forcing `codex_responses` on the provider name alone would
# break custom endpoints named "meta" that do not host the Responses API. See #63425.
from hermes_cli.providers import host_mandated_api_mode as _host_mandated_api_mode from hermes_cli.providers import host_mandated_api_mode as _host_mandated_api_mode
_mandated = _host_mandated_api_mode(base_url or "") _mandated = _host_mandated_api_mode(base_url or "")
except Exception: except Exception:
@@ -453,6 +458,8 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url):
def _finalize_routing(agent, api_mode, credential_pool): def _finalize_routing(agent, api_mode, credential_pool):
# Credential-pool validation runs AFTER provider auto-detection so a pool scoped to # Credential-pool validation runs AFTER provider auto-detection so a pool scoped to
# "anthropic" isn't rejected for provider=None + anthropic.com URL. # "anthropic" isn't rejected for provider=None + anthropic.com URL.
# Regression from #63048 which placed this check before the URL-based auto-detection block above (fixed
# #63425).
if credential_pool is not None: if credential_pool is not None:
try: try:
from agent.credential_pool import credential_pool_matches_provider from agent.credential_pool import credential_pool_matches_provider
@@ -481,6 +488,13 @@ def _finalize_routing(agent, api_mode, credential_pool):
# exceptions live in _provider_model_requires_responses_api. # exceptions live in _provider_model_requires_responses_api.
_base_lower = str(agent.base_url or "").lower() _base_lower = str(agent.base_url or "").lower()
if ( if (
# GPT-5.x models usually require the Responses API path, but some providers have exceptions (for
# example Copilot's gpt-5-mini still uses chat completions). ACP runtimes are excluded: an ACP
# client handles its own routing and does not implement the Responses API surface. Keyed on the
# `acp://` scheme, not one vendor, so every ACP client is covered. When api_mode was explicitly
# provided, respect it — the user knows what their endpoint supports (#10473). Exception: Azure
# OpenAI serves gpt-5.x on /chat/completions and does NOT support the Responses API — skip the
# upgrade for Azure (openai.azure.com), even though it looks OpenAI-compatible.
api_mode is None api_mode is None
and agent.api_mode == "chat_completions" and agent.api_mode == "chat_completions"
and agent.provider != "copilot-acp" and agent.provider != "copilot-acp"
@@ -647,6 +661,11 @@ def _init_prompt_cache_config(agent):
# unknown values keep "5m". A falsy/off value disables caching entirely (OAuth plans # unknown values keep "5m". A falsy/off value disables caching entirely (OAuth plans
# billing cache writes, proxies adding their own cache_control); the disable survives # billing cache writes, proxies adding their own cache_control); the disable survives
# /model switches and fallback re-derivation. # /model switches and fallback re-derivation.
# Anthropic supports "5m" (default) and "1h" cache TTL tiers. Read from config.yaml under
# prompt_caching.cache_ttl; unknown values keep "5m". 1h tier costs 2x on write vs 1.25x for 5m, but
# amortizes across long sessions with >5-minute pauses between turns (#14971). This is useful for OAuth
# subscription users where cache writes bill against "extra usage" or for third-party proxies that
# inject their own cache_control markers (#13477).
agent._cache_ttl = "5m" agent._cache_ttl = "5m"
with suppress(Exception): with suppress(Exception):
from hermes_cli.config import load_config_readonly as _load_pc_cfg from hermes_cli.config import load_config_readonly as _load_pc_cfg
@@ -721,6 +740,7 @@ def _init_anthropic_client(agent, api_key, base_url, _provider_timeout):
return return
# ANTHROPIC_TOKEN fallback only for native Anthropic — other anthropic_messages providers # ANTHROPIC_TOKEN fallback only for native Anthropic — other anthropic_messages providers
# must use their own key or Anthropic credentials leak to third-party endpoints. # must use their own key or Anthropic credentials leak to third-party endpoints.
# Falling back would send Anthropic credentials to third-party endpoints (Fixes #1739, #minimax-401).
_is_native_anthropic = agent.provider == "anthropic" _is_native_anthropic = agent.provider == "anthropic"
effective_key = api_key or (resolve_anthropic_token() if _is_native_anthropic else None) or "" effective_key = api_key or (resolve_anthropic_token() if _is_native_anthropic else None) or ""
@@ -742,6 +762,10 @@ def _init_anthropic_client(agent, api_key, base_url, _provider_timeout):
agent._anthropic_api_key = effective_key agent._anthropic_api_key = effective_key
# OAuth only for native Anthropic: third-party anthropic_messages providers must never # OAuth only for native Anthropic: third-party anthropic_messages providers must never
# trip OAuth paths — those inject Claude-Code identity headers → 401/403. # trip OAuth paths — those inject Claude-Code identity headers → 401/403.
# Only mark the session as OAuth-authenticated when the token genuinely belongs to native Anthropic.
# Third-party providers (MiniMax, Kimi, GLM, LiteLLM proxies) that accept the Anthropic protocol must
# never trip OAuth code paths — doing so injects Claude-Code identity headers and system prompts that
# cause 401/403 on their endpoints. See #1739.
from agent.anthropic_adapter import _is_oauth_token as _is_oat from agent.anthropic_adapter import _is_oauth_token as _is_oat
agent._is_anthropic_oauth = _is_oat(effective_key) if (_is_native_anthropic and isinstance(effective_key, str)) else False agent._is_anthropic_oauth = _is_oat(effective_key) if (_is_native_anthropic and isinstance(effective_key, str)) else False
agent._anthropic_client = build_anthropic_client(effective_key, base_url, timeout=_provider_timeout) agent._anthropic_client = build_anthropic_client(effective_key, base_url, timeout=_provider_timeout)
@@ -758,6 +782,12 @@ def _init_moa_client(agent, api_key):
# build_moa_facade relays "moa.*" events through tool_progress_callback so every surface # build_moa_facade relays "moa.*" events through tool_progress_callback so every surface
# shows each reference's answer before the aggregator acts. Display-only; shared with # shows each reference's answer before the aggregator acts. Display-only; shared with
# fallback-restore so a restored facade keeps emitting. # fallback-restore so a restored facade keeps emitting.
# build_moa_facade wires the reference relay that routes reference-model outputs to the agent's
# tool_progress_callback so every surface that already consumes it (CLI spinner/scrollback, TUI,
# desktop, gateway) can show each reference's answer as a labelled block before the aggregator acts. The
# facade emits "moa.reference", "moa.progress", "moa.phase", and "moa.aggregating" events, forwarded
# through the same callback the tool lifecycle uses. Best-effort and cache-safe — display-only events,
# they never touch the message history. See #53802.
agent.client = build_moa_facade(agent, agent.model) agent.client = build_moa_facade(agent, agent.model)
agent._client_kwargs = {} agent._client_kwargs = {}
agent.api_key = api_key or "moa-virtual-provider" agent.api_key = api_key or "moa-virtual-provider"
@@ -837,6 +867,9 @@ def _routed_client_kwargs(agent, fallback_model, _provider_timeout) -> Dict[str,
# No credentials: try the fallback chain BEFORE failing (an exhausted single-entry pool # No credentials: try the fallback chain BEFORE failing (an exhausted single-entry pool
# must not die with a misleading "No LLM provider configured"); only explicitly named # must not die with a misleading "No LLM provider configured"); only explicitly named
# providers keep the missing-key diagnostic. # providers keep the missing-key diagnostic.
# An exhausted single-entry pool (typically ``openrouter`` under free-tier daily quotas) must still
# reach the chain instead of dying at init with a misleading "No LLM provider configured" error. See
# #17929.
_explicit = (agent.provider or "").strip().lower() _explicit = (agent.provider or "").strip().lower()
for _fb in _fallback_entries(fallback_model): for _fb in _fallback_entries(fallback_model):
try: try:
@@ -1237,6 +1270,11 @@ def _init_memory(agent, _agent_cfg, skip_memory, platform):
agent._iters_since_skill = 0 agent._iters_since_skill = 0
# skip_memory skips the external *provider*; enabled_toolsets=["memory"] still gets the # skip_memory skips the external *provider*; enabled_toolsets=["memory"] still gets the
# built-in store so the memory tool never sees store=None. # built-in store so the memory tool never sees store=None.
# Flush/background agents can still pass enabled_toolsets=["memory"] so the built-in file store exists
# and the memory tool does not fail with store=None (#65429). A toolset on disabled_toolsets is not a
# request: a caller that denylists memory while its default toolset still names it must not get
# MEMORY.md loaded by an enabled-only check. (Cron agents now run with skip_memory=False and take the
# normal path here.)
_memory_toolset_requested = ( _memory_toolset_requested = (
"memory" in (agent.enabled_toolsets or []) "memory" in (agent.enabled_toolsets or [])
and "memory" not in (agent.disabled_toolsets or []) and "memory" not in (agent.disabled_toolsets or [])
@@ -1768,6 +1806,9 @@ def _select_context_engine(_agent_cfg):
# parent's. Uncopyable state (locks, DB conns) → built-in with an ACCURATE message. # parent's. Uncopyable state (locks, DB conns) → built-in with an ACCURATE message.
import copy import copy
try: try:
# Copy can fail for engines holding uncopyable state (locks, DB connections, clients); in
# that case fall back to the built-in compressor with an ACCURATE message rather than
# silently mislabelling it "not found". See #42449.
_selected_engine = copy.deepcopy(_candidate) _selected_engine = copy.deepcopy(_candidate)
except Exception as _copy_err: except Exception as _copy_err:
_copy_failed = True _copy_failed = True
@@ -1812,6 +1853,9 @@ def _build_context_engine(agent, _agent_cfg, cs, _custom_providers, _effective_c
# External engines own compaction policy — the host threshold (and its Codex # External engines own compaction policy — the host threshold (and its Codex
# autoraise) never reaches the plugin, so drop the notice. # autoraise) never reaches the plugin, so drop the notice.
agent._compression_threshold_autoraised = None agent._compression_threshold_autoraised = None
# External engines own compaction policy: the host compression threshold (including the Codex
# gpt-5.5 autoraise above) only configures the built-in ContextCompressor and never reaches the
# plugin, so the autoraise notice would announce a change that does not apply. (#44439)
from agent.model_metadata import get_model_context_length from agent.model_metadata import get_model_context_length
_plugin_ctx_len = get_model_context_length( _plugin_ctx_len = get_model_context_length(
agent.model, base_url=agent.base_url, api_key=getattr(agent, "api_key", ""), agent.model, base_url=agent.base_url, api_key=getattr(agent, "api_key", ""),
@@ -1924,6 +1968,14 @@ def _inject_context_engine_tools(agent):
# Context engine tool schemas (lcm_*), deduped against existing names (plugins may # Context engine tool schemas (lcm_*), deduped against existing names (plugins may
# register the same schemas; duplicates 400 provider-side) and gated on enabled_toolsets # register the same schemas; duplicates 400 provider-side) and gated on enabled_toolsets
# so `platform_toolsets: telegram: []` can't leak them. # so `platform_toolsets: telegram: []` can't leak them.
# Skip names that are already present — the _ra().get_tool_definitions() quiet_mode cache returned a
# shared list pre-#17335, so a stray mutation here would poison subsequent agent inits in the same
# Gateway process and trip provider-side 'duplicate tool name' errors. Even with the cache fix, dedup is
# the right defense against plugin paths that may register the same schemas via ctx.register_tool().
# Mirrors the memory tools dedup above. Respect the platform's enabled_toolsets configuration (#5544):
# context engine tools follow the same gating pattern as memory provider tools — without the gate,
# `platform_toolsets: telegram: []` would still leak lcm_* tools into the tool surface and incur the
# same local-model latency penalty.
agent._context_engine_tool_names: set = set() agent._context_engine_tool_names: set = set()
if ( if (
agent.context_compressor agent.context_compressor
@@ -1939,6 +1991,7 @@ def _inject_context_engine_tools(agent):
if _schema is None: if _schema is None:
# A nameless tool makes strict providers 400 and disables the whole toolset. # A nameless tool makes strict providers 400 and disables the whole toolset.
_ra().logger.warning( _ra().logger.warning(
# Skip it. See #47707.
"Context engine returned a tool schema with no resolvable " "Context engine returned a tool schema with no resolvable "
"name; skipping to avoid poisoning the request (%r)", "name; skipping to avoid poisoning the request (%r)",
_raw_schema, _raw_schema,
@@ -2002,6 +2055,11 @@ def _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length):
) )
# Recalibrate the compressor to the served window: every request runs at num_ctx, so a # Recalibrate the compressor to the served window: every request runs at num_ctx, so a
# trigger derived from the probed model window could sit above it and never fire. # trigger derived from the probed model window could sit above it and never fire.
# A config that sets only model.ollama_num_ctx (without model.context_length) previously left the
# compressor targeting the probed window while the server truncated/rejected at num_ctx — the compaction
# trigger could sit several times ABOVE the real served window and never fire. Clamp the compressor's
# window to the effective num_ctx so threshold math operates on the context the server actually serves.
# (Overlaps #60103's silent-clamp dead zone; this is the init-order half.)
_cc_window = getattr(agent.context_compressor, "context_length", 0) or 0 _cc_window = getattr(agent.context_compressor, "context_length", 0) or 0
if agent._ollama_num_ctx and agent._ollama_num_ctx > 0 and _cc_window and agent._ollama_num_ctx < _cc_window: if agent._ollama_num_ctx and agent._ollama_num_ctx > 0 and _cc_window and agent._ollama_num_ctx < _cc_window:
_ra().logger.info( _ra().logger.info(
@@ -2020,6 +2078,9 @@ def _emit_compression_summary(agent, cs):
_autoraise = agent._compression_threshold_autoraised or {} _autoraise = agent._compression_threshold_autoraised or {}
_autoraise_notice = None _autoraise_notice = None
if ( if (
# A change in the raised threshold (or the autoraised model) updates the marker state and
# re-notifies once. The config display gate (compression.codex_gpt55_autoraise_notice) still
# suppresses the banner entirely without disabling the threshold autoraise. See #54432.
bool(_autoraise) bool(_autoraise)
and cs.enabled and cs.enabled
and cs.autoraise_notice_enabled and cs.autoraise_notice_enabled
+236 -4
View File
@@ -223,6 +223,14 @@ def sanitize_tool_call_arguments(
``cursor["prefix"]`` holds strong refs (not ``id()``: address reuse aliases) to the ``cursor["prefix"]`` holds strong refs (not ``id()``: address reuse aliases) to the
messages validated last call; the ``is``-identical prefix is skipped. Safe because only messages validated last call; the ``is``-identical prefix is skipped. Safe because only
the surrogate sanitizers mutate live dicts; every other path replaces dicts, breaking identity. the surrogate sanitizers mutate live dicts; every other path replaces dicts, breaking identity.
Safety argument for skipping: a message in the matched prefix was fully scanned before — every tool_call
argument was either already valid JSON or was rewritten to ``"{}"`` (valid). The only code paths that
mutate ``function["arguments"]`` on live history dicts between calls are the surrogate / non-ASCII
sanitizers, which substitute characters *inside* JSON string values and cannot invalidate JSON syntax.
Compression, repair, undo, and steer paths replace or reorder message dicts, which breaks the identity
match and forces a re-scan. Holding strong references (the objects themselves, not ``id()``s) makes
address reuse aliasing (#50372-style) impossible.
""" """
log = logger or logging.getLogger(__name__) log = logger or logging.getLogger(__name__)
if not isinstance(messages, list): if not isinstance(messages, list):
@@ -252,6 +260,8 @@ def sanitize_tool_call_arguments(
continue continue
# Canonical ``call_id || id`` precedence so scan and stub share the id the pipeline # Canonical ``call_id || id`` precedence so scan and stub share the id the pipeline
# uses; bare ``id`` misses Codex call_id results and orphans a stub. # uses; bare ``id`` misses Codex call_id results and orphans a stub.
# Keying on bare ``id`` here would fail to find a result built with ``call_id`` (Codex Responses
# format) and insert a duplicate stub that itself becomes an orphan (#58168).
tool_call_id = _ra().AIAgent._get_tool_call_id_static(tool_call) or None tool_call_id = _ra().AIAgent._get_tool_call_id_static(tool_call) or None
function_name = function.get("name", "?") function_name = function.get("name", "?")
# Log the FULL (bounded) argument string: we are about to overwrite the only copy, which # Log the FULL (bounded) argument string: we are about to overwrite the only copy, which
@@ -365,6 +375,15 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None:
else: else:
# Drop a stale ``tool_calls: []`` at the source: strict providers (DeepSeek v4, Kimi) 400 on # Drop a stale ``tool_calls: []`` at the source: strict providers (DeepSeek v4, Kimi) 400 on
# it and it persists into replayed history. # it and it persists into replayed history.
# Neither turn carries tool calls, but the surviving turn may still carry a stale ``tool_calls: []``
# from the earlier message. An empty array is semantically "no tool calls", yet strict
# OpenAI-compatible providers (DeepSeek v4, Moonshot/Kimi) reject it with HTTP 400 ("Invalid
# 'messages[N].tool_calls': empty array..."). Drop the key HERE, at the source:
# ``sanitize_api_messages`` only fixes the per-call wire copy, so a ``[]`` left on the repaired turn
# survives in the live/persisted trajectory returned to callers (gateway/WebUI transcripts, session
# resume, subagents, cron) and is replayed on the next turn — which is how #58755 kept reproducing
# after the chokepoint fix (#77921). Popping is non-destructive: an empty array carries no
# information.
prev.pop("tool_calls", None) prev.pop("tool_calls", None)
# Concatenate plain-text content only; leave multimodal (list) content alone. # Concatenate plain-text content only; leave multimodal (list) content alone.
prev_content = prev.get("content") prev_content = prev.get("content")
@@ -374,6 +393,8 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None:
joined = "\n".join(p for p in (prev_content.strip(), new_content.strip()) if p) joined = "\n".join(p for p in (prev_content.strip(), new_content.strip()) if p)
prev["content"] = joined prev["content"] = joined
# A falsy new_content leaves ``joined`` == prev_content; that is not a rewrite. # A falsy new_content leaves ``joined`` == prev_content; that is not a rewrite.
# "") strips to nothing and ``joined`` collapses back to ``prev_content`` unchanged -- that must NOT
# count as a rewrite (wz-heng, #78063 review).
content_rewritten = joined != prev_content content_rewritten = joined != prev_content
elif not prev_content and new_content is not None: elif not prev_content and new_content is not None:
prev["content"] = new_content prev["content"] = new_content
@@ -384,6 +405,18 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None:
prev["reasoning_content"] = msg["reasoning_content"] prev["reasoning_content"] = msg["reasoning_content"]
# A stale ``api_content`` sidecar overrides ``content`` at API-build time and would replay # A stale ``api_content`` sidecar overrides ``content`` at API-build time and would replay
# pre-merge bytes; drop it only when content actually changed. # pre-merge bytes; drop it only when content actually changed.
# ``prev`` may carry an ``api_content`` sidecar (the exact bytes previously sent to the API, e.g. a
# sanitize-divergence stamp — see ``_flush_messages_to_session_db``) from BEFORE this merge. The sidecar
# takes priority over ``content`` at API-build time (``conversation_loop``'s ``api_messages`` build
# substitutes it back in for role ``assistant``), so leaving it in place while ``prev["content"]``
# changes would silently replay the pre-merge bytes and discard everything this merge just concatenated
# on — the same stale-field-survives-the-merge shape as the ``tool_calls`` gap above, just for a
# different field. Only drop it when the merge actually changed the resulting value (e.g. the later
# turn's content is ``None``, or either side is multimodal/list — both branches skip the reassignment
# and ``prev["content"]`` is untouched; a falsy ``new_content`` that strips to nothing also leaves
# ``joined`` equal to the original ``prev_content``): in those cases the sidecar is still the exact
# bytes previously sent for the UNCHANGED content, and dropping it would break the prompt-cache replay
# invariant for no reason (wz-heng, #78063 review).
if content_rewritten: if content_rewritten:
drop_stale_api_content(prev) drop_stale_api_content(prev)
@@ -416,6 +449,11 @@ def _drop_stray_tool_results(messages: List[Dict]) -> Tuple[List[Dict], int]:
alias is not replayed to strict providers.""" alias is not replayed to strict providers."""
repairs = 0 repairs = 0
known_tool_ids: Dict[str, int] = {} # alias -> group id; reset by assistant/user turns known_tool_ids: Dict[str, int] = {} # alias -> group id; reset by assistant/user turns
# Pass 1: drop stray tool messages that don't follow a known assistant tool call. A Responses call can
# have several equivalent spellings (call_id, id, response_item_id, or a composite ``call|item`` id), so
# consume the whole alias group when one spelling is matched. Alias expansion lives in
# ``agent.message_sanitization.tool_call_id_variants`` / ``tool_result_id_variants`` (single policy
# owner) — which also handles SDK tool_call objects, preserving the #91768 dict-or-object tolerance.
matched_tool_groups: set = set() matched_tool_groups: set = set()
next_tool_group = 0 next_tool_group = 0
filtered: List[Dict] = [] filtered: List[Dict] = []
@@ -644,6 +682,18 @@ def _is_entitlement_403(agent, status_code, error_context) -> bool:
if status_code != 403: if status_code != 403:
return False return False
haystack = " ".join( haystack = " ".join(
# Subscription/entitlement 403s look like auth failures on the wire but refresh cannot fix them —
# the OAuth token is already valid, the account simply lacks the entitlement. Without this guard,
# the refresh path keeps minting fresh tokens against the same unsubscribed account and the main
# agent loop spins re-issuing the same 403 until the user Ctrl+C's. Defense-in-depth for #26847:
# xAI's backend has been seen to 403 standard SuperGrok subscribers with bodies that don't match the
# existing entitlement keyword set in ``_is_entitlement_failure``. Any 403 against ``xai-oauth`` is
# treated as entitlement here so the refresh loop can't spin in those cases either. Exception
# (#29344): xAI's ``[WKE=unauthenticated:...]`` suffix and the ``OAuth2 access token could not be
# validated`` phrasing are xAI's authoritative "this is a stale token, not entitlement" signal. When
# either fires we must NOT apply the catch-all override — refresh is the recoverable path for these
# bodies, and blanket-classifying them as entitlement was the bug that left long-running TUI
# sessions stuck on stale tokens until the user exited and reopened.
str(error_context.get(k) or "").lower() str(error_context.get(k) or "").lower()
for k in ("message", "reason", "code", "error") for k in ("message", "reason", "code", "error")
if isinstance(error_context, dict) if isinstance(error_context, dict)
@@ -746,6 +796,11 @@ def recover_with_credential_pool(
# The pool belongs to the PRIMARY provider: acting on fallback errors would corrupt its state # The pool belongs to the PRIMARY provider: acting on fallback errors would corrupt its state
# and reset base_url to the primary endpoint. Empty pool provider means unscoped; empty agent # and reset base_url to the primary endpoint. Empty pool provider means unscoped; empty agent
# provider is a mismatch (swap would leave provider="" model=""). # provider is a mismatch (swap would leave provider="" model="").
# Defensive guard: if a fallback provider is active and its provider name doesn't match the pool's
# provider, the pool belongs to the PRIMARY provider. Mutating it based on fallback errors would corrupt
# the primary's credential state (see #33088) and, via _swap_credential, overwrite the agent's base_url
# back to the primary's endpoint — every subsequent request then goes to the wrong host and 404s (see
# #33163). The pool should only act when the agent is still on the same provider that seeded the pool.
current_provider = (getattr(agent, "provider", "") or "").strip().lower() current_provider = (getattr(agent, "provider", "") or "").strip().lower()
pool_provider = (getattr(pool, "provider", "") or "").strip().lower() pool_provider = (getattr(pool, "provider", "") or "").strip().lower()
if pool_provider and not credential_pool_matches_provider( if pool_provider and not credential_pool_matches_provider(
@@ -860,6 +915,10 @@ def _rebuild_primary_client(agent, rt: Dict[str, Any], *, reason: str) -> None:
# reference_callback relay survives recovery. # reference_callback relay survives recovery.
from agent.moa_loop import build_moa_facade from agent.moa_loop import build_moa_facade
agent.client = build_moa_facade(agent, agent.model) agent.client = build_moa_facade(agent, agent.model)
# MoA is a virtual chat-completions provider. It never has real OpenAI client kwargs; restoring it
# after a fallback must recreate the facade, not call OpenAI() with an empty api_key. Use the shared
# factory so the restored facade keeps the reference_callback relay wired at init — a bare
# MoAClient() would silently stop emitting moa.reference/moa.aggregating display events (#53802).
agent._anthropic_client = None agent._anthropic_client = None
elif agent.api_mode == "anthropic_messages": elif agent.api_mode == "anthropic_messages":
_build_anthropic_client_from_runtime(agent, rt) _build_anthropic_client_from_runtime(agent, rt)
@@ -885,6 +944,11 @@ def try_recover_primary_transport(
try: try:
# Never hard-close the shared client here: stale streaming workers may still be unwinding on # Never hard-close the shared client here: stale streaming workers may still be unwinding on
# the old pool; _retire_shared_openai_client defers FD release to GC. # the old pool; _retire_shared_openai_client defers FD release to GC.
# Retire the existing client to release stale connections. #70773: never hard-close the shared
# client here — this runs on the conversation-loop thread while workers from stale-killed streaming
# attempts may still be unwinding their SSL BIOs on the old pool. ``_retire_shared_openai_client``
# shuts the sockets down (FD-safe from any thread) and defers the FD release to GC, which cannot
# complete until every borrowing thread has unwound.
if getattr(agent, "client", None) is not None: if getattr(agent, "client", None) is not None:
with contextlib.suppress(Exception): with contextlib.suppress(Exception):
agent._retire_shared_openai_client(agent.client, reason="primary_recovery") agent._retire_shared_openai_client(agent.client, reason="primary_recovery")
@@ -893,6 +957,9 @@ def try_recover_primary_transport(
if agent.api_mode == "anthropic_messages": if agent.api_mode == "anthropic_messages":
_build_anthropic_client_from_runtime(agent, rt) _build_anthropic_client_from_runtime(agent, rt)
elif (agent.provider or "").strip().lower() == "moa": elif (agent.provider or "").strip().lower() == "moa":
# MoA is a virtual provider with empty client_kwargs — rebuilding via _create_openai_client
# would raise "api_key client option must be set". Recreate the facade through the shared
# factory so the reference_callback relay survives recovery (#53802).
from agent.moa_loop import build_moa_facade from agent.moa_loop import build_moa_facade
agent.client = build_moa_facade(agent, agent.model) agent.client = build_moa_facade(agent, agent.model)
else: else:
@@ -1021,6 +1088,8 @@ def _rebind_primary_credential_pool(agent, primary_provider, matches_primary, lo
return return
if matches_primary(entry): if matches_primary(entry):
# _swap_credential rebuilds the client and reapplies base-url-scoped headers. # _swap_credential rebuilds the client and reapplies base-url-scoped headers.
# ``_swap_credential`` rebuilds the OpenAI/Anthropic client, reapplies base-url-scoped headers, and
# carries the accumulated base_url / OAuth-detection fixes (#33163).
agent._swap_credential(entry) agent._swap_credential(entry)
logger.info( logger.info(
"Restore re-selected pool entry %s (%s)", "Restore re-selected pool entry %s (%s)",
@@ -1043,6 +1112,11 @@ def restore_primary_runtime(agent) -> bool:
# _fallback_index past the chain end and silently block future fallbacks. # _fallback_index past the chain end and silently block future fallbacks.
agent._fallback_index = 0 agent._fallback_index = 0
return False return False
# Reset the chain index even when no fallback was activated this turn. Without this, a turn where
# _try_activate_fallback() was called but returned False (chain exhausted or provider not configured)
# leaves _fallback_index >= len(_fallback_chain) while _fallback_activated stays False. The next turn
# skips this block entirely, stranding the index and silently blocking all future fallback attempts for
# the session. Fixes #20465.
if getattr(agent, "_rate_limited_until", 0) > time.monotonic(): if getattr(agent, "_rate_limited_until", 0) > time.monotonic():
return False # primary still in rate-limit cooldown, stay on fallback return False # primary still in rate-limit cooldown, stay on fallback
rt = agent._primary_runtime rt = agent._primary_runtime
@@ -1150,6 +1224,7 @@ def extract_reasoning(agent, assistant_message) -> Optional[str]:
if not parts and isinstance(content, list): if not parts and isinstance(content, list):
# DeepSeek V4 Pro returns typed content blocks ({"type": "thinking", ...}); dropping them # DeepSeek V4 Pro returns typed content blocks ({"type": "thinking", ...}); dropping them
# makes the next turn fail with HTTP 400 "thinking must be passed back". # makes the next turn fail with HTTP 400 "thinking must be passed back".
# Refs #21944.
for block in content: for block in content:
if isinstance(block, dict) and block.get("type") == "thinking": if isinstance(block, dict) and block.get("type") == "thinking":
_add((block.get("thinking") or block.get("text") or "").strip()) _add((block.get("thinking") or block.get("text") or "").strip())
@@ -1253,7 +1328,12 @@ def _raw_cache_ttl_from_config(default: Any) -> Any:
def prompt_caching_disabled_from_config() -> bool: def prompt_caching_disabled_from_config() -> bool:
"""True when ``prompt_caching.cache_ttl`` is configured as off (same detection as ``agent_init``).""" """True when ``prompt_caching.cache_ttl`` is configured as off (same detection as ``agent_init``).
Same disable detection as ``agent_init`` (via ``cache_ttl_means_disabled``) so stub-based policy paths
(MoA slot decoration, auxiliary fallback replan) honor the same config contract without holding a live
``AIAgent`` (#76085 / #33555).
"""
return cache_ttl_means_disabled(_raw_cache_ttl_from_config("5m")) return cache_ttl_means_disabled(_raw_cache_ttl_from_config("5m"))
@@ -1281,11 +1361,18 @@ def plan_cache_sections_for_destination(
"""Plan request-local cache sections for one resolved destination (MoA / auxiliary senders): """Plan request-local cache sections for one resolved destination (MoA / auxiliary senders):
stripped copies (non-caching route) or a ``build_prompt_cache_plan`` layout; never mutates stripped copies (non-caching route) or a ``build_prompt_cache_plan`` layout; never mutates
inputs. ``cache_disabled``/``cache_ttl`` default to live config so the operator's disable and inputs. ``cache_disabled``/``cache_ttl`` default to live config so the operator's disable and
tier are honored; ``static_system_prefix`` gives the system prompt the main loop's early breakpoint.""" tier are honored; ``static_system_prefix`` gives the system prompt the main loop's early breakpoint.
``cache_disabled`` threads the operator's ``prompt_caching.cache_ttl`` disable into the blank policy
stub. When omitted, the live config is consulted so MoA/auxiliary paths cannot re-enable markers after
the user turned caching off (#76085).
"""
from agent.prompt_caching import ( from agent.prompt_caching import (
build_prompt_cache_plan, effective_cache_ttl, envelope_tool_part_cache_markers_supported, build_prompt_cache_plan, effective_cache_ttl, envelope_tool_part_cache_markers_supported,
strip_anthropic_cache_control, strip_anthropic_tool_cache_control, strip_anthropic_cache_control, strip_anthropic_tool_cache_control,
) )
# The policy function reads agent.* only as fallbacks for kwargs we don't pass; blank_cache_policy_stub
# is the only sanctioned stub so _cache_disabled cannot be left off again (#76085).
stub = blank_cache_policy_stub(cache_disabled) stub = blank_cache_policy_stub(cache_disabled)
dest = dict(provider=provider, base_url=base_url, api_mode=api_mode, model=model) dest = dict(provider=provider, base_url=base_url, api_mode=api_mode, model=model)
should_cache, native_layout = anthropic_prompt_cache_policy(stub, **dest) should_cache, native_layout = anthropic_prompt_cache_policy(stub, **dest)
@@ -1386,6 +1473,11 @@ def anthropic_prompt_cache_policy(
envelope (OpenRouter / OpenAI-wire proxies; Qwen/Alibaba too). The operator disable is read envelope (OpenRouter / OpenAI-wire proxies; Qwen/Alibaba too). The operator disable is read
from ``_cache_disabled`` (not ``_cache_ttl``, unset during init) so it survives switches from ``_cache_disabled`` (not ``_cache_ttl``, unset during init) so it survives switches
and restores. Branch ORDER is load-bearing (see inline notes). and restores. Branch ORDER is load-bearing (see inline notes).
Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct Alibaba (DashScope) also honour
Anthropic-style ``cache_control`` markers on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi
#3393 documented this for opencode-go Qwen. Without markers these providers serve zero cache hits,
re-billing the full prompt on every turn.
""" """
if getattr(agent, "_cache_disabled", False): if getattr(agent, "_cache_disabled", False):
return (False, False) return (False, False)
@@ -1403,6 +1495,10 @@ def anthropic_prompt_cache_policy(
is_claude = "claude" in model_lower is_claude = "claude" in model_lower
# Kimi/Moonshot via OpenRouter uses the same envelope cache_control as Claude; without this it # Kimi/Moonshot via OpenRouter uses the same envelope cache_control as Claude; without this it
# serves ~1% cache hits. Family matcher covers bare k1./k2. slugs. # serves ~1% cache hits. Family matcher covers bare k1./k2. slugs.
# Without this branch moonshotai/kimi-k2.6 falls through to (False, False), serving ~1% cache hits on
# 64K-token prompts and re-billing the full prompt on every turn. Observed within-turn progression with
# cache enabled: 1% → 67% → 84% → 97% (#25970). Reuses the canonical family matcher (covers bare
# k1./k2./k25 release slugs the substring check missed).
from agent.anthropic_adapter import _model_name_is_kimi_family from agent.anthropic_adapter import _model_name_is_kimi_family
is_kimi = _model_name_is_kimi_family(eff_model) or "moonshot" in model_lower is_kimi = _model_name_is_kimi_family(eff_model) or "moonshot" in model_lower
is_openrouter = base_url_host_matches(eff_base_url, "openrouter.ai") is_openrouter = base_url_host_matches(eff_base_url, "openrouter.ai")
@@ -1471,6 +1567,11 @@ def anthropic_prompt_cache_policy(
# Qwen/Alibaba on OpenCode and DashScope accept envelope cache_control on the OpenAI wire # Qwen/Alibaba on OpenCode and DashScope accept envelope cache_control on the OpenAI wire
# (pi-mono's "alibaba" cacheControlFormat). DeepSeek on OpenCode is excluded: its relay 400s on # (pi-mono's "alibaba" cacheControlFormat). DeepSeek on OpenCode is excluded: its relay 400s on
# block-array content. Family set/predicate shared with the effective_cache_ttl clamp. # block-array content. Family set/predicate shared with the effective_cache_ttl clamp.
# Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire transport that accepts
# Anthropic-style cache_control markers and rewards them with real cache hits. Without this branch
# qwen3.6-plus on opencode-go reports 0% cached tokens and burns through the subscription on every turn.
# OpenCode Zen's relay rejects the Anthropic-style content block format that cache markers produce
# (content becomes a block array instead of a plain string), causing HTTP 400 (#77217).
from agent.prompt_caching import ALIBABA_FAMILY_PROVIDERS, is_qwen_model from agent.prompt_caching import ALIBABA_FAMILY_PROVIDERS, is_qwen_model
if provider_lower in ALIBABA_FAMILY_PROVIDERS and is_qwen_model(model_lower): if provider_lower in ALIBABA_FAMILY_PROVIDERS and is_qwen_model(model_lower):
return True, False return True, False
@@ -1569,9 +1670,20 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo
from agent.ssl_verify import resolve_httpx_verify from agent.ssl_verify import resolve_httpx_verify
# Treat client_kwargs as read-only: callers pass agent._client_kwargs, and in-place mutation # Treat client_kwargs as read-only: callers pass agent._client_kwargs, and in-place mutation
# leaks into later requests (a torn-down httpx transport got reused). # leaks into later requests (a torn-down httpx transport got reused).
# Callers pass agent._client_kwargs (or shallow copies of it) in; any in-place mutation leaks back into
# the stored dict and is reused on subsequent requests. #10933 hit this by injecting an httpx.Client
# transport that was torn down after the first request, so the next request wrapped a closed transport
# and raised "Cannot send a request, as the client has been closed" on every retry. The revert resolved
# that specific path; this copy locks the contract so future transport/keepalive work can't reintroduce
# the same class of bug.
client_kwargs = dict(client_kwargs) client_kwargs = dict(client_kwargs)
# The MoA virtual provider has no OpenAI wire endpoint; the facade *is* the client. Rebuild the # The MoA virtual provider has no OpenAI wire endpoint; the facade *is* the client. Rebuild the
# facade, never a native client (TypeError; relay re-wire). # facade, never a native client (TypeError; relay re-wire).
# Rebuilding a native OpenAI client while agent.provider == "moa" (client replacement, stream-retry pool
# cleanup, credential rotation, fallback+restore) drops the facade: the next primary call either raises
# a `_moa_prepared_request` TypeError (#78382) or, when _client_kwargs carry an unrelated relay
# base_url, leaks the request to a foreign gateway. Rebuild the facade instead (build_moa_facade also
# re-wires the reference relay, see #53802).
if (getattr(agent, "provider", "") or "").strip().lower() == "moa": if (getattr(agent, "provider", "") or "").strip().lower() == "moa":
from agent.moa_loop import build_moa_facade from agent.moa_loop import build_moa_facade
return build_moa_facade(agent, getattr(agent, "model", None) or "default") return build_moa_facade(agent, getattr(agent, "model", None) or "default")
@@ -1604,12 +1716,26 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo
# behind a per-client view whose ``close()`` is a no-op for the pool, so a closed wrapper # behind a per-client view whose ``close()`` is a no-op for the pool, so a closed wrapper
# never takes a sibling's (or the successor's) connections with it # never takes a sibling's (or the successor's) connections with it
# (tests/agent/test_shared_http_transport.py). # (tests/agent/test_shared_http_transport.py).
# Without this, a peer that drops mid-stream leaves the socket in a state where epoll_wait never fires,
# ``httpx`` read timeout may not trigger, and the agent hangs until manually killed. Probes after 30s
# idle, retry every 10s, give up after 3 → dead peer detected within ~60s. Safety against #10933: the
# ``client_kwargs = dict(client_kwargs)`` above means this injection only lands in the local per-call
# copy, never back into ``agent._client_kwargs``. Each ``_create_openai_client`` invocation therefore
# gets its OWN fresh ``httpx.Client`` whose lifetime is tied to the OpenAI client it is passed to. When
# the OpenAI client is closed (rebuild, teardown, credential rotation), the paired ``httpx.Client``
# closes with it, and the next call constructs a fresh one — no stale closed transport can be reused.
if "http_client" not in client_kwargs: if "http_client" not in client_kwargs:
keepalive_http = agent._build_keepalive_http_client(client_kwargs.get("base_url", ""), verify=httpx_verify) keepalive_http = agent._build_keepalive_http_client(client_kwargs.get("base_url", ""), verify=httpx_verify)
if keepalive_http is not None: if keepalive_http is not None:
client_kwargs["http_client"] = keepalive_http client_kwargs["http_client"] = keepalive_http
# Retries belong to the outer conversation loop (honors Retry-After); SDK retries would # Retries belong to the outer conversation loop (honors Retry-After); SDK retries would
# double-retry inside it. auxiliary_client keeps SDK retries as it isn't wrapped. # double-retry inside it. auxiliary_client keeps SDK retries as it isn't wrapped.
# Delegate all rate-limit / 5xx retry to hermes's outer conversation loop, which honors Retry-After and
# applies adaptive/jittered backoff. The OpenAI SDK default (max_retries=2) uses its own 1-2s backoff
# that ignores Retry-After and double-retries inside our loop — the same deadlock the Anthropic clients
# hit (#26293). This is the single chokepoint every primary OpenAI/aggregator client passes through
# (init, switch_model, recovery, restore, request-scoped); auxiliary_client builds its own clients and
# keeps SDK retries because it is NOT wrapped by the conversation loop.
client_kwargs.setdefault("max_retries", 0) client_kwargs.setdefault("max_retries", 0)
_ensure_copilot_headers(client_kwargs) _ensure_copilot_headers(client_kwargs)
# OpenCode Free is served anonymously: any unrecognized bearer is a 401, so an empty # OpenCode Free is served anonymously: any unrecognized bearer is a 401, so an empty
@@ -1768,6 +1894,9 @@ def _build_switched_client(agent, new_provider, api_key, base_url, api_mode, new
) )
# Read live config, not agent._custom_providers, so mid-session ssl_ca_cert / ssl_verify # Read live config, not agent._custom_providers, so mid-session ssl_ca_cert / ssl_verify
# edits are honored. # edits are honored.
# Read custom_providers from live config (not the init-time snapshot on ``agent._custom_providers``)
# so ssl_ca_cert / ssl_verify edits are honored when switching mid-session, matching the
# context-length reload below (#15779).
apply_custom_provider_tls_to_client_kwargs( apply_custom_provider_tls_to_client_kwargs(
agent._client_kwargs, str(effective_base or ""), agent._client_kwargs, str(effective_base or ""),
get_compatible_custom_providers(load_config_readonly()), get_compatible_custom_providers(load_config_readonly()),
@@ -1906,6 +2035,7 @@ def _build_primary_runtime_snapshot(agent, api_mode) -> Dict[str, Any]:
"reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False), "reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False),
# Overrides must travel with the switched-to identity or a later recovery/restore resurrects # Overrides must travel with the switched-to identity or a later recovery/restore resurrects
# PRE-switch overrides from the stale init snapshot. # PRE-switch overrides from the stale init snapshot.
# See #75091.
"request_overrides": dict(getattr(agent, "request_overrides", {}) or {}), "request_overrides": dict(getattr(agent, "request_overrides", {}) or {}),
"runtime_capabilities": dict(getattr(agent, "runtime_capabilities", {}) or {}), "runtime_capabilities": dict(getattr(agent, "runtime_capabilities", {}) or {}),
"compressor_model": getattr(cc, "model", agent.model), "compressor_model": getattr(cc, "model", agent.model),
@@ -1973,6 +2103,12 @@ def switch_model(
snapshot and re-raises (callers catch).""" snapshot and re-raises (callers catch)."""
old_model = agent.model old_model = agent.model
old_provider = agent.provider old_provider = agent.provider
# ── Reload credential pool for the new provider (issue #52727) ── Without this,
# ``recover_with_credential_pool`` sees a ``pool.provider != agent.provider`` mismatch and
# short-circuits, leaving the new provider with no rotation/recovery on 401/429 and burning the original
# pool's entries. Only reload when the provider actually changed (or the pool was missing) —
# re-selecting the same provider must not churn the pool reference. A reload failure is logged +
# swallowed: the switch itself must still complete.
old_norm = (old_provider or "").strip().lower() old_norm = (old_provider or "").strip().lower()
new_norm = (new_provider or "").strip().lower() new_norm = (new_provider or "").strip().lower()
api_mode, base_url, destination_capabilities = _resolve_switch_destination( api_mode, base_url, destination_capabilities = _resolve_switch_destination(
@@ -2131,6 +2267,12 @@ def repair_tool_call(agent, tool_name: str) -> str | None:
# VolcEngine api/plan leaks XML attribute fragments into tool_use.name (`terminal" # VolcEngine api/plan leaks XML attribute fragments into tool_use.name (`terminal"
# parameter="command" ...`); trim at the first quote/angle bracket. Do NOT split on whitespace: # parameter="command" ...`); trim at the first quote/angle bracket. Do NOT split on whitespace:
# "write file" must reach ``_norm`` -> ``write_file``. # "write file" must reach ``_norm`` -> ``write_file``.
# `terminal" parameter="command" string="true` `execute_code" parameter="code" string="true`
# `session_search" parameter="session_id" string="true` We trim at the first unambiguous XML/quote
# character so the rest of the repair pipeline (lowercase / snake_case / fuzzy match) can resolve the
# cleaned name to a real tool. Crucially we DO NOT split on whitespace: legitimate inputs like "write
# file" must keep flowing through ``_norm`` -> ``write_file`` (covered by test_space_to_underscore in
# tests/run_agent/test_repair_tool_call_name.py). See #33007.
for _xml_sep in ('"', "'", "<", ">"): for _xml_sep in ('"', "'", "<", ">"):
_idx = tool_name.find(_xml_sep) _idx = tool_name.find(_xml_sep)
if _idx > 0: if _idx > 0:
@@ -2171,12 +2313,19 @@ _INTERRUPTED_PLACEHOLDER = "[response interrupted]"
# Escalate repeated heals once per session window, then stay quiet. Default threshold; tunable via # Escalate repeated heals once per session window, then stay quiet. Default threshold; tunable via
# ``agent.sanitizer_heal_escalation_threshold`` (<= 0 disables). # ``agent.sanitizer_heal_escalation_threshold`` (<= 0 disables).
# Repeated heals of the same poisoned transcript used to WARNING on every send (#96870).
# ``_EMPTY_HEAL_ESCALATE_AFTER`` is the built-in default; deployments tune it via
# ``agent.sanitizer_heal_escalation_threshold`` in config.yaml (<= 0 disables escalation entirely — WARNINGs
# still fire per window).
_EMPTY_HEAL_ESCALATE_AFTER = 3 _EMPTY_HEAL_ESCALATE_AFTER = 3
_EMPTY_HEAL_WINDOW_S = 600.0 _EMPTY_HEAL_WINDOW_S = 600.0
_empty_heal_log_state: Dict[str, Dict[str, Any]] = {} _empty_heal_log_state: Dict[str, Dict[str, Any]] = {}
_empty_heal_log_lock = threading.Lock() _empty_heal_log_lock = threading.Lock()
# Sessions already told ONCE (out-of-band, never in conversation context); kept apart from the # Sessions already told ONCE (out-of-band, never in conversation context); kept apart from the
# windowed log state so a new window never re-arms the notice. # windowed log state so a new window never re-arms the notice.
# Session keys that already received the one-time user notice. Separate from the windowed log state so a new
# 10-minute window never re-notifies: the user is told ONCE per session, ever (#96870 — out-of-band,
# delivery channel only, never injected into conversation context).
_empty_heal_user_notified: set = set() _empty_heal_user_notified: set = set()
# One-shot pending notices keyed by session, drained via ``consume_pending_sanitizer_heal_notice`` # One-shot pending notices keyed by session, drained via ``consume_pending_sanitizer_heal_notice``
# and delivered via the status/warning callback. # and delivered via the status/warning callback.
@@ -2374,12 +2523,31 @@ def _drop_invalid_roles(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
def _drop_empty_tool_calls_arrays(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: def _drop_empty_tool_calls_arrays(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""Strict providers 400 on ``tool_calls: []``; normalize on shallow copies so history stays byte-stable.""" """Strict providers 400 on ``tool_calls: []``; normalize on shallow copies so history stays byte-stable."""
# --- Drop empty / malformed tool_calls arrays on assistant messages --- An assistant message carrying
# ``tool_calls: []`` (an empty array) — or a non-list value under the key — is semantically identical to
# an assistant message with no tool calls, but strict OpenAI-compatible providers reject the empty array
# outright: DeepSeek v4 returns HTTP 400 "Invalid 'messages[N].tool_calls': empty array. Expected an
# array with minimum length 1, but got an empty array instead." (#58755, follow-up to #56980). Empty
# arrays reach here from session resume, host-fed histories, or the consecutive-assistant merge in
# ``repair_message_sequence`` (which preserves a pre-existing ``[]`` on the surviving turn). This is the
# final pre-API chokepoint, so normalize defensively — and, per the #56980 review, do it HERE on the
# per-call copy rather than in ``repair_message_sequence``, which would destructively rewrite the
# persisted trajectory. Shallow-copy the message before dropping the key so stored history (and prompt
# caching) stays byte-stable.
normalized: List[Dict[str, Any]] = [] normalized: List[Dict[str, Any]] = []
dropped = 0 dropped = 0
for msg in messages: for msg in messages:
if ( if (
isinstance(msg, dict) isinstance(msg, dict)
and msg.get("role") == "assistant" and msg.get("role") == "assistant"
# Defense-in-depth: a strict OpenAI-compatible provider (e.g. onerouter / Qwen, DeepSeek v4)
# rejects an assistant message carrying ``tool_calls: []`` (empty array) with HTTP 400 "Empty
# tool_calls is not supported in message." The pre-API sanitizer in agent_runtime_helpers drops
# these, but only on the conversation_loop path — other routes can reach the wire without it.
# For every request that serializes through this transport (conversation loop and any caller
# using it), this is the last boundary, so normalize here. Requests built by fully separate
# payload paths (e.g. some auxiliary clients) never pass through this layer and are out of scope
# for it. (#58755 follow-up)
and "tool_calls" in msg and "tool_calls" in msg
and not (isinstance(msg["tool_calls"], list) and msg["tool_calls"]) and not (isinstance(msg["tool_calls"], list) and msg["tool_calls"])
): ):
@@ -2444,6 +2612,22 @@ def _pair_tool_calls_positionally(messages: List[Dict[str, Any]]) -> List[Dict[s
"""Positional tool_call <-> tool_result pairing: strict providers (DeepSeek v4, Kimi) require """Positional tool_call <-> tool_result pairing: strict providers (DeepSeek v4, Kimi) require
results IMMEDIATELY after their call. Drops positional orphans, stubs unanswered declared results IMMEDIATELY after their call. Drops positional orphans, stubs unanswered declared
ids; matching is alias-aware.""" ids; matching is alias-aware."""
# --- Positional tool_call <-> tool_result pairing --- Strict OpenAI-compatible providers (DeepSeek v4,
# Kimi) enforce the POSITIONAL invariant: an assistant message carrying tool_calls must be IMMEDIATELY
# followed by tool messages covering every tool_call_id. The previous implementation compared global id
# sets, which misses the failure mode where a result exists somewhere in the transcript but not in the
# run right after its call — an interrupted turn or a compression window can displace a result past a
# user turn. The id then survives in the global result set, so the call looks answered, no stub is
# injected, and the provider rejects the payload with HTTP 400 "An assistant message with 'tool_calls'
# must be followed by tool messages responding to each 'tool_call_id' (insufficient tool messages
# following tool_calls message)". Rewritten as a single rolling walk on the per-call copy (#94704): (a)
# tool results that do not immediately follow an assistant message declaring their id are dropped
# (positional orphans — includes results appearing BEFORE their call, which strict providers also
# reject); (b) declared ids not covered by the immediately-following tool run get a stub result injected
# at the end of that run, even when a mispositioned result exists elsewhere. Matching is variant-aware
# (``tool_call_id_variants`` / ``tool_result_id_variants``): a result keyed on ANY alias spelling
# (``id`` / ``call_id`` / ``response_item_id`` / composite bridge) answers the call, preserving the
# unified alias policy from #55626/#63000/#93251.
paired: List[Dict[str, Any]] = [] paired: List[Dict[str, Any]] = []
declared_calls: Dict[str, tuple] = {} declared_calls: Dict[str, tuple] = {}
dropped = 0 dropped = 0
@@ -2504,6 +2688,24 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
(not ids ever seen) because llama.cpp reuses one constant id, and whole variant groups so (not ids ever seen) because llama.cpp reuses one constant id, and whole variant groups so
alias-keyed results are not deleted.""" alias-keyed results are not deleted."""
outstanding: Dict[str, int] = {} # every alias of an unanswered call -> its group id outstanding: Dict[str, int] = {} # every alias of an unanswered call -> its group id
# 3. Deduplicate tool_call_ids. Strict providers (DeepSeek) reject a payload where the same tool_call_id
# appears more than once with HTTP 400 "Duplicate value for 'tool_call_id'" (#58327). Duplicates can
# arise from retries, crash/resume glitches, or a compression window that re-emits a tool result. This
# is the final pre-API chokepoint, so dedup defensively here even though repair_message_sequence also
# consumes matched ids. (a) collapse duplicate tool_calls WITHIN an assistant message (b) drop tool
# results that answer no OUTSTANDING tool call (b) tracks outstanding calls rather than every id ever
# seen, because ``tool_call_id`` is NOT globally unique in practice: llama.cpp emits a single constant
# id for every tool call it ever returns (verified: three separate completions from one server all
# carry the same id). A seen-once-drop-forever rule reads the SECOND legitimate tool result of such a
# session as a duplicate and deletes it, so from the second tool call onward the model never sees any
# result — it announces its next action and the turn dies with the work unfinished. Outstanding-call
# semantics keep both protections intact: a re-emitted result still answers no pending call and is
# still dropped, while a genuine new call that reuses the id re-arms that id first. Variant-group
# tracking: answering or deduping one spelling consumes its siblings too. A Codex/Responses tool_call
# registers ``id`` (fc_...), ``call_id`` (call_...), ``response_item_id``, and composite spellings
# (#55626/#58168/#63000); tracking only the coalesced id here made a result keyed on any OTHER variant
# look like it answered no outstanding call, so this pass deleted the very result step 2's
# variant-aware matching had just preserved (issue #93251 — whole parallel batches vanished).
outstanding_groups: Dict[int, frozenset] = {} outstanding_groups: Dict[int, frozenset] = {}
next_group_id = 0 next_group_id = 0
deduped: List[Dict[str, Any]] = [] deduped: List[Dict[str, Any]] = []
@@ -2536,6 +2738,9 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
continue continue
if candidate_groups: if candidate_groups:
# Consume EVERY variant of the matched call; ids are re-armed by the next call reusing them. # Consume EVERY variant of the matched call; ids are re-armed by the next call reusing them.
# Consume the whole alias group so a SECOND result replaying any sibling spelling falls into
# the drop branch below — strict providers reject duplicate tool_call_ids with HTTP 400
# (#58327, #66974). Credit: #55436.
group_id = min(candidate_groups) group_id = min(candidate_groups)
for variant in outstanding_groups.pop(group_id, frozenset()): for variant in outstanding_groups.pop(group_id, frozenset()):
if outstanding.get(variant) == group_id: if outstanding.get(variant) == group_id:
@@ -2552,6 +2757,22 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
def _realign_tool_result_names(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: def _realign_tool_result_names(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""Align each tool result's wire ``name`` with its call's function name (per-call copy only): """Align each tool result's wire ``name`` with its call's function name (per-call copy only):
Google 400s on a mismatch, routine when tool_search bridges via ``tool_call``.""" Google 400s on a mismatch, routine when tool_search bridges via ``tool_call``."""
# 4. Google matches functionResponse.name against functionCall.name and rejects a mismatch with HTTP 400
# "Request contains an invalid argument" (INVALID_ARGUMENT); behind an OpenAI-compatible gateway that
# surfaces only as a generic "Provider returned error". When tool_search defers MCP/plugin tools the
# model calls the bridge tool ``tool_call``, while ``make_tool_result_message()`` labels the result
# with the unwrapped internal tool name (``mcp__github__create_issue``) that dispatch, hooks, logging,
# and guardrails need. #72089 fixed exactly this for the native Gemini adapter, which now prefers
# ``tool_name_by_call_id`` over the result name; requests that reach Gemini through the
# OpenAI-compatible path (OpenRouter, Vertex/LiteLLM proxies, any OpenAI-shaped gateway) skip that
# translation entirely and still send the internal name on the wire. Normalizing here rather than in
# the OpenAI-compat serializer keeps it provider-agnostic: Gemini reaches Hermes under many model
# strings and base URLs, so sniffing for "is this really Google?" is unreliable, and every other
# provider either ignores the field or agrees with the call name. Runs on the per-call copy, so the
# stored trajectory keeps the real tool name for the session DB and the UI — only the wire payload
# changes. A no-op for the native Gemini path, which already resolves the same name. A result whose
# assistant call frame is missing entirely never reaches here — pass 1 above drops it as an orphan —
# so the only results this pass sees are ones whose call name is knowable.
call_names: Dict[str, str] = {} call_names: Dict[str, str] = {}
for msg in messages: for msg in messages:
if msg.get("role") == "assistant": if msg.get("role") == "assistant":
@@ -2688,7 +2909,14 @@ def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int:
"""Re-pad or strip assistant turns' reasoning_content for the CURRENT provider after a """Re-pad or strip assistant turns' reasoning_content for the CURRENT provider after a
fallback switch: ``api_messages`` is shaped for the primary; require-side providers fallback switch: ``api_messages`` is shaped for the primary; require-side providers
(DeepSeek/Kimi/MiMo) 400 without the pad, strict ones (Mistral, Cerebras, Groq) 400/422 (DeepSeek/Kimi/MiMo) 400 without the pad, strict ones (Mistral, Cerebras, Groq) 400/422
with it. Idempotent; returns the number of assistant turns changed.""" with it. Idempotent; returns the number of assistant turns changed.
* Switching TO a strict provider that rejects the field (Mistral, Cerebras, Groq, SambaNova, …):
assistant turns built under a reasoning primary carry a ``reasoning_content`` pad (often a single space
``" "``), and the strict provider rejects it with HTTP 400/422 ("Extra inputs are not permitted"). This
is the exact cross-provider fallback bug from #45655 — a DeepSeek primary pads history with ``" "``, the
request falls back to Mistral, and Mistral 422s on the stale pad.
"""
from agent.message_sanitization import reapply_reasoning_echo from agent.message_sanitization import reapply_reasoning_echo
return reapply_reasoning_echo(api_messages, agent._needs_thinking_reasoning_pad()) return reapply_reasoning_echo(api_messages, agent._needs_thinking_reasoning_pad())
@@ -2701,7 +2929,11 @@ def _iter_httpx_pools_with_owner(http_client: Any):
``owner`` is ``None`` for a pool this client owns outright, or the ``_SharedTransport`` view ``owner`` is ``None`` for a pool this client owns outright, or the ``_SharedTransport`` view
id when the pool is process-shared with other clients id when the pool is process-shared with other clients
(``process_bootstrap.build_keepalive_http_client``). Callers must then touch only the (``process_bootstrap.build_keepalive_http_client``). Callers must then touch only the
in-flight requests stamped with that owner.""" in-flight requests stamped with that owner.
Walking the default transport alone makes ``force_close_tcp_sockets`` return 0 while a stream is still
mid-recv — the interrupt logs success and the provider keeps burning the slot (#72975).
"""
seen_pools: set[int] = set() seen_pools: set[int] = set()
try: try:
transports = [getattr(http_client, "_transport", None)] transports = [getattr(http_client, "_transport", None)]
+14
View File
@@ -568,6 +568,17 @@ def build_anthropic_kwargs(
kwargs["tool_choice"] = _TOOL_CHOICE_MAP.get(tool_choice) or { kwargs["tool_choice"] = _TOOL_CHOICE_MAP.get(tool_choice) or {
"type": "tool", "name": to_wire(tool_choice) if to_wire else tool_choice "type": "tool", "name": to_wire(tool_choice) if to_wire else tool_choice
} }
# Map reasoning_config to Anthropic's thinking parameter. Claude 4.6+ models use adaptive thinking +
# output_config.effort. Older models use manual thinking with budget_tokens. MiniMax Anthropic-compat
# endpoints support thinking (manual mode only, not adaptive). Haiku does NOT support extended thinking
# — skip entirely. Kimi / Moonshot models also use adaptive thinking: their Anthropic-compatible
# endpoints (api.moonshot.cn/anthropic, api.kimi.com/coding) accept ``thinking.type="adaptive"`` +
# ``output_config.effort``, and the replay-validation 400s that originally motivated dropping the
# parameter (#13848) no longer occur. (Kimi on chat_completions enables thinking via extra_body in the
# ChatCompletionsTransport — see #13503.) On 4.7+ the `thinking.display` field defaults to "omitted",
# which silently hides reasoning text that Hermes surfaces in its CLI. We request "summarized" so the
# reasoning blocks stay populated — matching 4.6 behavior and preserving the activity-feed UX during
# long tool runs.
if reasoning_config and isinstance(reasoning_config, dict): if reasoning_config and isinstance(reasoning_config, dict):
kwargs.update(_thinking_kwargs(reasoning_config, model, effective_max_tokens)) kwargs.update(_thinking_kwargs(reasoning_config, model, effective_max_tokens))
# Safety net so upstream 4.6 -> 4.7 migrations don't need coordinated edits everywhere callers # Safety net so upstream 4.6 -> 4.7 migrations don't need coordinated edits everywhere callers
@@ -638,6 +649,9 @@ def _stream_final_message(stream_fn, api_kwargs, log_prefix, on_stream_event, on
try: try:
on_stream_event(event) on_stream_event(event)
except TimeoutError: except TimeoutError:
# The callback is the caller's deadline seam (#99692: the host waiting on this summary has
# already given up). Abandon the stream — the ``with`` closes it — instead of streaming an
# answer nobody will read.
raise raise
except Exception: except Exception:
logger.debug("%son_stream_event callback failed", log_prefix, exc_info=True) logger.debug("%son_stream_event callback failed", log_prefix, exc_info=True)
+6 -1
View File
@@ -76,7 +76,12 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool:
"""DeepSeek's ``/anthropic`` route. In thinking mode DeepSeek requires prior-turn ``thinking`` """DeepSeek's ``/anthropic`` route. In thinking mode DeepSeek requires prior-turn ``thinking``
blocks to round-trip while the generic third-party path strips them; its blocks are unsigned, blocks to round-trip while the generic third-party path strips them; its blocks are unsigned,
so it gets the same strip-signed / keep-unsigned policy as Kimi. Pinned to the ``/anthropic`` so it gets the same strip-signed / keep-unsigned policy as Kimi. Pinned to the ``/anthropic``
path so the OpenAI-compatible base URL is not misclassified.""" path so the OpenAI-compatible base URL is not misclassified.
Per DeepSeek's published compatibility matrix the blocks are unsigned (no Anthropic-proprietary
signature, no ``redacted_thinking`` support), so this endpoint is handled with the same strip-signed /
keep-unsigned policy used for Kimi's ``/coding`` endpoint. See hermes-agent#16748.
"""
return base_url_host_matches(base_url or "", "api.deepseek.com") and "/anthropic" in _normalized_lower(base_url) return base_url_host_matches(base_url or "", "api.deepseek.com") and "/anthropic" in _normalized_lower(base_url)
+18 -1
View File
@@ -109,6 +109,7 @@ def normalize_model_name(model: str, preserve_dots: bool = False) -> str:
if model.lower().startswith("anthropic/"): if model.lower().startswith("anthropic/"):
model = model[len("anthropic/"):] model = model[len("anthropic/"):]
if not preserve_dots and not _is_bedrock_model_id(model) and model.lower().startswith(("claude-", "anthropic/")): if not preserve_dots and not _is_bedrock_model_id(model) and model.lower().startswith(("claude-", "anthropic/")):
# Only convert dots to hyphens for Anthropic/Claude models. See issue #17171.
model = model.replace(".", "-") model = model.replace(".", "-")
return model return model
@@ -150,6 +151,8 @@ def convert_tools_to_anthropic(tools: List[Dict]) -> List[Dict]:
for t in tools or []: for t in tools or []:
fn = t.get("function", {}) fn = t.get("function", {})
name = fn.get("name", "") name = fn.get("name", "")
# Defensive dedup: Anthropic rejects requests with duplicate tool names. Upstream injection paths
# already dedup, but this guard converts a hard API failure into a warning. See: #18478
if name and name in seen_names: if name and name in seen_names:
logger.warning("convert_tools_to_anthropic: duplicate tool name '%s' — dropping second occurrence", name) logger.warning("convert_tools_to_anthropic: duplicate tool name '%s' — dropping second occurrence", name)
continue continue
@@ -370,6 +373,12 @@ def _replay_ordered_blocks(m: Dict[str, Any], ordered_blocks: List[Any]) -> Opti
def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
"""Assistant message -> Anthropic content blocks (thinking, text, tool_use, Kimi/DeepSeek """Assistant message -> Anthropic content blocks (thinking, text, tool_use, Kimi/DeepSeek
reasoning_content injection).""" reasoning_content injection)."""
# apply_anthropic_cache_control marks an assistant turn with non-empty text by writing cache_control
# INTO ``content`` (see _apply_cache_marker's list branch), not at the top level. This branch rebuilds
# the message from ordered_blocks and never reads ``content``, so that marker would be dropped -- and
# because _can_carry_marker already counted this message as a carrier, the breakpoint is burned rather
# than relocated. #56195 covered the complementary shape (blank content -> top-level marker); this is
# the interleaved thinking + preamble-text + tool_use shape.
content = m.get("content", "") content = m.get("content", "")
ordered_blocks = m.get("anthropic_content_blocks") ordered_blocks = m.get("anthropic_content_blocks")
if isinstance(ordered_blocks, list) and ordered_blocks: if isinstance(ordered_blocks, list) and ordered_blocks:
@@ -395,6 +404,8 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
# (injected as a fallback upstream). Prepend, since thinking must precede text/tool_use. Skip # (injected as a fallback upstream). Prepend, since thinking must precede text/tool_use. Skip
# when reasoning_details already supplied (signed) thinking blocks: a duplicate unsigned one # when reasoning_details already supplied (signed) thinking blocks: a duplicate unsigned one
# would be downgraded to a spurious text block on the last assistant message. # would be downgraded to a spurious text block on the last assistant message.
# See hermes-agent#13848. Accept empty string "" — _copy_reasoning_content_for_api() injects "" as a
# tier-3 fallback for Kimi tool-call messages that had no reasoning.
reasoning_content = m.get("reasoning_content") reasoning_content = m.get("reasoning_content")
if isinstance(reasoning_content, str) and not _has_block_type(blocks, _THINKING_TYPES): if isinstance(reasoning_content, str) and not _has_block_type(blocks, _THINKING_TYPES):
blocks.insert(0, {"type": "thinking", "thinking": reasoning_content}) blocks.insert(0, {"type": "thinking", "thinking": reasoning_content})
@@ -598,7 +609,13 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
"""Anthropic requires messages[0].role == user; prepend a placeholder turn otherwise. A second """Anthropic requires messages[0].role == user; prepend a placeholder turn otherwise. A second
auto-compaction can leave a role=assistant summary first, which the API rejects (often masked auto-compaction can leave a role=assistant summary first, which the API rejects (often masked
as a misleading tool_use/tool_result 400). The filler must be non-whitespace text or it trades as a misleading tool_use/tool_result 400). The filler must be non-whitespace text or it trades
that 400 for the blank-block one.""" that 400 for the blank-block one.
The inserted text block must be non-whitespace: Anthropic separately rejects any text content block
whose text is empty or whitespace-only ("text content blocks must contain non-whitespace text"), so a
single space here traded the "leading assistant turn" 400 for that one (#69512 class). Uses the same
placeholder as every other synthesized filler block in this module for consistency.
"""
if result and result[0].get("role") != "user": if result and result[0].get("role") != "user":
result.insert(0, {"role": "user", "content": [_text_block(_EMPTY_TEXT_PLACEHOLDER)]}) result.insert(0, {"role": "user", "content": [_text_block(_EMPTY_TEXT_PLACEHOLDER)]})
+12
View File
@@ -57,6 +57,15 @@ class ApiErrorSummaryMixin:
the pool. xAI returns the same permission-denied text for BOTH cases; a ``[WKE=unauthenticated:...]`` the pool. xAI returns the same permission-denied text for BOTH cases; a ``[WKE=unauthenticated:...]``
suffix (or "access token could not be validated") means stale token → return False so the refresh path suffix (or "access token could not be validated") means stale token → return False so the refresh path
runs. runs.
Disambiguator for xAI (#29344): the same ``code`` text ("The caller does not have permission to
execute the specified operation") is returned for BOTH an unsubscribed account AND a stale OAuth
access token. xAI ships an explicit signal in the ``error`` field that tells the two apart: a
``[WKE=unauthenticated:...]`` suffix (and/or the ``OAuth2 access token could not be validated``
phrasing) means the credentials failed validation — that's recoverable by refreshing the token, NOT
by surfacing an entitlement message. When either signal is present we return False eagerly so the
credential-pool refresh path runs, letting long-running TUI sessions recover from stale tokens
without an exit/reopen cycle.
""" """
if status_code not in {401, 403, None}: if status_code not in {401, 403, None}:
return False return False
@@ -162,6 +171,9 @@ class ApiErrorSummaryMixin:
# SDK may leave body empty while httpx has the payload. Redact: the body is attacker-influenced # SDK may leave body empty while httpx has the payload. Redact: the body is attacker-influenced
# and may echo Authorization / x-api-key / request JSON. # and may echo Authorization / x-api-key / request JSON.
# Redact before returning: the raw provider/proxy error body is attacker-influenced and may echo
# Authorization / x-api-key / request JSON, which would otherwise leak into final_response + logs
# (this path widens exposure vs the old empty-body "HTTP 400" string). See #36109.
response = getattr(error, "response", None) response = getattr(error, "response", None)
if response is not None: if response is not None:
try: try:
+214 -3
View File
@@ -177,6 +177,12 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any:
_apply_required_codex_headers(kwargs, access_token=api_key, base_url=base_url) _apply_required_codex_headers(kwargs, access_token=api_key, base_url=base_url)
# Hermes owns aux retry/fallback policy; the SDK default (max_retries=2) would triple # Hermes owns aux retry/fallback policy; the SDK default (max_retries=2) would triple
# wall time on a hung endpoint before Hermes sees one failure. # wall time on a hung endpoint before Hermes sees one failure.
# Hermes owns auxiliary retry + provider/model fallback policy (the same-provider transient retry in
# call_llm plus the except-chain fallback). The OpenAI SDK's own default (max_retries=2 → up to 3
# attempts) silently multiplies the effective wall time of every aux call by 3× on a slow/hung endpoint,
# so a 120s timeout can stall ~360s before Hermes sees a single failure (issue #54465). Disable
# SDK-internal retries by default and let Hermes control the budget; explicit callers can still override
# via kwargs.
kwargs.setdefault("max_retries", 0) kwargs.setdefault("max_retries", 0)
return OpenAI(api_key=api_key, base_url=base_url, **kwargs) return OpenAI(api_key=api_key, base_url=base_url, **kwargs)
@@ -184,6 +190,13 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any:
# Interrupt protection for atomic aux tasks: a compression summary killed by an ordinary # Interrupt protection for atomic aux tasks: a compression summary killed by an ordinary
# gateway interrupt degrades to a static marker, so a thread-local flag marks such calls # gateway interrupt degrades to a static marker, so a thread-local flag marks such calls
# protected. Explicit host cancel (Ctrl+C, /stop) still overrides it, timeouts still fire. # protected. Explicit host cancel (Ctrl+C, /stop) still overrides it, timeouts still fire.
# ── Interrupt protection for atomic auxiliary tasks ────────────────────── Some auxiliary tasks must NOT be
# aborted mid-flight by a gateway interrupt (e.g. an incoming user message while the agent is busy). Context
# compression is the prime case: if the summary LLM call is interrupted part-way, compression falls back to
# a static "summary unavailable" marker and the real handoff is lost (#23975). A thread-local flag lets such
# a task mark its in-flight LLM call as interrupt-protected; the Codex Responses stream's cancellation check
# honors it. TIMEOUTS still fire (a hung call must die), and all OTHER aux tasks (vision, web_extract,
# title_generation, …) remain freely interruptible.
_aux_interrupt_protection = threading.local() _aux_interrupt_protection = threading.local()
@@ -279,6 +292,12 @@ _aux_provider_response = threading.local()
# Absolute monotonic deadline of the waiting HOST. The stream's own ceiling # Absolute monotonic deadline of the waiting HOST. The stream's own ceiling
# (_aux_stream_total_ceiling, >= the host's and started later) would otherwise leave an # (_aux_stream_total_ceiling, >= the host's and started later) would otherwise leave an
# orphaned stream still billing after every host-ceiling timeout. # orphaned stream still billing after every host-ceiling timeout.
# Absolute wall-clock deadline (time.monotonic) of the HOST waiting for this auxiliary call, when it has one
# (#99692). Liveness alone is not enough: a host also stops waiting at its own total ceiling, and the
# streamed consumer below bounds itself only by _aux_stream_total_ceiling() — a budget derived from the aux
# request timeout, which is >= the host ceiling for every configured value AND starts counting later. So the
# stream that outlives its abandoned host is not an edge case; it is the guaranteed outcome of every
# total-ceiling timeout.
_aux_stream_deadline = threading.local() _aux_stream_deadline = threading.local()
@@ -411,6 +430,12 @@ def aux_stream_deadline(deadline: Optional[float]):
``None`` is a passthrough; re-entrant-safe. Host->worker return leg of the progress hook: ``None`` is a passthrough; re-entrant-safe. Host->worker return leg of the progress hook:
without it the isolated provider daemon streams to its own ceiling after the host stopped without it the isolated provider daemon streams to its own ceiling after the host stopped
waiting, billing a summary the commit fence refuses. waiting, billing a summary the commit fence refuses.
``8207862212`` releases the compression OWNER when the fence is cancelled, but the isolated provider
daemon (:func:`_run_protected_sync_provider_call`) that holds the socket keeps streaming to its own
``_aux_stream_total_ceiling`` budget — >= the host's ceiling by construction — billing an abandoned
summary the commit fence is already guaranteed to refuse, and stacking one fresh orphan per turn on a
session that compression never managed to shrink. See #99692.
""" """
previous = getattr(_aux_stream_deadline, "value", None) previous = getattr(_aux_stream_deadline, "value", None)
_aux_stream_deadline.value = deadline if isinstance(deadline, (int, float)) else previous _aux_stream_deadline.value = deadline if isinstance(deadline, (int, float)) else previous
@@ -446,6 +471,9 @@ def _run_protected_sync_provider_call(callback: Callable[[dict[str, Any]], Any],
dispatch_hook = getattr(_aux_dispatch, "hook", None) dispatch_hook = getattr(_aux_dispatch, "hook", None)
provider_response_hook = getattr(_aux_provider_response, "hook", None) provider_response_hook = getattr(_aux_provider_response, "hook", None)
host_deadline = _current_aux_stream_deadline() host_deadline = _current_aux_stream_deadline()
# #99692: the stream is consumed on the daemon below, and thread-locals do not cross that boundary — an
# owner-thread-only deadline would leave the fix inert on exactly the path large-session compression
# takes (protected call + hard-cancel source installed).
provider_context = contextvars.copy_context() provider_context = contextvars.copy_context()
done = threading.Event() done = threading.Event()
outcome: dict[str, Any] = {} outcome: dict[str, Any] = {}
@@ -1104,6 +1132,18 @@ class _CodexStreamGuard:
self.total_timeout = total_timeout self.total_timeout = total_timeout
self._start = time.monotonic() self._start = time.monotonic()
self.no_progress_timeout = _AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS self.no_progress_timeout = _AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS
# Progress-aware stream deadlines (supersedes the old single absolute kill at ``total_timeout``).
# Three regimes: 1. First token: the stream must produce its first substantive payload within
# ``no_progress_timeout`` (60s default) or we fail fast and let the caller's normal retry/fallback
# chain run — a dead (or keepalive-only zombie) Codex stream no longer holds the full 300s
# compression budget before falling back (masoria report, Aug 2026: 3 stacked 300s waits -> 20+ min
# stuck on "Summarizing"). 2. Streaming: every substantive event re-arms the deadline by
# ``no_progress_timeout`` — a live stream is never killed by an absolute total, so a long reasoning
# summary that is actually producing tokens completes instead of timing out at 300s and falling back
# (#54915's original complaint, fixed properly). Keepalive/lifecycle frames do NOT re-arm, mirroring
# the commit-fence progress gating (#96707). 3. Hard ceiling: an absolute backstop from
# ``_aux_stream_total_ceiling`` (max(600s, 4x configured timeout) — the same bound the streamed
# chat.completions path uses) so a pathological one-token-per-59s drip still terminates.
if total_timeout is not None: if total_timeout is not None:
self.no_progress_timeout = min(self.no_progress_timeout, float(total_timeout)) self.no_progress_timeout = min(self.no_progress_timeout, float(total_timeout))
self.hard_deadline = self._start + _aux_stream_total_ceiling(total_timeout) self.hard_deadline = self._start + _aux_stream_total_ceiling(total_timeout)
@@ -1193,6 +1233,9 @@ class _CodexStreamGuard:
# FD-safe — ``close()`` releases the raw TLS fd while the owner's OpenSSL BIO still # FD-safe — ``close()`` releases the raw TLS fd while the owner's OpenSSL BIO still
# caches it, the kernel recycles it (e.g. into a SQLite handle), and the owner's TLS # caches it, the kernel recycles it (e.g. into a SQLite handle), and the owner's TLS
# flush corrupts that file. The owner does the real close in its ``finally``. # flush corrupts that file. The owner does the real close in its ``finally``.
# This callback has two callers — ``_check_cancelled`` on the owning thread, and the daemon watchdog
# ``threading.Timer``, which is a stranger thread. The owning thread performs the real close in the
# ``finally`` below, which is where the FD release belongs. See #70773.
self.timeout_release_pending.set() self.timeout_release_pending.set()
if threading.get_ident() == self._owner_tid: if threading.get_ident() == self._owner_tid:
_close_quietly(self._client, "client close during timeout failed") _close_quietly(self._client, "client close during timeout failed")
@@ -1212,6 +1255,9 @@ class _CodexStreamGuard:
# The aux client cache wraps this same client; drop the entry so the next aux call # The aux client cache wraps this same client; drop the entry so the next aux call
# doesn't reuse the dead transport and fail fast. # doesn't reuse the dead transport and fail fast.
try: try:
# After we close the httpx transport above, the cache must drop that entry — otherwise the next
# auxiliary call (compression retry, memory flush, etc.) reuses the dead client and fails fast
# with a connection error. See issue #23432.
_evict_cached_client_instance(self._client) _evict_cached_client_instance(self._client)
except Exception: except Exception:
logger.debug("Codex auxiliary: cache eviction on timeout failed", exc_info=True) logger.debug("Codex auxiliary: cache eviction on timeout failed", exc_info=True)
@@ -1227,6 +1273,8 @@ class _CodexStreamGuard:
# interrupt (degraded fallback marker); explicit host cancel has its own exception. # interrupt (degraded fallback marker); explicit host cancel has its own exception.
if _aux_interrupt_cancel_requested(): if _aux_interrupt_cancel_requested():
raise AuxiliaryExplicitCancellation() raise AuxiliaryExplicitCancellation()
# Explicit host cancellation has its own frozen exception; timeouts above still fire and other
# aux tasks remain interruptible. See #23975.
if is_interrupted() and not _aux_interrupt_protected(): if is_interrupted() and not _aux_interrupt_protected():
raise InterruptedError("Codex auxiliary Responses stream interrupted") raise InterruptedError("Codex auxiliary Responses stream interrupted")
except InterruptedError: except InterruptedError:
@@ -1260,6 +1308,8 @@ class _CodexStreamGuard:
# TTFP telemetry records every frame, but forward progress (compression commit fence, # TTFP telemetry records every frame, but forward progress (compression commit fence,
# no-progress window) counts only substantive payloads — keepalives must not re-arm, # no-progress window) counts only substantive payloads — keepalives must not re-arm,
# so a zombie stream dies at the same window as a dead connection. # so a zombie stream dies at the same window as a dead connection.
# #93650: keep bulk wire-format payload out of the SDK's GIL-holding request transform on auxiliary
# calls too.
if _codex_event_has_content(_event): if _codex_event_has_content(_event):
self.record_progress() self.record_progress()
self.saw_content.set() self.saw_content.set()
@@ -1289,6 +1339,15 @@ class _CodexCompletionsAdapter:
def _build_responses_kwargs(self, kwargs: Dict[str, Any]) -> Tuple[Dict[str, Any], str, Any]: def _build_responses_kwargs(self, kwargs: Dict[str, Any]) -> Tuple[Dict[str, Any], str, Any]:
"""chat.completions kwargs → Responses API kwargs, ``(resp_kwargs, model, timeout)``; mirrors codex.py::build_kwargs.""" """chat.completions kwargs → Responses API kwargs, ``(resp_kwargs, model, timeout)``; mirrors codex.py::build_kwargs."""
from utils import base_url_host_matches from utils import base_url_host_matches
# Separate system/instructions from replayable conversation messages, then route the rest through
# the SINGLE shared chat->Responses converter used by the main agent transport
# (agent/transports/codex.py). Maintaining a private conversion loop here let chat-style messages
# with role="tool" leak straight into Responses input[] — which the Responses API rejects with
# "Invalid value: 'tool'. Supported values are: 'assistant', 'system', 'developer', and 'user'."
# (issue #5709, hit hard by flush_memories() / compression replaying real session history that
# includes assistant tool_calls + role="tool" results). The shared converter encodes assistant tool
# calls as `function_call` items and tool results as `function_call_output` items with a valid
# call_id, so every Responses path normalizes tool history identically and cannot drift.
from agent.codex_responses_adapter import _chat_messages_to_responses_input from agent.codex_responses_adapter import _chat_messages_to_responses_input
model = kwargs.get("model", self._model) model = kwargs.get("model", self._model)
host = str(getattr(self._client, "base_url", "") or "") host = str(getattr(self._client, "base_url", "") or "")
@@ -1309,6 +1368,9 @@ class _CodexCompletionsAdapter:
# Copilot binds replayed codex_message_items ids to a backend connection that doesn't # Copilot binds replayed codex_message_items ids to a backend connection that doesn't
# survive credential rotation (401 on replay) — same guard as build_kwargs. Aux calls # survive credential rotation (401 on replay) — same guard as build_kwargs. Aux calls
# never send ``context_management`` (main-turn feature): no compaction checkpoint. # never send ``context_management`` (main-turn feature): no compaction checkpoint.
# Auxiliary calls (context compression, flush_memories, MoA aggregation) go through this adapter
# instead of agent/transports/codex.py's build_kwargs, so they need the same guard applied
# independently. See #32716.
input_items = _chat_messages_to_responses_input( input_items = _chat_messages_to_responses_input(
replay_messages, is_github_responses=is_copilot, native_compaction_eligible=False replay_messages, is_github_responses=is_copilot, native_compaction_eligible=False
) )
@@ -1374,6 +1436,9 @@ class _CodexCompletionsAdapter:
# conversation (rotation-stable logical scope, else the physical session id). Skip the # conversation (rotation-stable logical scope, else the physical session id). Skip the
# key where the main transport does: xAI takes it in extra_body, GitHub opts out. # key where the main transport does: xAI takes it in extra_body, GitHub opts out.
try: try:
# Reuse the Responses transport's single authoritative hash algorithm and session-scope
# normalization so equivalent static prefixes route to the same cache bucket across modes,
# without concentrating unrelated sessions into one shared bucket (see #78941).
from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key
from agent.transports.codex import _default_prompt_cache_retention_for_request from agent.transports.codex import _default_prompt_cache_retention_for_request
if not (is_xai or is_github) and "prompt_cache_key" not in resp_kwargs: if not (is_xai or is_github) and "prompt_cache_key" not in resp_kwargs:
@@ -1471,6 +1536,11 @@ class _AsyncAuxiliaryClientBase:
self.api_key = sync_wrapper.api_key self.api_key = sync_wrapper.api_key
self.base_url = sync_wrapper.base_url self.base_url = sync_wrapper.base_url
if hasattr(sync_wrapper, "_real_client"): if hasattr(sync_wrapper, "_real_client"):
# Mirror the sync wrapper's _real_client so cache eviction by leaf OpenAI client (e.g.
# _close_client_on_timeout in #23482) drops this async entry too. Without this, sync and async
# cache entries diverge on poisoning: the sync entry is evicted but the async entry keeps
# reusing the closed transport, failing every subsequent async aux call with 'Connection error'
# until the gateway restarts.
self._real_client = sync_wrapper._real_client self._real_client = sync_wrapper._real_client
@@ -1587,6 +1657,9 @@ class _AnthropicCompletionsAdapter:
# response_format: top-level gets the same translation as the extra_body form; when both # response_format: top-level gets the same translation as the extra_body form; when both
# are present the extra_body form wins. Passthrough excludes ``reasoning``/``response_format`` # are present the extra_body form wins. Passthrough excludes ``reasoning``/``response_format``
# (already TRANSLATED to native fields — raw would 400 on strict gateways) and ``_`` Hermes plumbing. # (already TRANSLATED to native fields — raw would 400 on strict gateways) and ``_`` Hermes plumbing.
# The adapter builds the Messages body from a fixed allow-list of kwargs, so before this an
# unrecognized top-level kwarg was dropped on the floor: the request succeeded but the schema
# contract silently became prompt compliance (#85626 review, point 2).
top_level_response_format = kwargs.get("response_format") top_level_response_format = kwargs.get("response_format")
if top_level_response_format is not None: if top_level_response_format is not None:
_translate_anthropic_response_format(anthropic_kwargs, top_level_response_format) _translate_anthropic_response_format(anthropic_kwargs, top_level_response_format)
@@ -2471,6 +2544,10 @@ def set_runtime_main(
Context-local so concurrent gateway sessions don't clobber each other; legacy mirrors are Context-local so concurrent gateway sessions don't clobber each other; legacy mirrors are
updated for old readers. ``cache_scope`` is the rotation-stable logical cache scope, updated for old readers. ``cache_scope`` is the rotation-stable logical cache scope,
preferred over ``session_id`` for prompt_cache_key derivation. preferred over ``session_id`` for prompt_cache_key derivation.
``cache_scope`` is the rotation-stable logical cache scope (compression- lineage root —
agent/prompt_cache_scope.py) resolved once per turn by turn_context; auxiliary Responses calls prefer it
over ``session_id`` for prompt_cache_key derivation (#79017).
""" """
runtime = { runtime = {
"provider": (provider or "").strip().lower(), "provider": (provider or "").strip().lower(),
@@ -2538,6 +2615,8 @@ def _resolve_custom_runtime() -> Tuple[Optional[str], Optional[str], Optional[st
if base_url_host_matches(custom_base, "openrouter.ai"): if base_url_host_matches(custom_base, "openrouter.ai"):
return None, None, None # requested='custom' falls back to OpenRouter when unconfigured. return None, None, None # requested='custom' falls back to OpenRouter when unconfigured.
# Local servers (Ollama, vLLM, ...) ignore auth but the SDK needs a non-empty key. # Local servers (Ollama, vLLM, ...) ignore auth but the SDK needs a non-empty key.
# Use a placeholder key — the OpenAI SDK requires a non-empty string but local servers ignore the
# Authorization header. Same fix as cli.py _ensure_runtime_credentials() (PR #2556).
if not isinstance(custom_key, str) or not custom_key.strip(): if not isinstance(custom_key, str) or not custom_key.strip():
custom_key = "no-key-required" custom_key = "no-key-required"
if not isinstance(custom_mode, str) or not custom_mode.strip(): if not isinstance(custom_mode, str) or not custom_mode.strip():
@@ -2931,6 +3010,7 @@ def _is_rate_limit_error(exc: Exception) -> bool:
OpenAI's RateLimitError may omit .status_code — matched by class name. A generic 429 without OpenAI's RateLimitError may omit .status_code — matched by class name. A generic 429 without
billing keywords counts as a rate limit. billing keywords counts as a rate limit.
""" """
# (PR #8023 pattern)
if type(exc).__name__ == "RateLimitError": if type(exc).__name__ == "RateLimitError":
return True return True
if getattr(exc, "status_code", None) != 429: if getattr(exc, "status_code", None) != 429:
@@ -3116,6 +3196,8 @@ def _is_invalid_aux_response_error(exc: Exception) -> bool:
# Tasks on a user-visible critical path (compression blocks resuming an oversized session; vision # Tasks on a user-visible critical path (compression blocks resuming an oversized session; vision
# stalls the serialised turn queue). A same-provider retry after a full-budget timeout costs another # stalls the serialised turn queue). A same-provider retry after a full-budget timeout costs another
# whole ``timeout`` window, so they skip straight to fallback; fast blips still retry. # whole ``timeout`` window, so they skip straight to fallback; fast blips still retry.
# Fast blips (a streaming-close or a 5xx) still retry, since those are cheap. See issue #54465 for the
# compression case.
_TIMEOUT_NO_RETRY_TASKS = frozenset({"compression", "vision"}) _TIMEOUT_NO_RETRY_TASKS = frozenset({"compression", "vision"})
@@ -3294,6 +3376,10 @@ def _prepare_same_provider_retry(
) )
# Preserve per-request attribution headers (e.g. Copilot ``x-initiator``) so the retry keeps capability gating. # Preserve per-request attribution headers (e.g. Copilot ``x-initiator``) so the retry keeps capability gating.
if extra_headers: if extra_headers:
# Copilot's ``x-initiator: user``) across the rebuilt-client retry — dropping them here would let a
# recovery retry silently lose capability gating (#60293).
# Preserve per-request attribution headers across the rebuilt-client retry — see the sync variant
# above (#60293).
retry_kwargs["extra_headers"] = dict(extra_headers) retry_kwargs["extra_headers"] = dict(extra_headers)
if _is_anthropic_compat_endpoint(resolved_provider, retry_base): if _is_anthropic_compat_endpoint(resolved_provider, retry_base):
retry_kwargs["messages"] = _convert_openai_images_to_anthropic(retry_kwargs["messages"]) retry_kwargs["messages"] = _convert_openai_images_to_anthropic(retry_kwargs["messages"])
@@ -3432,7 +3518,13 @@ def _coerce_positive_timeout(raw: Any) -> Optional[float]:
def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[float]: def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[float]:
"""Per-entry ``timeout`` for a configured fallback candidate, or None (keep the task-level """Per-entry ``timeout`` for a configured fallback candidate, or None (keep the task-level
timeout). Inheriting the primary's deadline used to kill healthy-but-slower fallbacks.""" timeout). Inheriting the primary's deadline used to kill healthy-but-slower fallbacks.
A fallback candidate previously inherited the exact timeout the primary provider was called with. When
that deadline was tuned for the primary (or the primary simply consumed its whole budget before failing
over), the fallback aborted on the same clock even when independently healthy — a 163k-token compression
that needs ~90s on the fallback died at the primary's 30s deadline every turn (#62452).
"""
entry = _fallback_chain_entry(task, fb_label) entry = _fallback_chain_entry(task, fb_label)
return _coerce_positive_timeout(entry.get("timeout") if entry else None) return _coerce_positive_timeout(entry.get("timeout") if entry else None)
@@ -3598,7 +3690,12 @@ def _call_fallback_candidate_sync(
) -> Optional[Any]: ) -> Optional[Any]:
"""Call one fallback candidate with stale-credential recovery: on an auth error refresh its """Call one fallback candidate with stale-credential recovery: on an auth error refresh its
credentials and retry once with a rebuilt client; if that also auth-fails, quarantine the credentials and retry once with a rebuilt client; if that also auth-fails, quarantine the
provider and return None so the caller moves on. Non-auth errors raise.""" provider and return None so the caller moves on. Non-auth errors raise.
``effective_timeout`` is the task-level deadline; a configured-chain candidate with its own ``timeout``
entry gets that instead, so a fallback tuned differently from the primary is allowed its own budget
(#62452).
"""
destination, fb_kwargs, rebuild = _plan_fallback_candidate( destination, fb_kwargs, rebuild = _plan_fallback_candidate(
fb_client, fb_model, fb_label, task=task, effective_timeout=effective_timeout, fb_client, fb_model, fb_label, task=task, effective_timeout=effective_timeout,
apply_fast_lane=True, messages=messages, tools=tools, temperature=temperature, apply_fast_lane=True, messages=messages, tools=tools, temperature=temperature,
@@ -3751,6 +3848,17 @@ def _try_main_agent_model_fallback(
# too-small aux models; runtime chains must too, or compression stops at a reachable-but-too-small # too-small aux models; runtime chains must too, or compression stops at a reachable-but-too-small
# candidate. ``None`` (unknown) passes through. # candidate. ``None`` (unknown) passes through.
# ── Context-window screening for runtime fallback chains (issue #52392) ── When the runtime auxiliary
# fallback chain selects a candidate that is reachable but has a context window smaller than the compression
# task requires, the call errors out instead of continuing to the next, viable candidate. The startup
# feasibility check in ``agent.conversation_compression.check_compression_model_feasibility`` already
# filters too-small auxiliary models at startup, but the runtime fallback chain
# (``_try_configured_fallback_chain`` and ``_try_main_fallback_chain``) does not apply the same filter, so
# compression can stop at the first alive door even if the room behind it is too small. The helpers below
# screen each candidate by its effective context window before it is returned. ``None`` results from
# ``get_model_context_length`` are passed through (we cannot prove a model is too small, so we do not block
# it). This preserves the existing fallback surface for unrecognised/custom models while closing the gap on
# the well-known ones.
def _task_minimum_context_length(task: Optional[str]) -> Optional[int]: def _task_minimum_context_length(task: Optional[str]) -> Optional[int]:
"""Minimum context length for an auxiliary task; None = no floor (only ``compression`` has one).""" """Minimum context length for an auxiliary task; None = no floor (only ``compression`` has one)."""
return MINIMUM_CONTEXT_LENGTH if task == "compression" else None return MINIMUM_CONTEXT_LENGTH if task == "compression" else None
@@ -3991,6 +4099,7 @@ def _try_main_provider_route(
explicit_base_url = None explicit_base_url = None
elif runtime_base_url: elif runtime_base_url:
# Config-less named custom provider (live runtime only): anonymous custom arm + runtime key. # Config-less named custom provider (live runtime only): anonymous custom arm + runtime key.
# See #34777.
resolved_provider = "custom" resolved_provider = "custom"
explicit_api_key = runtime_api_key or None explicit_api_key = runtime_api_key or None
elif runtime_api_key: elif runtime_api_key:
@@ -4121,6 +4230,7 @@ def _to_async_client(sync_client, model: str, is_vision: bool = False):
_apply_required_codex_headers(async_kwargs, access_token=sync_client.api_key, base_url=sync_base_url) _apply_required_codex_headers(async_kwargs, access_token=sync_client.api_key, base_url=sync_base_url)
async_kwargs = {**_openai_http_client_kwargs(sync_base_url, async_mode=True), **async_kwargs} async_kwargs = {**_openai_http_client_kwargs(sync_base_url, async_mode=True), **async_kwargs}
# Hermes owns the auxiliary retry/timeout budget; disable SDK-internal retries. # Hermes owns the auxiliary retry/timeout budget; disable SDK-internal retries.
# See #54465.
async_kwargs.setdefault("max_retries", 0) async_kwargs.setdefault("max_retries", 0)
return AsyncOpenAI(**async_kwargs), model return AsyncOpenAI(**async_kwargs), model
@@ -4187,6 +4297,7 @@ def _build_bedrock_client(provider: str, model: Optional[str], *, raw_codex: boo
return None, None return None, None
# Region must match the main runtime's resolution (bedrock.region in config first, then # Region must match the main runtime's resolution (bedrock.region in config first, then
# env/profile) so aux calls never leave the primary runtime's configured region. # env/profile) so aux calls never leave the primary runtime's configured region.
# See #53880, #65076.
region = resolve_bedrock_runtime_region() region = resolve_bedrock_runtime_region()
default_model = "anthropic.claude-haiku-4-5-20251001-v1:0" default_model = "anthropic.claude-haiku-4-5-20251001-v1:0"
final_model = _normalize_resolved_model(model or default_model, provider) or default_model final_model = _normalize_resolved_model(model or default_model, provider) or default_model
@@ -4406,6 +4517,8 @@ def _resolve_custom_branch(req: _ResolveRequest) -> _ResolveResult:
elif main_runtime: elif main_runtime:
# Reuse main_runtime's concrete base_url + api_key for a named custom provider; # Reuse main_runtime's concrete base_url + api_key for a named custom provider;
# re-resolving from bare "custom" loses the name and lands on the wrong provider. # re-resolving from bare "custom" loses the name and lands on the wrong provider.
# Re-resolution loses the provider name and falls back to OpenRouter or a wrong API-key provider —
# the main agent already solved this, we just need to reuse its answer. (#45472)
_main_base = str(main_runtime.get("base_url") or "").strip().rstrip("/") _main_base = str(main_runtime.get("base_url") or "").strip().rstrip("/")
_main_key = str(main_runtime.get("api_key") or "").strip() _main_key = str(main_runtime.get("api_key") or "").strip()
if _main_base and _main_key: if _main_base and _main_key:
@@ -4487,6 +4600,7 @@ def _resolve_named_custom_branch(req: _ResolveRequest) -> Optional[_ResolveResul
provider, final_model, entry_api_mode or "chat_completions") provider, final_model, entry_api_mode or "chat_completions")
# anthropic_messages: route via AnthropicAuxiliaryClient (mirrors _try_custom_endpoint); # anthropic_messages: route via AnthropicAuxiliaryClient (mirrors _try_custom_endpoint);
# the Anthropic SDK sees the original (un-rewritten) URL. # the Anthropic SDK sees the original (un-rewritten) URL.
# Mirrors the anonymous-custom branch in _try_custom_endpoint(). See #15033.
if entry_api_mode == "anthropic_messages": if entry_api_mode == "anthropic_messages":
try: try:
from agent.anthropic_adapter import build_anthropic_client from agent.anthropic_adapter import build_anthropic_client
@@ -4710,6 +4824,29 @@ def resolve_provider_client(
# Excluded: ``auto`` (a stale main slug could pair with any picked provider) and Nous + vision (the # Excluded: ``auto`` (a stale main slug could pair with any picked provider) and Nous + vision (the
# Portal's tier-aware vision recommendation must win over a text-only model). # Portal's tier-aware vision recommendation must win over a text-only model).
if not model and provider != "auto" and not (provider == "nous" and is_vision): if not model and provider != "auto" and not (provider == "nous" and is_vision):
# ``auto`` is intentionally excluded: `_resolve_auto(main_runtime=...)` returns the model paired
# with the provider it actually selected. Pre-filling an auto call from `_read_main_model()` can
# leak a stale process-global runtime into a different provider (for example Claude model slug on
# Codex OAuth) and override that correctly resolved model. 1. ``model`` argument (caller knew what
# they wanted) 2. Provider's catalog default — cheap/fast model the provider registered via
# ``ProviderProfile.default_aux_model`` or the legacy ``_API_KEY_PROVIDER_AUX_MODELS_FALLBACK``
# dict. 3. User's main model from ``model.model`` in config.yaml. This is the load-bearing step for
# OAuth providers: an xai-oauth user with grok-4.3 configured gets grok-4.3 for title generation
# instead of silently dropping to whatever Step-2 fallback (#31845). When the main provider is MoA,
# ``_read_main_model_for_aux()`` substitutes the preset's aggregator model — the preset NAME is
# never a valid wire model id, so unset aux models default to the preset's acting model instead.
# Each provider branch below sees a non-empty ``model`` whenever the user has *anything* configured
# — no provider-specific empty-model guards needed. When the user has NOTHING configured (fresh
# install, main_model also empty), the branches still hit their own missing-credentials returns and
# ``_resolve_auto`` falls through to the Step-2 chain as before. Do NOT pre-fill a blank ``auto``
# request from the config/main default here. Claude model sent to Codex after the main lane fell
# back to gpt-5.5). Let _resolve_auto() return the actual current runtime model when the caller did
# not explicitly request one. (# compression-current-model) Nous + vision is the one carve-out: the
# branch below resolves its model from the Portal's tier-aware vision recommendation
# (``_try_nous(vision= True)``), and ``final_model = model or default`` means anything pre-filled
# here wins over that. The main chat model is routinely text-only (e.g. a ``:free`` chat SKU), so
# pre-filling it sends the image to a model that cannot accept one and the Portal 404s. Leave
# ``model`` unset and let the Portal slot through; only an explicit caller model may override it.
model = _get_aux_model_for_provider(provider) or _read_main_model_for_aux() or model model = _get_aux_model_for_provider(provider) or _read_main_model_for_aux() or model
req = _ResolveRequest( req = _ResolveRequest(
provider, original_provider, model, async_mode, raw_codex, provider, original_provider, model, async_mode, raw_codex,
@@ -4974,6 +5111,8 @@ def auxiliary_max_tokens_param(value: int, *, model: Optional[str] = None) -> di
# Client cache: (provider, async_mode, base_url, api_key, api_mode, runtime_key) -> (client, default_model, loop) # Client cache: (provider, async_mode, base_url, api_key, api_mode, runtime_key) -> (client, default_model, loop)
# Loop identity is NOT part of the key: stale-loop entries are replaced in place on async hits, # Loop identity is NOT part of the key: stale-loop entries are replaced in place on async hits,
# bounding growth to one entry per provider config (avoids fd accumulation in gateways). # bounding growth to one entry per provider config (avoids fd accumulation in gateways).
# This bounds cache growth to one entry per unique provider config rather than one per (config ×
# event-loop), which previously caused unbounded fd accumulation in long-running gateway processes (#10200).
_client_cache: Dict[tuple, tuple] = {} _client_cache: Dict[tuple, tuple] = {}
_client_cache_lock = threading.Lock() _client_cache_lock = threading.Lock()
_CLIENT_CACHE_MAX_SIZE = 64 # safety belt — evict oldest when exceeded _CLIENT_CACHE_MAX_SIZE = 64 # safety belt — evict oldest when exceeded
@@ -5056,6 +5195,12 @@ def _refresh_nous_auxiliary_client(
to ``_get_cached_client`` when the stale client was acquired — so the fresh client overwrites to ``_get_cached_client`` when the stale client was acquired — so the fresh client overwrites
the exact entry the stale one is served from. Keying on the resolved model or an empty task the exact entry the stale one is served from. Keying on the resolved model or an empty task
would leave the expired client immortal and every auxiliary call 401ing forever. would leave the expired client immortal and every auxiliary call 401ing forever.
See #56889.
For ``provider == "auto"`` the task participates in the cache key (task-specific fallback policy), so it
MUST be carried into the key here for the same reason as ``lookup_model``; otherwise an auto-provider
client refreshed on a 401 lands under the ``task=""`` key while the stale entry survives under the
task-scoped key (#58894).
""" """
runtime = _resolve_nous_runtime_api(force_refresh=True, stale_access_token=api_key) runtime = _resolve_nous_runtime_api(force_refresh=True, stale_access_token=api_key)
if runtime is None: if runtime is None:
@@ -5212,6 +5357,10 @@ def _get_cached_client(
Async clients bind to the loop they were created on, so every async hit validates the cached Async clients bind to the loop they were created on, so every async hit validates the cached
loop is the current, open loop; stale entries are replaced in place (bounded, no cross-loop reuse). loop is the current, open loop; stale entries are replaced in place (bounded, no cross-loop reuse).
This keeps cache size bounded to one entry per unique provider config, preventing the fd-exhaustion that
previously occurred in long-running gateways where recycled worker threads created unbounded entries
(#10200).
""" """
current_loop = _current_event_loop() if async_mode else None current_loop = _current_event_loop() if async_mode else None
runtime = _normalize_main_runtime(main_runtime) runtime = _normalize_main_runtime(main_runtime)
@@ -5265,6 +5414,14 @@ def _get_cached_client(
_AUX_DIRECT_API_BASE_URLS: Dict[str, str] = {"openai": "https://api.openai.com/v1"} _AUX_DIRECT_API_BASE_URLS: Dict[str, str] = {"openai": "https://api.openai.com/v1"}
# MoA virtual provider: an *explicit* `provider: moa` override (either the caller-passed `provider` arg or
# `auxiliary.<task>.provider` in config.yaml) reaches this function directly — it never goes through
# _resolve_auto(), which only unwraps the *implicit* "main provider is moa" case (#53827). Left as-is, "moa"
# is returned verbatim and resolve_provider_client() looks it up in PROVIDER_REGISTRY (which has no "moa"
# entry — it's not a real HTTP provider), falls to the unknown-provider dead end, and call_llm surfaces a
# nonsensical "MOA_API_KEY environment variable" error for a provider that was never meant to be reached
# over the wire. Auxiliary tasks don't need the reference fan-out — resolve to the preset's aggregator slot
# instead, exactly like the implicit path does (shared helper: _resolve_moa_aggregator).
def _unwrap_moa_provider(prov: str, mdl: Optional[str]) -> Tuple[str, Optional[str]]: def _unwrap_moa_provider(prov: str, mdl: Optional[str]) -> Tuple[str, Optional[str]]:
"""Resolve an *explicit* ``provider: moa`` to its preset's aggregator slot (_resolve_auto() """Resolve an *explicit* ``provider: moa`` to its preset's aggregator slot (_resolve_auto()
only unwraps the implicit case; "moa" isn't in PROVIDER_REGISTRY and would dead-end).""" only unwraps the implicit case; "moa" isn't in PROVIDER_REGISTRY and would dead-end)."""
@@ -5350,6 +5507,7 @@ def _resolve_task_provider_model(
# An explicit provider without base_url adopts the task's configured endpoint (same or # An explicit provider without base_url adopts the task's configured endpoint (same or
# unnamed provider) so the early return below carries it. Explicit "auto" is excluded — it # unnamed provider) so the early return below carries it. Explicit "auto" is excluded — it
# must keep flowing through auto-resolution. # must keep flowing through auto-resolution.
# See #58515.
if provider and provider != "auto" and not base_url and cfg_base_url and cfg_provider in (None, provider): if provider and provider != "auto" and not base_url and cfg_base_url and cfg_provider in (None, provider):
base_url = cfg_base_url base_url = cfg_base_url
if not api_key: if not api_key:
@@ -5375,6 +5533,11 @@ _DEFAULT_AUX_TIMEOUT = 30.0
# Reasoning compression models can exceed the default 120 s config timeout, falling back to the # Reasoning compression models can exceed the default 120 s config timeout, falling back to the
# deterministic marker. Bounded *floor* for config-derived compression timeouts only; never # deterministic marker. Bounded *floor* for config-derived compression timeouts only; never
# overrides an explicit per-call timeout. # overrides an explicit per-call timeout.
# Compression summarises large conversation histories; a reasoning auxiliary model (e.g. Codex / GPT-5.5)
# can legitimately take longer than the default ``auxiliary.compression.timeout`` (120 s), causing the
# stream to time out and the compressor to fall back to the deterministic context marker (#54915). A floor
# is harmless for fast compression models (they finish before the deadline) and is a minimum, so a higher
# config value is kept unchanged.
_COMPRESSION_TIMEOUT_FLOOR_SECONDS = 300.0 _COMPRESSION_TIMEOUT_FLOOR_SECONDS = 300.0
@@ -5531,6 +5694,9 @@ def _get_task_extra_body(task: str) -> Dict[str, Any]:
# Per-task concurrency limiting: many sessions can spawn unbounded background aux calls, each # Per-task concurrency limiting: many sessions can spawn unbounded background aux calls, each
# retrying across the fallback chain during incidents. # retrying across the fallback chain during incidents.
# During provider incidents each call also retries / fans out across the fallback chain, multiplying request
# volume on already-degraded endpoints. A per-task semaphore caps in-flight calls so retry amplification
# stays bounded. See #23324.
_aux_sync_semaphores: Dict[str, Tuple[int, threading.BoundedSemaphore]] = {} _aux_sync_semaphores: Dict[str, Tuple[int, threading.BoundedSemaphore]] = {}
_aux_async_semaphores: Dict[Tuple[str, int], Tuple[int, Any]] = {} _aux_async_semaphores: Dict[Tuple[str, int], Tuple[int, Any]] = {}
_aux_sem_lock = threading.Lock() _aux_sem_lock = threading.Lock()
@@ -5833,6 +5999,11 @@ def _validate_llm_response(
Also the single aux-usage accounting chokepoint: every successful non-streaming response Also the single aux-usage accounting chokepoint: every successful non-streaming response
passes here exactly once; *provider*/*base_url* are optional hints. passes here exactly once; *provider*/*base_url* are optional hints.
See #7264.
Recording is best-effort and never affects validation. *provider*/*base_url* are optional accounting
hints — fallback-path calls omit them and the row keeps the model (read from the response itself) with
an empty route. See #23270.
""" """
if response is None: if response is None:
raise RuntimeError(f"Auxiliary {task or 'call'}: LLM returned None response") raise RuntimeError(f"Auxiliary {task or 'call'}: LLM returned None response")
@@ -6007,6 +6178,7 @@ _AFFORDABLE_TOKENS_RE = re.compile(r"can only afford\s+([0-9][0-9,]*)", re.IGNOR
# Below the floor the affordable budget can't fit a useful aux output — treat as exhaustion; # Below the floor the affordable budget can't fit a useful aux output — treat as exhaustion;
# the margin keeps provider-side token-count rounding from 402-ing the retry. # the margin keeps provider-side token-count rounding from 402-ing the retry.
_AFFORDABLE_RETRY_FLOOR_TOKENS = 512 _AFFORDABLE_RETRY_FLOOR_TOKENS = 512
# See #49785.
_AFFORDABLE_RETRY_MARGIN_TOKENS = 64 _AFFORDABLE_RETRY_MARGIN_TOKENS = 64
@@ -6067,7 +6239,18 @@ def _create_with_progress_once(
"""create() that streams (and re-aggregates, ticking the hook per substantive chunk) when a """create() that streams (and re-aggregates, ticking the hook per substantive chunk) when a
progress hook is active or the provider is stream-only; plain ``create(**kwargs)`` otherwise progress hook is active or the provider is stream-only; plain ``create(**kwargs)`` otherwise
or when the adapter streams internally. Streaming rejections fall back to a plain call — or when the adapter streams internally. Streaming rejections fall back to a plain call —
except under ``force_stream``.""" except under ``force_stream``.
Behavior is byte-for-byte identical to a plain ``create(**kwargs)`` when neither trigger applies (every
existing caller/task) or when the client's wire adapter streams internally. With a hook + a
chunk-capable client, the request is sent with ``stream=True`` and aggregated, ticking the hook only for
substantive chunks. The configured ``timeout`` acts per stream read (idle) rather than as a total
budget, and outer liveness watchdogs see tokens moving. ``force_stream=True`` (stream-only providers
such as Tencent Copilot — credit @kudi88, PR #60686) takes the same streamed path even without a hook.
Providers that reject the streamed request fall back to the plain non-streaming call — except under
``force_stream``, where a stream-only provider rejects the plain call by definition, so the original
error is surfaced to the normal recovery chains instead.
"""
_notify_aux_dispatch() _notify_aux_dispatch()
_notify_aux_progress() # Preserve the watchdog's historical dispatch tick. _notify_aux_progress() # Preserve the watchdog's historical dispatch tick.
if (not _aux_progress_active() and not force_stream) or _client_streams_internally(client): if (not _aux_progress_active() and not force_stream) or _client_streams_internally(client):
@@ -6141,6 +6324,9 @@ class _ChatStreamAccumulator:
self._total_ceiling = total_ceiling self._total_ceiling = total_ceiling
# Absolute instant the waiting host gives up; checked alongside (not instead of) the # Absolute instant the waiting host gives up; checked alongside (not instead of) the
# ceiling, and unaffected by pre-construction dispatch/TTFT. # ceiling, and unaffected by pre-construction dispatch/TTFT.
# Checked as well as (not instead of) the ceiling above: the ceiling still bounds callers with no
# host deadline, and the host deadline is absolute, so it is unaffected by however long dispatch and
# TTFT took before this accumulator was constructed. See #99692.
self._host_deadline = host_deadline self._host_deadline = host_deadline
self.content_parts: List[str] = [] self.content_parts: List[str] = []
self.reasoning_parts: List[str] = [] self.reasoning_parts: List[str] = []
@@ -6626,6 +6812,12 @@ def _ladder_provider_fallback(first_err: Exception, route: _LadderRoute):
response) bypass the explicit-provider gate — the provider cannot serve this request response) bypass the explicit-provider gate — the provider cannot serve this request
regardless of user intent. Auth errors only fall back in auto mode.""" regardless of user intent. Auth errors only fall back in auto mode."""
task, tag, resolved_provider = route.task, route.tag, route.resolved_provider task, tag, resolved_provider = route.task, route.tag, route.resolved_provider
# Respect explicit provider choice for transient errors (auth, request validation, etc.) but allow
# fallback when the provider clearly cannot serve the request due to capacity: payment/quota exhaustion
# and connection failures are capacity problems, not request constraints. See #26803: daily token quota
# (429 + "too many tokens per day") must fall back just like a 402 credit error.
# Rate limits are included: after retries are exhausted, a 429 means the provider is at capacity. See
# #52228. See #26803: daily token quota must fall back like a 402 credit error.
is_auto = resolved_provider in {"auto", "", None} is_auto = resolved_provider in {"auto", "", None}
reason = next((label for predicate, label in _FALLBACK_REASONS if predicate(first_err)), None) reason = next((label for predicate, label in _FALLBACK_REASONS if predicate(first_err)), None)
is_capacity_error = any( is_capacity_error = any(
@@ -6669,6 +6861,9 @@ def _ladder_provider_fallback(first_err: Exception, route: _LadderRoute):
break break
# All fallback layers exhausted — one user-visible warning, then re-raise. # All fallback layers exhausted — one user-visible warning, then re-raise.
logger.warning("Auxiliary %s%s: %s on %s and all fallbacks exhausted " logger.warning("Auxiliary %s%s: %s on %s and all fallbacks exhausted "
# All fallback layers exhausted — emit a single user-visible warning so the operator
# knows aux task is about to fail. (#26882) The error itself is re-raised below.
# (#26882)
"(fallback_chain + main agent model). Raising original error.", "(fallback_chain + main agent model). Raising original error.",
task or "call", tag, reason, resolved_provider) task or "call", tag, reason, resolved_provider)
return None return None
@@ -6705,6 +6900,10 @@ def _aux_recovery_ladder(
return resp return resp
# Connection/timeout errors poison the cached client (closed transport, half-read # Connection/timeout errors poison the cached client (closed transport, half-read
# stream); evict so the next aux call rebuilds a fresh one. # stream); evict so the next aux call rebuilds a fresh one.
# Drop it from the cache regardless of whether we found a fallback above so the next auxiliary call
# rebuilds a fresh client instead of reusing the dead one. See issue #23432.
# Mirror the sync path: drop poisoned clients on connection/timeout so the next aux call rebuilds. See
# issue #23432.
if _is_connection_error(first_err): if _is_connection_error(first_err):
try: try:
_evict_cached_client_instance(client) _evict_cached_client_instance(client)
@@ -6924,6 +7123,17 @@ def _call_llm_impl(
return _relay_sync_stream(client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode) return _relay_sync_stream(client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode)
def _primary(**validate_kw: Any) -> Any: def _primary(**validate_kw: Any) -> Any:
# Retry on the same provider for a transient transport blip (connection reset / streaming-close /
# incomplete chunked read / 5xx / 408) before the except-chain below escalates to provider/model
# fallback. A dropped connection shouldn't abandon an otherwise-healthy provider — this especially
# matters for pinned auxiliary calls like MoA reference advisors, where "fallback to another
# provider" is not a meaningful recovery (the advisor is a specific model), so a transient blip that
# isn't retried simply loses that advisor for the turn (root of the run2 double-advisor "Connection
# error" collapse — a genuine upstream blip hitting both parallel advisors at once). Attempts are
# bounded and use exponential backoff. Count is configurable via auxiliary.transient_retries
# (default 2 retries → 3 total attempts); a second/third failure or any non-transient error falls
# through to ``first_err`` and the existing fallback handling unchanged. Unified home for the
# transient retry every auxiliary task shares. (PR #16587)
return _validate_llm_response( return _validate_llm_response(
_relay_sync_completion( _relay_sync_completion(
client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode, client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode,
@@ -7087,6 +7297,7 @@ async def _async_call_llm_impl(
client, kwargs, request_provider = req.client, req.kwargs, req.request_provider client, kwargs, request_provider = req.client, req.kwargs, req.request_provider
try: try:
# Retry ONCE on the same provider for a transient blip before fallback (see call_llm()). # Retry ONCE on the same provider for a transient blip before fallback (see call_llm()).
# (PR #16587)
_force_stream_async = ( _force_stream_async = (
_provider_requires_stream(request_provider, req.base_info or req.resolved_base_url) _provider_requires_stream(request_provider, req.base_info or req.resolved_base_url)
and not isinstance(client, ( and not isinstance(client, (
+5
View File
@@ -78,6 +78,9 @@ def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool:
(stranded failover). Same label = same configured credential; custom entries can each carry (stranded failover). Same label = same configured credential; custom entries can each carry
their own api_key, so a shared URL alone is only a weak signal when a label is missing.""" their own api_key, so a shared URL alone is only a weak signal when a label is missing."""
if a.provider and b.provider: if a.provider and b.provider:
# Different labels = different credential config (first-class registry providers explicitly so —
# #70893; custom entries can each carry their own api_key, so sameness is unprovable and we must not
# skip).
return a.provider == b.provider return a.provider == b.provider
return bool(a.base_url and a.base_url == b.base_url) return bool(a.base_url and a.base_url == b.base_url)
@@ -100,6 +103,8 @@ def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool:
if not (a.provider and b.provider and a.provider == b.provider): if not (a.provider and b.provider and a.provider == b.provider):
return bool( return bool(
a.base_url a.base_url
# Same-host different-label shims: same URL + same model IS the same deployment even when the
# alias labels differ (#22548) — unless both labels are first-class registry providers (#70893).
and a.base_url == b.base_url and a.base_url == b.base_url
and a.model and a.model
and a.model == b.model and a.model == b.model
+55 -3
View File
@@ -119,11 +119,18 @@ def _interrupt_background_review(review_agent: Any) -> None:
def cancel_background_review_for_live_turn(agent: Any) -> None: def cancel_background_review_for_live_turn(agent: Any) -> None:
"""Cancel the current review and await its request-phase acknowledgement. Foreground priority: """Cancel the current review and await its request-phase acknowledgement. Foreground priority:
past the bounded deadline, warn and let the live turn proceed — self-improvement work must past the bounded deadline, warn and let the live turn proceed — self-improvement work must
never block a user-facing turn.""" never block a user-facing turn.
Foreground priority is preserved: if the review does not acknowledge within the bounded deadline, a
warning is logged and the live turn proceeds anyway. See #84423.
"""
with _optional_lock(agent, "_background_review_lock"): with _optional_lock(agent, "_background_review_lock"):
run = getattr(agent, "_background_review_run", None) run = getattr(agent, "_background_review_run", None)
legacy_agent = getattr(agent, "_background_review_agent", None) legacy_agent = getattr(agent, "_background_review_agent", None)
review_agent = legacy_agent if run is None else run.cancel() review_agent = legacy_agent if run is None else run.cancel()
# Attribute the review fork's usage to the PARENT session. Snapshot BEFORE unregister/close so counters
# survive teardown. Placed in this finally so a fork that consumed tokens and THEN raised is still
# attributed (issue #87250). Best-effort: the recorder never raises into the review thread.
if review_agent is not None: if review_agent is not None:
_interrupt_background_review(review_agent) _interrupt_background_review(review_agent)
if run is None: if run is None:
@@ -594,7 +601,10 @@ def summarize_background_review_actions(
skill-management tool results from the review agent's messages, skipping tool messages already skill-management tool results from the review agent's messages, skipping tool messages already
present in ``prior_snapshot`` so inherited results are not re-surfaced as fresh work. present in ``prior_snapshot`` so inherited results are not re-surfaced as fresh work.
``notification_mode``: ``off`` -> no actions; ``on`` -> generic "Memory updated"/tool messages; ``notification_mode``: ``off`` -> no actions; ``on`` -> generic "Memory updated"/tool messages;
``verbose`` -> content previews from the tool-call arguments.""" ``verbose`` -> content previews from the tool-call arguments.
See #14944.
"""
mode = str(notification_mode or "on").lower() mode = str(notification_mode or "on").lower()
if mode == "off": if mode == "off":
return [] return []
@@ -814,6 +824,13 @@ def build_cache_parity_fork(
# Same model only: share the warm cached system prompt (~26% cost cut; a rebuilt prompt misses # Same model only: share the warm cached system prompt (~26% cost cut; a rebuilt prompt misses
# the byte-exact prefix key) and pin session_start so any re-render (compression, plugin # the byte-exact prefix key) and pin session_start so any re-render (compression, plugin
# hooks) stays byte-identical. # hooks) stays byte-identical.
# Inherit the parent's cached system prompt verbatim so the review fork's outbound HTTP request hits the
# same Anthropic/OpenRouter prefix cache the parent warmed. Without this, the fork rebuilds the system
# prompt from scratch (fresh _hermes_now() timestamp, fresh session_id, narrower toolset → different
# skills_prompt) and the byte-exact prefix-cache key misses. See issue #25322 and PR #17276 for the full
# analysis + measured impact (~26% end-to-end cost reduction on Sonnet 4.5). When routed to a different
# model the parent's cached prompt is for the wrong model/cache key and would miss anyway, so let the
# routed fork build its own.
if not _routed: if not _routed:
review_agent._cached_system_prompt = agent._cached_system_prompt review_agent._cached_system_prompt = agent._cached_system_prompt
review_agent.session_start = agent.session_start review_agent.session_start = agent.session_start
@@ -824,6 +841,9 @@ def build_cache_parity_fork(
return review_agent, _rt, _routed return review_agent, _rt, _routed
# Install a non-interactive approval callback on this worker thread so any dangerous-command guard the
# review agent trips resolves to "deny" instead of falling back to input() -- which deadlocks against the
# parent's prompt_toolkit TUI (#15216). Same pattern as _subagent_auto_deny in tools/delegate_tool.py.
def _bg_review_auto_deny(command, description, **kwargs): def _bg_review_auto_deny(command, description, **kwargs):
"""Non-interactive approval: dangerous-command guards resolve to "deny" instead of input(), """Non-interactive approval: dangerous-command guards resolve to "deny" instead of input(),
which would deadlock against the parent's TUI.""" which would deadlock against the parent's TUI."""
@@ -876,6 +896,21 @@ def _review_tool_whitelist(review_agent: Any, task_cfg: Optional[Dict[str, Any]]
whitelist |= {"read_file", "search_files"} whitelist |= {"read_file", "search_files"}
# ``extra_tools`` admits named parent tools (e.g. a human-gated proposal tool). The whitelist # ``extra_tools`` admits named parent tools (e.g. a human-gated proposal tool). The whitelist
# can only admit, never advertise: a listed tool must already exist in the inherited schema. # can only admit, never advertise: a listed tool must already exist in the inherited schema.
# Read-only file tools are whitelisted too (#61521, #39996): the model naturally reaches for
# read_file/search_files to inspect a skill before patching it. Denying them caused a per-review denial
# storm (~142 denials + ~204 read-before-write refusals over 2 days on one deployment) that starved the
# self-improvement loop — the model never loaded SKILL.md the way the read-before-write guard requires,
# so almost no patch landed. This is a DISPATCH-side change only: the advertised ``tools[]`` stays
# byte-identical to the parent's, so prompt-cache parity is untouched. read_file registers the read with
# the read-before-write guard (tools/file_tools.py), so a read_file → skill_manage(patch) sequence now
# succeeds. Write tools (write_file/patch/terminal) stay denied — autonomous maintenance must go through
# skill_manage's validation, and the deny message below names that substitute so one denial redirects
# the model instead of a storm.
# Profile-configured opt-in tools (#44672, salvage #82146 by @BrinShadewater):
# ``auxiliary.background_review.extra_tools`` admits named parent tools to the review whitelist — e.g. a
# human-gated proposal tool or a memory-provider write surface. Read from task_cfg (the
# auxiliary.background_review block already loaded for this spawn) so no extra config I/O happens per
# review.
configured_extra_tools: set = set() configured_extra_tools: set = set()
try: try:
extra_raw = _background_review_task_config(task_cfg).get("extra_tools", []) extra_raw = _background_review_task_config(task_cfg).get("extra_tools", [])
@@ -972,7 +1007,10 @@ def _run_review_in_thread(
"""Daemon-thread worker: build the fork, run the prompt, surface the action summary via """Daemon-thread worker: build the fork, run the prompt, surface the action summary via
``agent._safe_print`` / ``background_review_callback``. ``review_run`` (from ``agent._safe_print`` / ``background_review_callback``. ``review_run`` (from
:func:`prepare_background_review_run`) cancelled before the first provider call aborts :func:`prepare_background_review_run`) cancelled before the first provider call aborts
without entering ``run_conversation()``.""" without entering ``run_conversation()``.
See #84423.
"""
if review_run is not None and review_run.cancel_requested.is_set(): if review_run is not None and review_run.cancel_requested.is_set():
finish_background_review_run(agent, review_run) finish_background_review_run(agent, review_run)
return return
@@ -993,11 +1031,25 @@ def _run_review_in_thread(
try: try:
# Silence stdout/stderr for THIS thread only: a process-global redirect would blank every # Silence stdout/stderr for THIS thread only: a process-global redirect would blank every
# other thread's console for the whole review. # other thread's console for the whole review.
# A process-global ``contextlib.redirect_stdout(devnull)`` here would also blank
# ``sys.stdout``/``sys.stderr`` for every other thread — including a gateway event-loop thread
# driving a Telegram long-poll — for the full duration of the review (tens of seconds), swallowing
# their console output (#55769 / #55925). ``thread_scoped_silence`` routes only this thread's writes
# to devnull and leaves all other threads on the real streams.
with thread_scoped_silence(): with thread_scoped_silence():
_run_review_fork(agent, messages_snapshot, prompt, task_cfg, review_run, st) _run_review_fork(agent, messages_snapshot, prompt, task_cfg, review_run, st)
# A buggy/legacy tool response shape must NOT take down the whole review (the outer # A buggy/legacy tool response shape must NOT take down the whole review (the outer
# except would discard every action the fork DID complete), so coerce to an empty list. # except would discard every action the fork DID complete), so coerce to an empty list.
try: try:
# Scan the review agent's messages for successful tool actions and surface a compact summary to
# the user. Tool messages already present in messages_snapshot must be skipped, since the review
# agent inherits that history and would otherwise re-surface stale "created"/"updated" messages
# from the prior conversation as if they just happened (issue #14944). ``_change`` returned as a
# list instead of a dict, #59437) must NOT take down the whole review with an AttributeError,
# since the caller's outer except logs only "Background memory/skill review failed" and discards
# every successful action the fork DID complete before the crash. Coerce an exception into an
# empty actions list so the partial valid actions from earlier in the messages are returned
# instead.
actions = summarize_background_review_actions( actions = summarize_background_review_actions(
st.review_messages, messages_snapshot, st.review_messages, messages_snapshot,
notification_mode=getattr(agent, "memory_notifications", "on"), notification_mode=getattr(agent, "memory_notifications", "on"),
+14 -1
View File
@@ -24,6 +24,11 @@ logger = logging.getLogger(__name__)
# boto3 is not in the [all] extras; lazy_deps installs it on demand. # boto3 is not in the [all] extras; lazy_deps installs it on demand.
try: try:
# --------------------------------------------------------------------------- Ensure boto3/botocore are
# installed before any code in this module runs. Upstream removed boto3 from [all] extras (PRs #24220,
# #24515); lazy_deps handles on-demand installation so the Bedrock provider still works in the EKS
# deployment without baking boto3 into the base image.
# ---------------------------------------------------------------------------
from tools.lazy_deps import ensure from tools.lazy_deps import ensure
ensure("provider.bedrock", prompt=False) ensure("provider.bedrock", prompt=False)
except Exception: except Exception:
@@ -260,7 +265,12 @@ def resolve_aws_auth_env_var(env: Optional[Dict[str, str]] = None) -> Optional[s
def has_aws_credentials(env: Optional[Dict[str, str]] = None) -> bool: def has_aws_credentials(env: Optional[Dict[str, str]] = None) -> bool:
"""True if any AWS credential source (env vars or boto3 chain) is detected.""" """True if any AWS credential source (env vars or boto3 chain) is detected.
This two-tier approach mirrors the pattern from OpenClaw PR #62673: cloud environments (EC2, ECS,
Lambda) provide credentials via instance metadata, not environment variables. The env-var check is a
fast path for local development; the boto3 fallback covers all cloud deployments.
"""
return resolve_aws_auth_env_var(env) is not None or _boto3_chain_has_credentials() return resolve_aws_auth_env_var(env) is not None or _boto3_chain_has_credentials()
@@ -431,6 +441,8 @@ def convert_tools_to_converse(tools: List[Dict]) -> List[Dict]:
# Converse rejects empty OR whitespace-only text blocks, so the placeholder must be non-whitespace. # Converse rejects empty OR whitespace-only text blocks, so the placeholder must be non-whitespace.
# A lone space is whitespace and is rejected too — the placeholder MUST itself be non-whitespace. Ref: issue
# #9486.
_EMPTY_TEXT_PLACEHOLDER = "(empty)" _EMPTY_TEXT_PLACEHOLDER = "(empty)"
_PLACEHOLDER_BLOCK = {"text": _EMPTY_TEXT_PLACEHOLDER} _PLACEHOLDER_BLOCK = {"text": _EMPTY_TEXT_PLACEHOLDER}
@@ -447,6 +459,7 @@ def _image_block_from_data_url(url: str) -> Dict:
header, _, data = url.partition(",") header, _, data = url.partition(",")
media_type = (header[5:].split(";")[0] if header.startswith("data:") else "") or "image/jpeg" media_type = (header[5:].split(";")[0] if header.startswith("data:") else "") or "image/jpeg"
try: try:
# Ref: #33317.
raw_bytes = base64.b64decode(data) raw_bytes = base64.b64decode(data)
except Exception: except Exception:
raw_bytes = data.encode("utf-8") raw_bytes = data.encode("utf-8")
+7
View File
@@ -58,6 +58,13 @@ class BrowserProvider(ProviderBase):
# Legacy ``CloudBrowserProvider`` names still used by ``tools.browser_tool`` and out-of-tree subclasses. # Legacy ``CloudBrowserProvider`` names still used by ``tools.browser_tool`` and out-of-tree subclasses.
# ------------------------------------------------------------------ Backward-compat shims for the
# legacy CloudBrowserProvider API ------------------------------------------------------------------ The
# pre-PR-#25214 ABC exposed ``is_configured()`` and ``provider_name()``; ``tools.browser_tool`` has ~6
# callers that still use those names. Rather than churn every callsite (and break out-of-tree downstream
# code that subclassed CloudBrowserProvider), we expose the old names as thin delegations to the new
# API. Subclasses MUST implement :meth:`is_available` and :attr:`name`; they may override
# ``is_configured`` / ``provider_name`` for compatibility with the legacy ABC but it is not required.
def is_configured(self) -> bool: def is_configured(self) -> bool:
"""Backward-compat alias for :meth:`is_available`.""" """Backward-compat alias for :meth:`is_available`."""
return self.is_available() return self.is_available()
+47
View File
@@ -30,6 +30,9 @@ from agent.errors import EmptyStreamError
from agent.fast_mode import effective_request_overrides from agent.fast_mode import effective_request_overrides
from agent.turn_context import substitute_api_content from agent.turn_context import substitute_api_content
from agent.gemini_native_adapter import is_native_gemini_base_url from agent.gemini_native_adapter import is_native_gemini_base_url
# Remote endpoints must never be fingerprinted: the probe waterfall is only valid for local/LM-Studio/Ollama
# boxes. Non-Ollama remotes (sglang, vLLM, OpenAI-compat) expose Ollama-compat endpoints that can
# misidentify and, without an api_key, return 401 on every leg (issue #89863).
from agent.model_metadata import is_local_endpoint from agent.model_metadata import is_local_endpoint
from agent.message_content import flatten_message_text from agent.message_content import flatten_message_text
from agent.message_metadata import append_message, stamp_message_timestamp from agent.message_metadata import append_message, stamp_message_timestamp
@@ -1630,6 +1633,18 @@ def _assistant_tool_call_dict(agent, tool_call, index: int) -> dict:
"function": {"name": tool_call.function.name, "arguments": tool_call.function.arguments}} "function": {"name": tool_call.function.name, "arguments": tool_call.function.arguments}}
# Preserve extra_content (Gemini thought_signature) or Gemini 3 thinking # Preserve extra_content (Gemini thought_signature) or Gemini 3 thinking
# models 400 on the next request. # models 400 on the next request.
# Tool-call arguments are intentionally NOT redacted here. This dict enters the in-memory conversation
# history that is replayed to the model on every subsequent turn AND persisted to state.db, which is
# itself replayed verbatim on session resume (get_messages_as_conversation). Masking a credential to
# `***` here poisons that replay: the model reads back its own `PGPASSWORD='***' psql ...` call and
# copies the placeholder into the next tool call, breaking every credential-dependent command on the
# second turn (#43083). The masking also provided no real protection — the same secret still leaks
# verbatim through tool OUTPUT (file contents, command output, diffs, the compaction block), none of
# which this pass ever touched. Keeping secrets out of the replayable store is a separate
# tokenization/vault concern, not something arg-redaction can deliver without breaking replay.
# Storage-time redaction remains governed by the `security.redact_secrets` toggle. (#19798 introduced
# this; #43083 removed it.) Preserve extra_content (e.g. Gemini thought_signature) so it is sent back on
# subsequent API calls. Without this, Gemini 3 thinking models reject the request with a 400 error.
extra = getattr(tool_call, "extra_content", None) extra = getattr(tool_call, "extra_content", None)
if extra is not None: if extra is not None:
tc_dict["extra_content"] = _dump_if_model(extra) tc_dict["extra_content"] = _dump_if_model(extra)
@@ -1658,6 +1673,11 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic
elif assistant_tool_calls and agent._needs_thinking_reasoning_pad(): elif assistant_tool_calls and agent._needs_thinking_reasoning_pad():
# DeepSeek v4 / Kimi thinking modes 400 on a replayed tool-call message without # DeepSeek v4 / Kimi thinking modes 400 on a replayed tool-call message without
# reasoning_content; pad with a single space (empty string is rejected too). # reasoning_content; pad with a single space (empty string is rejected too).
# Without it, replaying the persisted message causes HTTP 400 ("The reasoning_content in the
# thinking mode must be passed back to the API"). Include streamed reasoning text when captured;
# otherwise pad with a single space — DeepSeek V4 Pro tightened validation and rejects empty string
# ("The reasoning content in the thinking mode must be passed back to the API"). A space satisfies
# non-empty checks everywhere without leaking fabricated reasoning. Refs #15250, #17400, #17341.
msg["reasoning_content"] = reasoning_text or " " msg["reasoning_content"] = reasoning_text or " "
elif reasoning_text: elif reasoning_text:
# Streaming-only providers accumulate reasoning via deltas and never set # Streaming-only providers accumulate reasoning via deltas and never set
@@ -1665,6 +1685,16 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic
# Promote ONLY when nothing set the field: SDK reasoning_content and the # Promote ONLY when nothing set the field: SDK reasoning_content and the
# tool-call pad win, and reasoning-less turns leave the field absent so # tool-call pad win, and reasoning-less turns leave the field absent so
# the replay-time leak guard and promotion tiers still apply. # the replay-time leak guard and promotion tiers still apply.
# Additive fallback (refs #16844, #16884). Streaming-only providers (glm, MiniMax, gpt-5.x via aigw,
# Anthropic via openai-compat shims) accumulate reasoning through ``delta.reasoning_content`` chunks
# but never land it on the message object as a top-level attribute, so neither branch above fires
# and the chain-of-thought is stored only under the internal ``reasoning`` key. When the user later
# replays that history through a DeepSeek-v4 / Kimi thinking model, the missing
# ``reasoning_content`` causes HTTP 400 ("The reasoning_content in the thinking mode must be passed
# back to the API."). Promote the already-sanitized streamed ``reasoning_text`` to
# ``reasoning_content`` at write time, but ONLY when no prior branch already set it AND we actually
# captured reasoning text. This preserves every existing behavior: - SDK-exposed
# ``reasoning_content`` (OpenAI/Moonshot/DeepSeek SDK) still wins.
msg["reasoning_content"] = reasoning_text msg["reasoning_content"] = reasoning_text
if getattr(assistant_message, "reasoning_details", None): if getattr(assistant_message, "reasoning_details", None):
@@ -1874,6 +1904,8 @@ def _should_skip_fallback_candidate(agent, fb: dict, fb_key: tuple, fb_provider:
return True return True
# Identity semantics (axes, shim aliases, credential surfaces, multi-endpoint pools) # Identity semantics (axes, shim aliases, credential surfaces, multi-endpoint pools)
# are owned by agent.backend_identity — do not re-implement comparisons here. # are owned by agent.backend_identity — do not re-implement comparisons here.
# Skip entries that resolve to the same backend that just failed — falling back to it loops the failure.
# See #22548, #62984, #70893.
from agent.backend_identity import BackendIdentity, should_skip_candidate from agent.backend_identity import BackendIdentity, should_skip_candidate
current_ident = BackendIdentity.build(provider=getattr(agent, "provider", ""), current_ident = BackendIdentity.build(provider=getattr(agent, "provider", ""),
model=getattr(agent, "model", ""), base_url=str(getattr(agent, "base_url", "") or "")) model=getattr(agent, "model", ""), base_url=str(getattr(agent, "base_url", "") or ""))
@@ -1937,6 +1969,8 @@ def _reresolve_fallback_reasoning_config(agent) -> None:
"""Per-model override > global reasoning_effort (YAML False = disabled); a config load """Per-model override > global reasoning_effort (YAML False = disabled); a config load
failure keeps the current reasoning_config rather than killing the swap.""" failure keeps the current reasoning_config rather than killing the swap."""
try: try:
# Re-resolve reasoning_config for the new fallback model (Closes #21256). Wrapped in try/except
# because a config load failure must not kill the swap.
from hermes_cli.config import load_config from hermes_cli.config import load_config
from hermes_constants import resolve_reasoning_config from hermes_constants import resolve_reasoning_config
agent.reasoning_config = resolve_reasoning_config(load_config() or {}, agent.model) agent.reasoning_config = resolve_reasoning_config(load_config() or {}, agent.model)
@@ -2032,6 +2066,7 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool
# Clear the per-config context_length override so the fallback model's own context # Clear the per-config context_length override so the fallback model's own context
# window is resolved instead of the previous model's stale value. # window is resolved instead of the previous model's stale value.
# See #22387.
agent._config_context_length = None agent._config_context_length = None
agent.model, agent.provider, agent.requested_provider = fb_model, fb_provider, fb_provider agent.model, agent.provider, agent.requested_provider = fb_model, fb_provider, fb_provider
agent.base_url, agent.api_mode = fb_base_url, fb_api_mode agent.base_url, agent.api_mode = fb_base_url, fb_api_mode
@@ -2102,6 +2137,13 @@ def _iteration_summary_api_messages(agent, messages: list) -> list:
api_msg.pop(key, None) api_msg.pop(key, None)
# api_content holds the exact bytes the main loop sent; substituting (not popping) # api_content holds the exact bytes the main loop sent; substituting (not popping)
# keeps the summary's prefix identical instead of re-prefilling the largest context. # keeps the summary's prefix identical instead of re-prefilling the largest context.
# Strict OpenAI-compatible gateways (Fireworks-backed OpenCode Go, Mistral, Moonshot/Kimi) reject
# any message key outside the Chat Completions schema. The main loop drops these via
# ChatCompletionsTransport.convert_messages(), but the summary path hand-builds messages and calls
# chat.completions.create() directly, bypassing the transport — so mirror that sanitization here:
# tool_name (SQLite FTS bookkeeping), the codex_* reasoning carriers, timestamp (preserved on
# gateway user replay entries for the stale-confirmation expiry check — #47868 rejection class), and
# every Hermes-internal underscore-prefixed scaffolding key.
substitute_api_content(api_msg) substitute_api_content(api_msg)
if needs_sanitize: if needs_sanitize:
agent._sanitize_tool_calls_for_strict_api(api_msg, model=sanitize_model) agent._sanitize_tool_calls_for_strict_api(api_msg, model=sanitize_model)
@@ -2243,6 +2285,8 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str:
if getattr(agent, "suppress_status_output", False): if getattr(agent, "suppress_status_output", False):
# Strict machine-readable mode (-Q, oneshot): keep diagnostics off stdout. quiet_mode is # Strict machine-readable mode (-Q, oneshot): keep diagnostics off stdout. quiet_mode is
# NOT the gate — the interactive CLI runs quiet_mode=True by default and must see this. # NOT the gate — the interactive CLI runs quiet_mode=True by default and must see this.
# Strict machine-readable mode (hermes chat -Q, oneshot, background review): keep diagnostics out of
# stdout so wrappers receive only the final assistant content (#93220 class).
logger.warning(warning) logger.warning(warning)
else: else:
agent._safe_print(warning) agent._safe_print(warning)
@@ -2797,6 +2841,8 @@ class _StreamingCall:
as in-stream chunks (choices=None + error_type/error_message), which as in-stream chunks (choices=None + error_type/error_message), which
would otherwise surface as a misleading EmptyStreamError plus retries.""" would otherwise surface as a misleading EmptyStreamError plus retries."""
usage = chunk.usage if hasattr(chunk, "usage") and chunk.usage else None # final usage chunk usage = chunk.usage if hasattr(chunk, "usage") and chunk.usage else None # final usage chunk
# Without this check the error is silently dropped and the stream ends empty → EmptyStreamError →
# misleading "empty stream" message and pointless retries on the same bad request. (#65631)
_err_type = getattr(chunk, "error_type", None) _err_type = getattr(chunk, "error_type", None)
_err_msg = getattr(chunk, "error_message", None) _err_msg = getattr(chunk, "error_message", None)
if _err_type or _err_msg: if _err_type or _err_msg:
@@ -2806,6 +2852,7 @@ class _StreamingCall:
raise ProviderStreamError(status_code=_status, body=body, raw_text=f"{_err_type}: {_err_msg}") raise ProviderStreamError(status_code=_status, body=body, raw_text=f"{_err_type}: {_err_msg}")
# Nous Portal usage frames (choices=[] + lastOne=true, no [DONE]) are a # Nous Portal usage frames (choices=[] + lastOne=true, no [DONE]) are a
# clean terminal, not a drop; relabelled upstreams send 1 / "true". # clean terminal, not a drop; relabelled upstreams send 1 / "true".
# See #90848.
last_one = getattr(chunk, "lastOne", None) last_one = getattr(chunk, "lastOne", None)
if last_one is None and isinstance(getattr(chunk, "model_extra", None), dict): if last_one is None and isinstance(getattr(chunk, "model_extra", None), dict):
last_one = chunk.model_extra.get("lastOne") last_one = chunk.model_extra.get("lastOne")
+28 -1
View File
@@ -117,6 +117,13 @@ class ClientLifecycleMixin:
``close()`` releases FDs from the calling thread while other threads may still hold the fd in an SSL BIO; ``close()`` releases FDs from the calling thread while other threads may still hold the fd in an SSL BIO;
a recycled fd then gets a TLS record written into an unrelated file (SQLite-header corruption). a recycled fd then gets a TLS record written into an unrelated file (SQLite-header corruption).
The shared primary client has no single owning thread — worker threads from stale-killed attempts
may still be unwinding their SSL BIOs, and the codex-direct / MoA paths stream on the shared client
itself. If we release an FD while another thread's SSL layer still caches the raw integer fd, the
kernel can recycle it into an unrelated ``open()`` (e.g. ``kanban.db``) and the unwinding TLS flush
then writes an application-data record into that file — the SQLite-header corruption documented in
#29507/#70773.
""" """
if client is None: if client is None:
return return
@@ -134,6 +141,12 @@ class ClientLifecycleMixin:
The worker may be blocked in an OpenSSL read; hard-closing from the timeout thread releases FDs under a The worker may be blocked in an OpenSSL read; hard-closing from the timeout thread releases FDs under a
live BIO (native corruption / SIGSEGV). Only ``shutdown()`` so the read sees EOF and the worker closes itself. live BIO (native corruption / SIGSEGV). Only ``shutdown()`` so the read sees EOF and the worker closes itself.
See #94248.
A delegation deadline abandons this agent's daemon worker while it may still be blocked inside an
in-flight OpenSSL ``read`` (Codex Responses stream, httpx request). This helper only ``shutdown()``s
pooled sockets (safe from any thread), settling blocked reads with EOF/EPIPE so the worker can
unwind and run the real close from its own thread. See #70773, #94248.
""" """
drained = 0 drained = 0
# Shared primary client (codex-direct / MoA stream on it directly). # Shared primary client (codex-direct / MoA stream on it directly).
@@ -194,6 +207,11 @@ class ClientLifecycleMixin:
return False return False
self.client = new_client self.client = new_client
# Never hard-close the replaced shared client (another thread may still be unwinding on the old pool). # Never hard-close the replaced shared client (another thread may still be unwinding on the old pool).
# #70773: never hard-close the replaced shared client from here — the caller may not be the thread
# whose request is still unwinding on the old pool (credential rotation and dead-connection cleanup
# run on the turn thread while stale-killed workers unwind; the codex-direct path streams on the
# shared client itself). Retire it instead: sockets are shut down (FD-safe), FD release deferred to
# GC.
self._retire_shared_openai_client(old_client, reason=f"replace:{reason}") self._retire_shared_openai_client(old_client, reason=f"replace:{reason}")
return True return True
@@ -310,6 +328,10 @@ class ClientLifecycleMixin:
try: try:
shutdown_count = self._force_close_tcp_sockets(client) shutdown_count = self._force_close_tcp_sockets(client)
# Zero sockets shut down means the worker stays blocked — WARN, not success. # Zero sockets shut down means the worker stays blocked — WARN, not success.
# tcp_force_closed=0 means the stranger-thread abort found no sockets to shut down — the worker
# stays blocked in recv and the provider keeps the slot (#72975). Surface that as WARNING so it
# cannot be mistaken for a successful abort in the logs.
# See #72975.
_log = logger.warning if shutdown_count == 0 else logger.info _log = logger.warning if shutdown_count == 0 else logger.info
_log( _log(
"%s client aborted (%s, shared=False, tcp_force_closed=%d, deferred_close=stranger_thread) %s%s", "%s client aborted (%s, shared=False, tcp_force_closed=%d, deferred_close=stranger_thread) %s%s",
@@ -572,7 +594,12 @@ class ClientLifecycleMixin:
def _try_refresh_env_client_credentials(self) -> bool: def _try_refresh_env_client_credentials(self) -> bool:
"""Adopt ``~/.hermes/.env`` credential/base-url edits at the turn boundary (a Settings save updates ``.env`` """Adopt ``~/.hermes/.env`` credential/base-url edits at the turn boundary (a Settings save updates ``.env``
but a live worker keeps init-time values). Adoption rule: ``_should_adopt_env_credentials``.""" but a live worker keeps init-time values). Adoption rule: ``_should_adopt_env_credentials``.
Covers api-key registry providers and named custom providers with a ``key_env`` (#67935) — the
latter resolve to ``provider="custom"`` with no registry entry, so they are matched through the
runtime provider's config lookup instead.
"""
if self.api_mode != "chat_completions" or getattr(self, "_fallback_activated", False): if self.api_mode != "chat_completions" or getattr(self, "_fallback_activated", False):
return False return False
resolved = self._resolve_env_credentials() resolved = self._resolve_env_credentials()
+52 -3
View File
@@ -426,10 +426,38 @@ def _chat_messages_to_responses_input(
``native_compaction_eligible``: THIS request carries ``context_management``; gates both replaying ``compaction`` ``native_compaction_eligible``: THIS request carries ``context_management``; gates both replaying ``compaction``
checkpoints and ``prune_pre_checkpoint_items``. Checkpoints persist across model swaps / compression flips / resume, checkpoints and ``prune_pre_checkpoint_items``. Checkpoints persist across model swaps / compression flips / resume,
so without the gate one checkpoint would erase pre-checkpoint history on a model that cannot decrypt it (lossless: so without the gate one checkpoint would erase pre-checkpoint history on a model that cannot decrypt it (lossless:
local history is never truncated).""" local history is never truncated).
Earlier (PR #26644, May 2026) we believed xAI's OAuth/SuperGrok ``/v1/responses`` surface rejected
replayed ``encrypted_content`` reasoning items minted by prior turns, and we stripped them. That
decision was wrong — xAI explicitly relies on Hermes threading encrypted reasoning back across turns for
cross-turn coherence (the whole point of their partnership integration). We now replay encrypted
reasoning on every Responses transport (xAI, native Codex, custom relays) and let xAI tell us explicitly
if a specific surface ever rejects a payload.
The Copilot backend (api.githubcopilot.com/responses) binds these ids to a specific backend "connection"
— credential-pool rotation, a gateway restart, or routine load-balancer churn between turns all
invalidate it — and rejects a stale id with HTTP 401 "input item ID does not belong to this connection"
even for short ids (see #32716). ``phase``/ ``status``/``content`` are still replayed; only ``id`` is
unsafe to reuse across a Copilot connection.
``native_compaction_eligible`` mirrors, for THIS request, the decision made by
``native_compaction.native_compaction_context_management`` — it is True only when that gate returned a
payload, i.e. when the request actually carries ``context_management``. It controls two things that must
never outlive the gate: replaying ``type: "compaction"`` checkpoint items, and restructuring the wire
around them (``prune_pre_checkpoint_items``). Checkpoints are persisted in the ``codex_reasoning_items``
sidecar and survive a mid-session model swap, a ``compression.enabled: false`` flip, the rejection kill
switch and a resumed session; without this flag a single captured checkpoint would keep deleting every
pre-checkpoint item from every later request, on a model that cannot decrypt the blob (#85914). Default
False = pre-feature wire, which is also correct for every caller that never sends ``context_management``
(auxiliary/compression client, ad-hoc ``convert_messages``). Dropping the checkpoint costs nothing:
Hermes' local history is never truncated by native compaction, so the full conversation is still on the
wire.
"""
items: List[Dict[str, Any]] = [] items: List[Dict[str, Any]] = []
# Parallel to ``items``: source chat message per item. Pruning reads a summary # Parallel to ``items``: source chat message per item. Pruning reads a summary
# carrier's provenance from the source; the converted item may be a lossy shape. # carrier's provenance from the source; the converted item may be a lossy shape.
# Pruning needs this to read a canonical summary carrier's up-to-date, provenance-tagged content
# directly — the converted `item` can be a lossy shape (stale exact-replay, or a typed
# `function_call_output` wrapper) that no longer carries it (#90976).
item_sources: List[Optional[Dict[str, Any]]] = [] item_sources: List[Optional[Dict[str, Any]]] = []
seen_item_ids: set = set() seen_item_ids: set = set()
def emit(new_items: List[Dict[str, Any]], msg: Dict[str, Any]) -> None: def emit(new_items: List[Dict[str, Any]], msg: Dict[str, Any]) -> None:
@@ -470,6 +498,17 @@ def _chat_messages_to_responses_input(
# The server renders nothing placed before a compaction item, so pre-checkpoint history is # The server renders nothing placed before a compaction item, so pre-checkpoint history is
# dead weight and plaintext asks / merged summaries silently vanish. Keep the newest checkpoint # dead weight and plaintext asks / merged summaries silently vanish. Keep the newest checkpoint
# first, retain pre-checkpoint USER and SUMMARY messages within a token budget, leave the tail. # first, retain pre-checkpoint USER and SUMMARY messages within a token budget, leave the tail.
# Native server-side compaction: when a replayed checkpoint is present, restructure the wire around it.
# Gated on the CURRENT request's native eligibility, not merely on the presence of a checkpoint: a
# persisted checkpoint outlives the gate, and pruning for a request that carries no
# ``context_management`` deletes history the server never compacted. ``item_sources`` (parallel to
# ``items``) carries the raw chat message each converted item came from. A canonical summary carrier's
# content can be lost or gone stale by the time it becomes a Responses item — a merge-into-tail
# tool-result carrier becomes a typed ``function_call_output`` (no ``content``/``role`` at all), and a
# merge-into-tail assistant carrier can be shadowed by a stale exact ``codex_message_items`` replay from
# before the merge rewrote its content. Pruning reads the source message's own up-to-date,
# provenance-tagged content directly instead of trying to recover it from whatever shape the conversion
# produced (#90976).
if not native_compaction_eligible: if not native_compaction_eligible:
return items return items
from agent.native_compaction import prune_pre_checkpoint_items from agent.native_compaction import prune_pre_checkpoint_items
@@ -478,7 +517,12 @@ def _chat_messages_to_responses_input(
class ResponsesRouteFlags(NamedTuple): class ResponsesRouteFlags(NamedTuple):
"""Which special Responses-API route an agent is talking to. Single owner of the """Which special Responses-API route an agent is talking to. Single owner of the
codex/xai/github predicates — every site must call :func:`classify_responses_route`.""" codex/xai/github predicates — every site must call :func:`classify_responses_route`.
Every site that needs these flags (request kwargs build, preflight estimation, silent- reject hints)
must call :func:`classify_responses_route` instead of re-implementing the string comparisons inline —
inline copies drift (backend-identity class: #22548/#70893/#59561/#72468).
"""
is_codex_backend: bool is_codex_backend: bool
is_xai_responses: bool is_xai_responses: bool
is_github_responses: bool is_github_responses: bool
@@ -505,7 +549,12 @@ def estimate_native_responses_preflight_tokens(
agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None, agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None,
) -> Optional[int]: ) -> Optional[int]:
"""Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted """Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted
session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails.""" session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails.
Automatic preflight previously counted the full durable transcript. On a natively compacted Codex
session that overstates the wire by several times and fires local compression against history the main
request will never send (#96155).
"""
if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list): if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list):
return None return None
route = classify_responses_route(agent)._asdict() route = classify_responses_route(agent)._asdict()
+8
View File
@@ -307,6 +307,9 @@ def make_codex_app_server_event_bridge(agent) -> Callable[[dict], None]:
def _fire_delta(params: dict, attr: str) -> None: def _fire_delta(params: dict, attr: str) -> None:
text = params.get("delta") or params.get("text") or "" text = params.get("delta") or params.get("text") or ""
# Single-writer guard (#65991): a superseded stream must not pollute the turn's accumulated text
# (which also feeds the interim-visible-text de-dup comparison), even when a caller reaches this
# directly (the tool-suppressed content path) rather than through _fire_stream_delta.
if isinstance(text, str) and text: if isinstance(text, str) and text:
agent_cb(attr, f"{attr} raised", args=(text,)) agent_cb(attr, f"{attr} raised", args=(text,))
@@ -380,6 +383,11 @@ def _ensure_codex_session(agent) -> None:
auto_approve_requests = is_approval_bypass_active() auto_approve_requests = is_approval_bypass_active()
except Exception: except Exception:
logger.debug("codex app-server: approval-bypass lookup failed; keeping fail-closed default", exc_info=True) logger.debug("codex app-server: approval-bypass lookup failed; keeping fail-closed default", exc_info=True)
# Bridge codex JSON-RPC notifications (item/started, item/completed, item/agentMessage/delta, ...) into
# Hermes' gateway UI callbacks (tool_progress_callback, _fire_stream_delta,
# _emit_interim_assistant_message). Without this, Discord/Telegram users see no live tool-progress or
# interim commentary while codex_app_server is running — only the final answer (#33200). Supersedes the
# narrower item/started-only bridge from #38835.
agent._codex_session = CodexAppServerSession( agent._codex_session = CodexAppServerSession(
cwd=getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()), approval_callback=approval_callback, cwd=getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()), approval_callback=approval_callback,
request_routing=_ServerRequestRouting(auto_approve_exec=auto_approve_requests, auto_approve_apply_patch=auto_approve_requests), request_routing=_ServerRequestRouting(auto_approve_exec=auto_approve_requests, auto_approve_apply_patch=auto_approve_requests),
+12
View File
@@ -11,6 +11,18 @@ _COMPACTION_INTERNAL_FIELDS = (
"tool_calls", "tool_calls",
"finish_reason", "finish_reason",
"reasoning", "reasoning",
# Provider replay/metadata fields that ride the wire on every request but are invisible to
# ``msg["content"]``/``msg["tool_calls"]`` accounting. Codex Responses sessions in particular carry
# ``codex_reasoning_items`` blobs of ``encrypted_content`` that can dominate the serialized session (a
# measured 214-turn session held ~115K tokens / 27% of its payload there — #55572).
# ``reasoning_details`` is handled separately (see ``_reasoning_details_text_chars``): its signed/base64
# envelope is excluded from the budget, mirroring the preflight estimator's exclusion in
# ``model_metadata._estimate_message_tokens_without_images`` (#73298).
# An assistant turn may carry only reasoning/thinking content with no visible text (extended-thinking
# turns, thinking-only recovery responses). Such a turn is persisted with its reasoning fields and is
# recallable from the transcript, but dropping it here as "empty" makes it vanish from the
# resumed/reloaded session view while the desktop's reasoning disclosure has nothing to render. Keep it
# when it carries reasoning so the "Thinking…" block still shows. (#44022)
"reasoning_content", "reasoning_content",
"reasoning_details", "reasoning_details",
"codex_reasoning_items", "codex_reasoning_items",
+21 -1
View File
@@ -114,6 +114,14 @@ def _run_under_progress_timeout(
from agent.conversation_compression import CompressionCommitFence, run_compress_context_with_progress_timeout from agent.conversation_compression import CompressionCommitFence, run_compress_context_with_progress_timeout
def _snapshot_worker(fence=None): def _snapshot_worker(fence=None):
# #76354 review F3: the pooled worker must NEVER share the caller's live transcript. Plugin/legacy
# context engines are allowed to mutate their input list in place; after a host timeout the worker
# stays alive, so a shared list would let a late engine rewrite the live conversation (roles,
# ordering, persisted content) behind the caller's back. Deep-snapshot here, on the worker thread,
# so the caller's list object is never touched by pooled code. Results are published to
# caller-visible state only via the returned value of an ADMITTED commit (the host discards results
# on timeout/cancel); durable SessionDB mutation is already gated behind the commit fence inside
# compress_context.
snapshot = copy.deepcopy(messages) snapshot = copy.deepcopy(messages)
result_msgs, result_prompt = run(fence, target_messages=snapshot) result_msgs, result_prompt = run(fence, target_messages=snapshot)
return (messages if result_msgs is snapshot else result_msgs), result_prompt return (messages if result_msgs is snapshot else result_msgs), result_prompt
@@ -185,9 +193,18 @@ class CompressionFacadeMixin:
) -> tuple: ) -> tuple:
"""Forwarder — see ``agent.conversation_compression.compress_context``. """Forwarder — see ``agent.conversation_compression.compress_context``.
``force=True`` (manual /compress) bypasses the summary-failure cooldown; ``bypass_cooldown=True`` ``force=True`` (manual /compress) bypasses the summary-failure cooldown; ``bypass_cooldown=True``
(provider-proven overflow recovery) runs one real attempt while the cooldown stays armed.""" (provider-proven overflow recovery) runs one real attempt while the cooldown stays armed.
``force=True`` is passed by the manual ``/compress`` slash command so users can bypass the
summary-failure cooldown after an auto-compress abort. Auto-compress callers use the default
``force=False``. See #100661.
"""
# Per-attempt timeout signal for turn-start preflight and in-loop consumers: a stalled # Per-attempt timeout signal for turn-start preflight and in-loop consumers: a stalled
# compression must not be mistaken for a structural no-op. Thread-local + per-agent lock. # compression must not be mistaken for a structural no-op. Thread-local + per-agent lock.
# A stalled compression must not be mistaken for a structural no-op and followed by the oversized
# provider request it was meant to prevent. The typed helper upgrades the simple attribute to
# thread-local state guarded by a per-agent lock so overlapping automatic/manual entrypoints cannot
# clobber each other's outcome (#98741).
from agent.conversation_compression import ( from agent.conversation_compression import (
CompressionCommitFence, compress_context, reset_context_compression_timeout_outcome, CompressionCommitFence, compress_context, reset_context_compression_timeout_outcome,
resolve_context_compression_timeouts, resolve_context_compression_timeouts,
@@ -206,6 +223,9 @@ class CompressionFacadeMixin:
root = self._conversation_root_id() root = self._conversation_root_id()
if root: if root:
token = set_conversation_context(root) token = set_conversation_context(root)
# Initialized alongside `token`: the turn-lease timeout/interrupt early returns leave the try block
# before set_affinity_scope() runs, and the finally reads this name unconditionally
# (UnboundLocalError otherwise — the 4 red cross-process lease tests on PR #97158).
affinity_token = None affinity_token = None
if get_affinity_scope() is None: if get_affinity_scope() is None:
declared = declared_conversation_scope_safe(self) declared = declared_conversation_scope_safe(self)
+285 -9
View File
@@ -46,6 +46,19 @@ def _safe_int(value: Any) -> int | None:
# summary sees it while the detached stalled worker does not. A stall raises nothing, so the aux client's # summary sees it while the detached stalled worker does not. A stall raises nothing, so the aux client's
# exception-path fallback never fires; the host pins a fallback route for exactly ONE retry (the sole aux # exception-path fallback never fires; the host pins a fallback route for exactly ONE retry (the sole aux
# call per compaction). The main-model retry must NOT re-issue the pin. # call per compaction). The main-model retry must NOT re-issue the pin.
# ── Pinned summary route ───────────────────────────────────────────────── The summary call normally
# resolves its provider/model from ``auxiliary.compression``. One caller needs to override that for a single
# attempt: after the host's progress-aware timeout aborts a stalled summary (#78981),
# ``agent.conversation_compression`` re-runs compression with the route pinned to a configured
# ``fallback_chain`` entry. Nothing raised out of the stalled call, so the auxiliary client's own fallback
# handling — which only runs from its exception path — never saw that failure. A ContextVar, not an
# attribute on the compressor: the aborted worker is detached and still alive on the pool, and the
# compressor object is shared with it. Context is copied per worker (``propagate_context_to_thread``), so
# the pin reaches the retry's whole synchronous call chain and cannot leak into the stalled attempt or any
# unrelated auxiliary call. Coverage is the single ``_generate_summary`` LLM call only. That is one call per
# compression run (its only non-recursive call site is the compress path; the two recursive calls are the
# deliberate main-model retry that must NOT re-issue the pin). The summary call is the ONLY auxiliary LLM
# call a lean compaction attempt makes (#96603) — there are no sibling digest calls.
_SUMMARY_ROUTE_PIN: contextvars.ContextVar[Optional[Dict[str, Any]]] = ( _SUMMARY_ROUTE_PIN: contextvars.ContextVar[Optional[Dict[str, Any]]] = (
contextvars.ContextVar("hermes_summary_route_pin", default=None) contextvars.ContextVar("hermes_summary_route_pin", default=None)
) )
@@ -93,7 +106,10 @@ _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS: tuple[str, ...] = (
def _is_hygiene_preagent_only_cooldown(error: object) -> bool: def _is_hygiene_preagent_only_cooldown(error: object) -> bool:
"""Return True for a cooldown that belongs only to pre-agent hygiene. """Return True for a cooldown that belongs only to pre-agent hygiene.
Hygiene watchdog timeouts / turn-hold deferrals are not evidence of an auxiliary-model failure and Hygiene watchdog timeouts / turn-hold deferrals are not evidence of an auxiliary-model failure and
must never block the in-agent compressor.""" must never block the in-agent compressor.
See #74136, #86972.
"""
text = str(error or "").strip().casefold() text = str(error or "").strip().casefold()
return any(marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS) return any(marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS)
@@ -114,6 +130,10 @@ def _response_finish_reason(response: Any) -> str:
# Marker for a length-stopped (PARTIAL) summary; the except-branch classifier keys # Marker for a length-stopped (PARTIAL) summary; the except-branch classifier keys
# on this exact substring, so keep raise sites and classifier in sync. # on this exact substring, so keep raise sites and classifier in sync.
# RuntimeError marker raised when the summarizer's generation stopped on the output-token cap
# (``finish_reason == "length"``). A length stop means the summary text is PARTIAL — persisting it as a
# compaction checkpoint would silently truncate the conversation's memory and feed the cut-off text back
# into every subsequent iterative-update prompt. (Ported from earendil-works/pi#7048 / commit 97fa14e39.)
_TRUNCATED_SUMMARY_MARKER = "finish_reason=length" _TRUNCATED_SUMMARY_MARKER = "finish_reason=length"
@@ -123,6 +143,11 @@ def _is_summary_access_or_quota_error(exc: Exception) -> bool:
# No active secret scope is a missing-credential failure of our own making; # No active secret scope is a missing-credential failure of our own making;
# classify as credential so compress() preserves the session unchanged. # classify as credential so compress() preserves the session unchanged.
try: try:
# A credential read that failed closed because no profile secret scope was active (multiplexed
# gateway, worker thread without the caller's ContextVars) is a missing-credential failure of our
# own making: the summary model cannot be reached until the spawn site is fixed, and a placeholder
# summary would only destroy the middle window for nothing. Classify it with the credential class so
# compress() preserves the session unchanged (#100849 bundle: every hygiene pass truncated).
from agent.secret_scope import UnscopedSecretError from agent.secret_scope import UnscopedSecretError
except Exception: # pragma: no cover - import guard except Exception: # pragma: no cover - import guard
UnscopedSecretError = () # type: ignore[assignment] UnscopedSecretError = () # type: ignore[assignment]
@@ -150,6 +175,12 @@ HISTORICAL_TASK_HEADING = "## Historical Task Snapshot"
SUMMARY_PREFIX = ( SUMMARY_PREFIX = (
# Jul 2026 (#65848 class): identical to the pre-#69619 prefix except it lacked the explicit "tools
# remain fully active" clause — the strong REFERENCE ONLY framing bled into general tool-use suppression
# (observed: 7 consecutive narration-only turns immediately after a compression event on a production
# deployment).
# Carveout era (#41607/#38364/#42812): "consistent → use as background" licensed stale-task resumption
# on topic overlap.
"[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted " "[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted "
"into the summary below. This is a handoff from a previous context " "into the summary below. This is a handoff from a previous context "
"window — treat it as background reference, NOT as active instructions. " "window — treat it as background reference, NOT as active instructions. "
@@ -193,6 +224,21 @@ COMPRESSED_SUMMARY_HAS_USER_TURN_KEY = "_compressed_summary_has_user_turn"
# Only micro markers may be superseded/defragged/rehydrated: a batch marker's # Only micro markers may be superseded/defragged/rehydrated: a batch marker's
# content is NOT in the rolling micro summary, so rewriting one destroys history. # content is NOT in the rolling micro summary, so rewriting one destroys history.
MICRO_COMPACT_MARKER_KEY = "_micro_compact_marker" MICRO_COMPACT_MARKER_KEY = "_micro_compact_marker"
# Intrinsic marker stamped on a message dict once it has been written to the SQLite session store. Used by
# ``_flush_messages_to_session_db`` to decide what is already durable. An object-identity (``id(msg)``)
# dedup set cannot be trusted across turns: once a flushed message dict is dropped from the live list (e.g.
# by scaffolding rewind or in-place compaction) and garbage- collected, CPython is free to hand its address
# to a brand-new assistant/tool message, whose ``id()`` then collides with the stale entry and the real turn
# is silently never persisted. A marker bound to the dict itself cannot be aliased that way. The ``_``
# prefix is mandatory: the wire sanitizers (agent/transports/chat_completions.py,
# agent/chat_completion_helpers.py) strip every top-level ``_``-prefixed key before the request leaves the
# process, so this never reaches a strict OpenAI-compatible gateway. CONTRACT (#92231): the marker asserts
# "this dict's CONTENT is durable as written". Loaded rows are stamped at materialization time
# (hermes_state._rows_to_conversation), so any code that mutates a loaded or flushed dict's content in place
# and needs the change persisted MUST pop the marker (and invalidate _db_flush_scan_prefix if the dict may
# sit inside the bounded-scan prefix) — see agent/turn_finalizer.py (fill-empty-tail) and
# agent/context_compressor.py (micro-compaction defrag) for the two canonical pop sites. Mutating without
# popping leaves the DB silently stale.
_DB_PERSISTED_MARKER = "_db_persisted" _DB_PERSISTED_MARKER = "_db_persisted"
# Carried-forward tail rows archive as rewind-style (active=0, compacted=0) so # Carried-forward tail rows archive as rewind-style (active=0, compacted=0) so
# they don't duplicate live copies in recall; never persisted (unknown column). # they don't duplicate live copies in recall; never persisted (unknown column).
@@ -630,6 +676,12 @@ def _is_clarify_non_response_sentinel(response: Any) -> bool:
# Ghost-skill defense: the ONE canonical prune marker; emit sites and presence # Ghost-skill defense: the ONE canonical prune marker; emit sites and presence
# checks must use the same string so they cannot drift. # checks must use the same string so they cannot drift.
# Ghost-skill defense (#32106): when compaction reduces an old ``skill_view`` result to a 1-line metadata
# summary, the model still believes the skill is loaded even though its instructions are gone. The marker
# below is the ONE canonical prune signal — ``_skill_pruned_marker()`` builds it and every presence check
# matches against the same string, so the emit side and the check side can never drift apart (the original
# PR #44166 emitted ``[SKILL_PRUNED:`` but presence-checked ``[SKILL_PRUNED]``, making re-injection fire
# even when the marker had survived).
SKILL_PRUNED_MARKER_PREFIX = "[SKILL_PRUNED:" SKILL_PRUNED_MARKER_PREFIX = "[SKILL_PRUNED:"
# Small skill_view results stay verbatim; shared by emit site and summarizer scan. # Small skill_view results stay verbatim; shared by emit site and summarizer scan.
_SKILL_VIEW_PRUNE_MIN_CHARS = 5000 _SKILL_VIEW_PRUNE_MIN_CHARS = 5000
@@ -783,6 +835,13 @@ def _build_recovery_footer(session_id: str, region_len: int) -> str:
# Detailed session log comes from the SAME single summary request (one aux LLM # Detailed session log comes from the SAME single summary request (one aux LLM
# call per attempt); coverage via input sampling, exact needles via anchor index. # call per attempt); coverage via input sampling, exact needles via anchor index.
# One flat 2-3K-token summary cannot carry a 400K+ region's specifics — the eval showed recall collapsing to
# ~33% when the big tail (which accidentally archived restated facts) shrank. The detailed,
# identifier-preserving session log is produced by the SAME single summary request as the narrative summary
# (one auxiliary LLM call per compaction attempt, total — #96603: the earlier per-chunk digest loop made up
# to 28 extra aux calls and pushed compactions to 7-11 minutes on slow aux routes). Coverage over oversized
# regions comes from even input sampling (see ``_sample_summary_input``), and exact-needle defense comes
# from the LLM-free anchor index below.
_LEAN_SESSION_LOG_HEADING = "## Detailed Session Log (oldest first)" _LEAN_SESSION_LOG_HEADING = "## Detailed Session Log (oldest first)"
# Extra output-token guidance for the session-log section (single response). # Extra output-token guidance for the session-log section (single response).
_LEAN_SESSION_LOG_BUDGET_TOKENS = 4_000 _LEAN_SESSION_LOG_BUDGET_TOKENS = 4_000
@@ -852,6 +911,8 @@ def _build_anchor_index(turns: List[Dict[str, Any]]) -> str:
# Message-count window (distinct from the token-based tail boundary) in which a # Message-count window (distinct from the token-based tail boundary) in which a
# just-loaded skill_view body must survive the Phase-1 prune. # just-loaded skill_view body must survive the Phase-1 prune.
# A skill_view call within this many trailing messages counts as "just loaded": its full instruction body
# must survive the Phase-1 prune even when the token-budget boundary would otherwise demote it (#32106).
_SKILL_PRUNE_RECENT_WINDOW = 10 _SKILL_PRUNE_RECENT_WINDOW = 10
@@ -912,12 +973,15 @@ _MAX_TAIL_MESSAGE_FLOOR = 8
# Skip the LLM call when the compressible middle is below this fraction of the # Skip the LLM call when the compressible middle is below this fraction of the
# threshold (and a prior ineffectiveness strike exists); dropping alone suffices. # threshold (and a prior ineffectiveness strike exists); dropping alone suffices.
# See #60451.
_FEASIBILITY_SKIP_MIDDLE_FRACTION = 0.10 _FEASIBILITY_SKIP_MIDDLE_FRACTION = 0.10
# Under pressure, demote large tool outputs even inside the protected region but # Under pressure, demote large tool outputs even inside the protected region but
# keep this many trailing messages verbatim. # keep this many trailing messages verbatim.
_PRESSURE_KEEP_RECENT_MESSAGES = 3 _PRESSURE_KEEP_RECENT_MESSAGES = 3
# Newest image-bearing tool results kept verbatim; older image payloads retire # Newest image-bearing tool results kept verbatim; older image payloads retire
# even inside protect_last_n (matches the Anthropic adapter's keep-window). # even inside protect_last_n (matches the Anthropic adapter's keep-window).
# Native vision_analyze / computer_use screenshots that sit inside the protected tail cannot be demoted by
# pass 2, so they ride every later request until anti-thrash disables compression (#92699).
_MAX_KEEP_TOOL_IMAGES = 3 _MAX_KEEP_TOOL_IMAGES = 3
# Below this window the threshold is floored (raise-only): at 50% the incompressible # Below this window the threshold is floored (raise-only): at 50% the incompressible
@@ -929,6 +993,8 @@ _SMALL_CTX_THRESHOLD_PERCENT = 0.75
_PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+") _PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+")
# MEDIA directives must not reach the summarizer or they get re-emitted as active. # MEDIA directives must not reach the summarizer or they get re-emitted as active.
# MEDIA delivery directives must not reach the summarizer — if one leaks into the summary, the downstream
# model may re-emit it as an active directive on the next turn, triggering bogus attachment sends (#14665).
_MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+") _MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+")
_HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)") _HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)")
@@ -1079,6 +1145,9 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) -
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
# Charge only thinking TEXT, never the signed/base64 envelope; skip when the # Charge only thinking TEXT, never the signed/base64 envelope; skip when the
# same text already rides in reasoning/reasoning_content. # same text already rides in reasoning/reasoning_content.
# When the same thinking text already rides in ``reasoning``/``reasoning_content`` (measured
# byte-identical on Anthropic-wire sessions), skip it here entirely so the prose is not charged twice on
# top of the envelope exclusion. See #73298.
if not (msg.get("reasoning") or msg.get("reasoning_content")): if not (msg.get("reasoning") or msg.get("reasoning_content")):
tokens += _reasoning_details_text_chars(msg.get("reasoning_details")) // _CHARS_PER_TOKEN tokens += _reasoning_details_text_chars(msg.get("reasoning_details")) // _CHARS_PER_TOKEN
return tokens return tokens
@@ -1256,6 +1325,11 @@ def _strip_historical_media(messages: List[Dict[str, Any]]) -> List[Dict[str, An
# non-empty for the zero-user-turn guard). Rule 2: superseded tool-result image, even in the tail. # non-empty for the zero-user-turn guard). Rule 2: superseded tool-result image, even in the tail.
return ( return (
(0 < anchor and index < anchor) (0 < anchor and index < anchor)
# When the ONLY image-bearing user message is the very first one (``anchor == 0``) and newer
# tool-result images exist, the model has moved on — but the opening base64 blob used to survive
# every compaction forever, which is half the wedge in #89938 (the reported session opened with
# a ~200KB poster). When nothing newer exists the opening image IS the newest image and is kept,
# consistent with keep-newest everywhere else.
or (anchor == 0 and index == 0 and tool_anchor > 0) or (anchor == 0 and index == 0 and tool_anchor > 0)
or (message.get("role") == "tool" and index != tool_anchor) or (message.get("role") == "tool" and index != tool_anchor)
) )
@@ -1437,6 +1511,8 @@ def _json_dict(text: Any) -> dict:
"""Parse ``text`` as a JSON object; ``{}`` for empty, invalid, or non-object input.""" """Parse ``text`` as a JSON object; ``{}`` for empty, invalid, or non-object input."""
try: try:
parsed = json.loads(text) if text else {} parsed = json.loads(text) if text else {}
# Just-loaded / actively-referenced skills survive verbatim (#32106). Pass-4 pressure demotion overrides
# this.
except (json.JSONDecodeError, TypeError): except (json.JSONDecodeError, TypeError):
return {} return {}
return parsed if isinstance(parsed, dict) else {} return parsed if isinstance(parsed, dict) else {}
@@ -1484,6 +1560,10 @@ def _memory_provider_section(memory_context: str) -> str:
def _today_for_prompt() -> str: def _today_for_prompt() -> str:
"""Date-only (user tz) for temporal anchoring; "" when the clock fails. Cache-safe: the summary is outside the prefix.""" """Date-only (user tz) for temporal anchoring; "" when the clock fails. Cache-safe: the summary is outside the prefix."""
try: try:
# Date-only granularity matches system_prompt.py:337 (PR #20451) and the user's configured timezone
# via hermes_time.now(). The compaction summary is a mid-conversation message that is NOT part of
# the cached prefix, so a date here never affects prompt-cache stability. Resolved defensively — a
# clock failure must never block compaction.
from hermes_time import now as _hermes_now from hermes_time import now as _hermes_now
return _hermes_now().strftime("%Y-%m-%d") return _hermes_now().strftime("%Y-%m-%d")
except Exception: # pragma: no cover - clock resolution is best-effort except Exception: # pragma: no cover - clock resolution is best-effort
@@ -1724,7 +1804,17 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
"""Clear all per-session compaction state at a real session boundary. """Clear all per-session compaction state at a real session boundary.
Session end (CLI exit, gateway expiry, id rotation) — NOT /new or /reset. Every per-session Session end (CLI exit, gateway expiry, id rotation) — NOT /new or /reset. Every per-session
flag/counter can contaminate the next live session (suppressed compression, stale cooldowns, flag/counter can contaminate the next live session (suppressed compression, stale cooldowns,
misleading warnings), so the whole surface is reset here.""" misleading warnings), so the whole surface is reset here.
Session end (CLI exit, gateway expiry, session-id rotation) goes through this method rather than
``on_session_reset()`` (/new, /reset). The original fix (#38788) only cleared ``_previous_summary``,
but the same cross-session contamination risk applies to every per-session variable that
``on_session_reset()`` clears: stale ``_ineffective_compression_count`` can suppress compression in
a subsequent live session; ``_summary_failure_cooldown_until`` can block summary generation;
``_last_compress_aborted`` can make callers think compression is still aborted;
``_last_aux_model_failure_*`` can surface stale error warnings; ``_last_summary_dropped_count`` /
``_last_summary_fallback_used`` can produce misleading user warnings.
"""
self._reset_session_compaction_state() self._reset_session_compaction_state()
def _reset_real_usage_pairing(self) -> None: def _reset_real_usage_pairing(self) -> None:
@@ -1880,7 +1970,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._durable_write("set_compression_ineffective_count", "compression ineffective count", self._ineffective_compression_count) self._durable_write("set_compression_ineffective_count", "compression ineffective count", self._ineffective_compression_count)
def _load_anti_thrash_recovery_deadline(self) -> None: def _load_anti_thrash_recovery_deadline(self) -> None:
"""Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed.""" """Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed.
See #100185.
"""
self._load_durable("_anti_thrash_recovery_deadline", "get_compression_recovery_deadline", "compression recovery deadline", float, 0.0) self._load_durable("_anti_thrash_recovery_deadline", "get_compression_recovery_deadline", "compression recovery deadline", float, 0.0)
def _set_anti_thrash_recovery_deadline(self, deadline: float) -> None: def _set_anti_thrash_recovery_deadline(self, deadline: float) -> None:
@@ -1924,6 +2017,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._verify_compaction_cleared_threshold = True self._verify_compaction_cleared_threshold = True
if feasibility_skip: if feasibility_skip:
# A pre-LLM feasibility skip is not a summary-quality verdict: it must neither extend nor reset the streak. # A pre-LLM feasibility skip is not a summary-quality verdict: it must neither extend nor reset the streak.
# A deliberate pre-LLM feasibility skip (#60451) is not a summary-quality verdict: it must
# neither extend a fallback streak (two skips would otherwise latch the >= 2 breaker and disable
# compression entirely — including the cheap deterministic dropping the skip exists to reach)
# nor reset one (a skip proves nothing about the summary model's health).
if not self.quiet_mode: if not self.quiet_mode:
logger.info( logger.info(
"Compaction completed via pre-LLM feasibility skip; fallback_compression_streak unchanged (%d)", "Compaction completed via pre-LLM feasibility skip; fallback_compression_streak unchanged (%d)",
@@ -1980,6 +2077,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
return None return None
# Hygiene-only cooldowns share the column but are not a 429/aux fault; the in-agent compressor may run. # Hygiene-only cooldowns share the column but are not a 429/aux fault; the in-agent compressor may run.
# A hygiene write may have overwritten an aux-model row; drop the in-memory cooldown too. # A hygiene write may have overwritten an aux-model row; drop the in-memory cooldown too.
# Hygiene watchdog timeouts and turn-hold deferrals persist the same column so the pre-agent pass
# can skip (#74136), but they are not evidence of a 429/aux-model fault. The in-conversation
# compressor has its own budget and must still be allowed to run (#86972).
if _is_hygiene_preagent_only_cooldown(state.get("error")): if _is_hygiene_preagent_only_cooldown(state.get("error")):
self._summary_failure_cooldown_until, self._last_summary_error = 0.0, None self._summary_failure_cooldown_until, self._last_summary_error = 0.0, None
return None return None
@@ -1994,6 +2094,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
def _record_compression_failure_cooldown(self, cooldown_seconds: float, error: Optional[str]) -> None: def _record_compression_failure_cooldown(self, cooldown_seconds: float, error: Optional[str]) -> None:
# Never shorten a longer live deadline; record the latest error text only. # Never shorten a longer live deadline; record the latest error text only.
self._summary_failure_cooldown_until = max(self._summary_failure_cooldown_until, time.monotonic() + float(cooldown_seconds)) self._summary_failure_cooldown_until = max(self._summary_failure_cooldown_until, time.monotonic() + float(cooldown_seconds))
# A later stall or timeout records the latest error text but keeps the later of the two clocks. See
# #96775.
self._last_summary_error = error self._last_summary_error = error
cooldown_until = time.time() + max(0.0, self._summary_failure_cooldown_until - time.monotonic()) cooldown_until = time.time() + max(0.0, self._summary_failure_cooldown_until - time.monotonic())
if not getattr(self, "_session_db", None) or not getattr(self, "_session_id", ""): if not getattr(self, "_session_db", None) or not getattr(self, "_session_id", ""):
@@ -2020,6 +2122,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
def _compression_cancelled(self) -> bool: def _compression_cancelled(self) -> bool:
"""Read the host-owned cooperative cancellation signal, if installed.""" """Read the host-owned cooperative cancellation signal, if installed."""
# #76354 review F4: fence check BEFORE cooldown-clear. A late worker whose host already timed out
# (and recorded a timeout cooldown) must not undo that cooldown when its summary eventually
# succeeds. The hook is installed by compress_context for the duration of the fenced call; when it
# reports cancellation, keep the host's cooldown.
cancelled_check = getattr(self, "_compression_cancelled_check", None) cancelled_check = getattr(self, "_compression_cancelled_check", None)
if not callable(cancelled_check): if not callable(cancelled_check):
return False return False
@@ -2042,6 +2148,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._base_threshold_percent = resolve_model_threshold(model, self.model_thresholds, _config_pct) self._base_threshold_percent = resolve_model_threshold(model, self.model_thresholds, _config_pct)
self.threshold_percent = self._effective_threshold_percent(context_length, self._base_threshold_percent) self.threshold_percent = self._effective_threshold_percent(context_length, self._base_threshold_percent)
# max_tokens=None means "unspecified": keep the existing output reservation. # max_tokens=None means "unspecified": keep the existing output reservation.
# A switch that genuinely changes the output budget passes the new value explicitly. (#43547)
if max_tokens is not None: if max_tokens is not None:
self.max_tokens = self._coerce_max_tokens(max_tokens) self.max_tokens = self._coerce_max_tokens(max_tokens)
self.threshold_tokens = self._compute_threshold_tokens(context_length, self.threshold_percent, self.max_tokens) self.threshold_tokens = self._compute_threshold_tokens(context_length, self.threshold_percent, self.max_tokens)
@@ -2072,6 +2179,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
_MIN_CTX_TRIGGER_RATIO = 0.85 _MIN_CTX_TRIGGER_RATIO = 0.85
# Anti-thrash recovery: after this long blocked, allow ONE probe (counters drop to 1 strike). # Anti-thrash recovery: after this long blocked, allow ONE probe (counters drop to 1 strike).
# Anti-thrash recovery window (#14694): once the ineffective/fallback breaker trips, automatic
# compaction stays blocked for this long, then ONE probe attempt is allowed (counters drop to 1 strike,
# so another ineffective pass re-trips immediately). Long enough that a genuinely incompressible session
# isn't compacting in a loop; short enough that a session which has since grown real compressible
# material recovers well before it rides into the provider's hard context limit.
_ANTI_THRASH_RECOVERY_SECONDS = 300.0 _ANTI_THRASH_RECOVERY_SECONDS = 300.0
# Structural no-op (nothing eligible) is not an ineffective attempt: defer retries instead of striking. # Structural no-op (nothing eligible) is not an ineffective attempt: defer retries instead of striking.
@@ -2109,7 +2221,21 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
) -> int: ) -> int:
"""Compute the compaction trigger in tokens from the effective input budget. """Compute the compaction trigger in tokens from the effective input budget.
Base is ``(context_length - max_tokens) * threshold_percent`` floored at MINIMUM_CONTEXT_LENGTH; Base is ``(context_length - max_tokens) * threshold_percent`` floored at MINIMUM_CONTEXT_LENGTH;
when the floor binds it is capped at 85% of the budget so small windows can still fire.""" when the floor binds it is capped at 85% of the budget so small windows can still fire.
The base value is ``effective_input_budget * threshold_percent``, floored at
``MINIMUM_CONTEXT_LENGTH`` so large-context models don't compress prematurely at 50%. BUT that floor
degenerates at small windows: for a model whose ``context_length`` is at/below the minimum (e.g. a
64K local model), ``max(0.5*64000, 64000) == 64000`` makes the threshold equal the ENTIRE window —
auto-compression can never fire because the provider rejects the request before usage reaches 100%
(#14690).
The provider reserves ``max_tokens`` of output space out of the same window, so the usable INPUT
budget is ``context_length - max_tokens``. With a large ``max_tokens`` (e.g. 65536 on a custom
provider) the input budget is materially smaller than the raw window, and a threshold based on the
full window lets the session hit a provider 400 before compaction fires (#43547). The percentage and
the degenerate-window check below both operate on the effective input budget. ``max_tokens=None``
(provider default) conservatively assumes no reservation (full window).
"""
effective_window = context_length - (max_tokens or 0) effective_window = context_length - (max_tokens or 0)
if effective_window <= 0: if effective_window <= 0:
effective_window = context_length effective_window = context_length
@@ -2166,6 +2292,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
# Usable input = context_length - max_tokens; only a positive int counts as a reservation. # Usable input = context_length - max_tokens; only a positive int counts as a reservation.
self.max_tokens = self._coerce_max_tokens(max_tokens) self.max_tokens = self._coerce_max_tokens(max_tokens)
# True: summary failure aborts (messages unchanged); False: insert deterministic handoff and drop middle. # True: summary failure aborts (messages unchanged); False: insert deterministic handoff and drop middle.
# Output-token reservation: the provider carves max_tokens out of the context window, so the usable
# input budget is context_length - max_tokens. None = provider default => assume no reservation.
# (#43547) Coerce defensively: only a positive int is a real reservation; any other value (None,
# non-numeric, <=0) means "no reservation" so the threshold arithmetic never sees a non-int (e.g. a
# test MagicMock).
self.abort_on_summary_failure = abort_on_summary_failure self.abort_on_summary_failure = abort_on_summary_failure
# Micro-compaction is OFF by default: each pass breaks the prompt-cache prefix every turn. # Micro-compaction is OFF by default: each pass breaks the prompt-cache prefix every turn.
@@ -2173,18 +2304,30 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._reset_micro_compact_cursor_state() self._reset_micro_compact_cursor_state()
self._micro_compact_defrag_threshold_tokens = 2000 self._micro_compact_defrag_threshold_tokens = 2000
# Set when _defrag_rolling_summary pops _DB_PERSISTED_MARKER in place; finalize_turn resets the flush cursor. # Set when _defrag_rolling_summary pops _DB_PERSISTED_MARKER in place; finalize_turn resets the flush cursor.
# Set by _defrag_rolling_summary when it pops _DB_PERSISTED_MARKER from a live dict in place;
# consumed by finalize_turn to invalidate the agent's bounded flush-scan cursor (sibling of the
# #75170 site).
self._flush_scan_cursor_invalidated: bool = False self._flush_scan_cursor_invalidated: bool = False
self._micro_compact_passes = self._micro_compact_tokens_saved_total = self._micro_compact_turns_since_pass = 0 self._micro_compact_passes = self._micro_compact_tokens_saved_total = self._micro_compact_turns_since_pass = 0
# Cadence dial: how often the cache-breaking pass is paid. 1 = every turn. # Cadence dial: how often the cache-breaking pass is paid. 1 = every turn.
self._micro_compact_every_n_turns: int = 1 self._micro_compact_every_n_turns: int = 1
# Deferred: get_model_context_length() may issue a sync HTTP probe that must not block construction. # Deferred: get_model_context_length() may issue a sync HTTP probe that must not block construction.
# Floor and cap are applied on first resolution (see _resolve_context_length / threshold_tokens). # Floor and cap are applied on first resolution (see _resolve_context_length / threshold_tokens).
# The small-context threshold floor and the absolute threshold cap both need the resolved window, so
# they are applied on first resolution (see _resolve_context_length / the threshold_tokens property)
# instead of here. update_model() re-derives the floor for a new window from
# _config_threshold_percent (the raw config value snapshotted above), so switching small -> large
# correctly drops back to the configured value. See #32221.
self._config_context_length = config_context_length self._config_context_length = config_context_length
self._configured_threshold_percent = self.threshold_percent self._configured_threshold_percent = self.threshold_percent
self._resolved_context_length: int | None = None self._resolved_context_length: int | None = None
self._threshold_tokens = self._tail_token_budget = self._max_summary_tokens = None self._threshold_tokens = self._tail_token_budget = self._max_summary_tokens = None
self.compression_count = 0 self.compression_count = 0
# The init log reports resolved budgets; emit it on first resolution to keep construction non-blocking. # The init log reports resolved budgets; emit it on first resolution to keep construction non-blocking.
# The "initialized" log reports resolved token budgets, which would force the deferred
# get_model_context_length() probe to run inside __init__ and re-introduce the exact synchronous
# blocking this change removes (#32221). Emit it on first context-length resolution instead so
# construction stays non-blocking on every path (not just quiet).
self._log_init_summary = not quiet_mode self._log_init_summary = not quiet_mode
self._context_probed = False # True after a step-down from context error self._context_probed = False # True after a step-down from context error
self.last_prompt_tokens = self.last_completion_tokens = 0 self.last_prompt_tokens = self.last_completion_tokens = 0
@@ -2227,6 +2370,16 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._pending_request_rough_tokens = 0 self._pending_request_rough_tokens = 0
# Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count, # Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count,
# not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it. # not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it.
# Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's
# real prompt count for the just-compacted conversation. Effectiveness is "did the prompt get
# under the threshold?", not "did the message list shrink?": compaction can only shrink
# messages, while the system prompt and tool schemas are an incompressible floor (with 50+
# tools, 20-30K tokens — see #14695). When that floor alone meets the threshold, every pass
# shrinks messages by a healthy margin yet leaves the prompt over the line, so the next turn
# compacts again, forever. It must NOT live in should_compress(): that runs twice per turn with
# two different measures (a rough preflight estimate and the real post-response count, #36718),
# and the rough one can dip below the threshold and reset the strike every turn, re-opening the
# loop. Keying on real usage compares like with like and fires exactly once per compaction.
if self._verify_compaction_cleared_threshold: if self._verify_compaction_cleared_threshold:
if self.last_prompt_tokens >= self.threshold_tokens: if self.last_prompt_tokens >= self.threshold_tokens:
self._record_ineffective_compression_verdict(self._ineffective_compression_count + 1) self._record_ineffective_compression_verdict(self._ineffective_compression_count + 1)
@@ -2351,6 +2504,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
# probe by dropping counters to 1 strike (persisted). Deadline is armed lazily and persisted on the row. # probe by dropping counters to 1 strike (persisted). Deadline is armed lazily and persisted on the row.
if self._tripped(): if self._tripped():
# Wall clock: the deadline is persisted so a rebuilt compressor resumes the SAME window. # Wall clock: the deadline is persisted so a rebuilt compressor resumes the SAME window.
# Wall clock, not monotonic: the deadline is persisted on the session row (#100185) so a fresh
# compressor bound to the same session — the gateway rebuilds the AIAgent on every cache
# eviction — resumes the SAME window instead of restarting it. Without that, a blocked messaging
# session never earned its probe and stayed blocked forever.
_now = time.time() _now = time.time()
if self._anti_thrash_recovery_deadline <= 0.0 or ( if self._anti_thrash_recovery_deadline <= 0.0 or (
# Clock jumped backwards: never wait longer than one window from now. # Clock jumped backwards: never wait longer than one window from now.
@@ -2359,6 +2516,20 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._set_anti_thrash_recovery_deadline(_now + self._ANTI_THRASH_RECOVERY_SECONDS) self._set_anti_thrash_recovery_deadline(_now + self._ANTI_THRASH_RECOVERY_SECONDS)
elif _now >= self._anti_thrash_recovery_deadline: elif _now >= self._anti_thrash_recovery_deadline:
self._set_anti_thrash_recovery_deadline(0.0) self._set_anti_thrash_recovery_deadline(0.0)
# Anti-thrashing: back off if recent compressions were ineffective. The back-off must not be
# permanent (#14694): the tripped state was judged against the transcript as it existed THEN
# (e.g. a middle region too small to matter), but the conversation keeps growing and can
# accumulate plenty of compressible material later. Without a recovery path the session
# never auto-compacts again and rides into the provider's hard context limit. Recovery is a
# probation probe: after _ANTI_THRASH_RECOVERY_SECONDS of continuous block, allow ONE
# attempt by dropping the tripped counter(s) to 1 strike (persisted, so sibling agents on
# the same session row unblock too). If the probe is ineffective again the very next verdict
# re-trips the guard, so the worst case in the truly-incompressible state is one compaction
# attempt per recovery window — bounded, not thrash. The clock is armed lazily on the first
# BLOCKED evaluation and persisted on the session row (#100185): a fresh process/compressor
# that loads a durable tripped counter (#69872) with no stored deadline starts a full window
# blocked, preserving the restart-must-not-disarm contract (#54923) — but one that loads an
# already-armed deadline resumes that window instead of restarting it.
if self._ineffective_compression_count >= 2: if self._ineffective_compression_count >= 2:
self._record_ineffective_compression_verdict(1) self._record_ineffective_compression_verdict(1)
if self._fallback_compression_streak >= 2: if self._fallback_compression_streak >= 2:
@@ -2548,6 +2719,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
prune_boundary = self._prune_boundary(result, protect_tail_count, protect_tail_tokens) prune_boundary = self._prune_boundary(result, protect_tail_count, protect_tail_tokens)
pruned = self._dedupe_tool_results(result) pruned = self._dedupe_tool_results(result)
# Just-loaded / tail-referenced skills keep full skill_view bodies through the ordinary passes. # Just-loaded / tail-referenced skills keep full skill_view bodies through the ordinary passes.
# Without this, a skill loaded moments before a compaction can be demoted to metadata while the
# model still believes its instructions are in context. See #32106.
protected_skills = _collect_protected_skill_names(result, prune_boundary) protected_skills = _collect_protected_skill_names(result, prune_boundary)
# Pass 2: summarize old tool results. Pass 3: shrink large tool_call arguments INSIDE the parsed JSON so # Pass 2: summarize old tool results. Pass 3: shrink large tool_call arguments INSIDE the parsed JSON so
# the result stays valid; otherwise providers 400 on every turn until the call leaves the window. # the result stays valid; otherwise providers 400 on every turn until the call leaves the window.
@@ -2559,6 +2732,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
self._truncate_tool_call_args_at(result, i) self._truncate_tool_call_args_at(result, i)
# Pass 3.5: retire image payloads inside the protected tail; re-sent embeds otherwise make # Pass 3.5: retire image payloads inside the protected tail; re-sent embeds otherwise make
# compression look ineffective and trip anti-thrash. Newest frames stay live. # compression look ineffective and trip anti-thrash. Newest frames stay live.
# Newest frames stay live for follow-up QA; older ones become placeholders. See #92699.
pruned += _retire_stale_tool_result_images(result) pruned += _retire_stale_tool_result_images(result)
if protect_tail_tokens is not None and protect_tail_tokens > 0 and result: if protect_tail_tokens is not None and protect_tail_tokens > 0 and result:
pruned += self._pressure_demote_tail( pruned += self._pressure_demote_tail(
@@ -2645,7 +2819,18 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
object as ``(messages, 0)``. The rearm gate is measured on message bodies only, so it is object as ``(messages, 0)``. The rearm gate is measured on message bodies only, so it is
bypassed (never the reclaim gate) when a provider-billed ``current_tokens`` reading already bypassed (never the reclaim gate) when a provider-billed ``current_tokens`` reading already
puts the request over ``threshold_tokens`` (#101889); every no-op taken while over threshold puts the request over ``threshold_tokens`` (#101889); every no-op taken while over threshold
is logged once per distinct reason.""" is logged once per distinct reason.
``_prune_old_tool_results`` runs all deterministic passes: (1) dedup byte-identical tool results —
keeps the newest full copy and back-references older exact duplicates ANYWHERE in the list
(including the protected tail), so no unique content is ever lost; (2) summarize non-tail tool
results larger than ``min_prune_chars``; (3) truncate oversized tool_call arguments on non-tail
assistant messages; (3.5) retire image payloads on all but the newest ``_MAX_KEEP_TOOL_IMAGES``
image-bearing tool results — tail-agnostic and lossy by design (#92699). Only pass (2)'s floor is
raised by ``proactive_prune_min_result_chars``; passes (1) and (3) keep their own fixed floors. The
recent-tail protection applies to passes (2) and (3); pass (1) is tail-agnostic by design because
dedup is lossless.
"""
if self.proactive_prune_tokens <= 0 or ( if self.proactive_prune_tokens <= 0 or (
current_tokens is not None and current_tokens < self.proactive_prune_tokens current_tokens is not None and current_tokens < self.proactive_prune_tokens
): ):
@@ -2689,6 +2874,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
logger.warning("Proactive tool-result prune DB commit failed; keeping the original transcript: %s", exc) logger.warning("Proactive tool-result prune DB commit failed; keeping the original transcript: %s", exc)
return messages, 0 return messages, 0
# Shared post-commit stamp site with the in-place commit and micro-compaction sync. # Shared post-commit stamp site with the in-place commit and micro-compaction sync.
# See #98450.
stamp_db_persisted_markers(pruned_msgs) stamp_db_persisted_markers(pruned_msgs)
self._proactive_prune_rearm_tokens = next_rearm_tokens self._proactive_prune_rearm_tokens = next_rearm_tokens
# Reclamation just ran: let a future lockout warn again. # Reclamation just ran: let a future lockout warn again.
@@ -2864,6 +3050,10 @@ None recoverable from deterministic fallback.
## Critical Context ## Critical Context
Summary generation was unavailable, so this is a best-effort deterministic fallback for {len(turns_to_summarize)} compacted message(s).{reason_text}""" Summary generation was unavailable, so this is a best-effort deterministic fallback for {len(turns_to_summarize)} compacted message(s).{reason_text}"""
# Per-turn truncation cuts [SKILL_PRUNED] markers; re-derive from raw turns and re-inject. # Per-turn truncation cuts [SKILL_PRUNED] markers; re-derive from raw turns and re-inject.
# Ghost-skill defense (#32106): the fallback's per-turn truncation (``_FALLBACK_TURN_MAX_CHARS``)
# routinely cuts [SKILL_PRUNED: ...] markers out of the compacted turns. Re-derive the ghosted
# skills from the raw turn contents and re-inject deterministically, exactly like the LLM-summary
# path.
_pruned_names = _collect_ghosted_skill_names(turns_to_summarize) _pruned_names = _collect_ghosted_skill_names(turns_to_summarize)
del _pruned_names[_MAX_PRUNED_SKILL_MARKERS:] del _pruned_names[_MAX_PRUNED_SKILL_MARKERS:]
summary = self._with_summary_prefix(_redact_compaction_text(body.strip())) summary = self._with_summary_prefix(_redact_compaction_text(body.strip()))
@@ -3001,6 +3191,10 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
call_kwargs["model"] = self.summary_model call_kwargs["model"] = self.summary_model
# Pinned route (stall fallback) overrides task routing so the retry leaves the stalled backend. # Pinned route (stall fallback) overrides task routing so the retry leaves the stalled backend.
call_kwargs.update(_pinned_summary_call_kwargs()) call_kwargs.update(_pinned_summary_call_kwargs())
# Compression is atomic: protect the in-flight summary call from a mid-turn gateway interrupt.
# Without this, an incoming user message aborts the summary and compression falls back to a degraded
# static marker, losing the real handoff (#23975). Re-entrant: a main-model retry (_generate_summary
# recursion) re-enters harmlessly.
_aux_call_start = time.monotonic() _aux_call_start = time.monotonic()
_latency_info: Dict[str, int] = {"prompt_build_ms": max(0, int((_aux_call_start - prompt_started_at) * 1000))} _latency_info: Dict[str, int] = {"prompt_build_ms": max(0, int((_aux_call_start - prompt_started_at) * 1000))}
call_kwargs["latency_info"] = _latency_info call_kwargs["latency_info"] = _latency_info
@@ -3026,8 +3220,25 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
# Reasoning-field fallback (DeepSeek/Qwen/Kimi put the summary in reasoning_content); capped. # Reasoning-field fallback (DeepSeek/Qwen/Kimi put the summary in reasoning_content); capped.
content = extract_content_or_reasoning(response, max_reasoning_chars=8000) content = extract_content_or_reasoning(response, max_reasoning_chars=8000)
where = f"(provider={self.provider or 'auto'} model={self.summary_model or self.model})" where = f"(provider={self.provider or 'auto'} model={self.summary_model or self.model})"
# Some OpenAI-compatible proxies (e.g. cmkey.cn, one-api channels) return a well-formed HTTP 200
# with an empty or whitespace-only ``content`` instead of an error or empty ``choices``. That
# payload passes ``_validate_llm_response`` (a ``message`` exists), so it reaches here and would
# otherwise be stored as a prefix-only summary with no body — silently wiping the compacted turns
# and making the model forget the in-progress task (#11978, #11914). Treat empty content as a
# failure so it routes through the same main-model fallback + cooldown machinery as a transport
# error, rather than replacing real context with an empty summary.
if not content.strip(): if not content.strip():
raise RuntimeError(f"Context compression LLM returned empty content {where}") raise RuntimeError(f"Context compression LLM returned empty content {where}")
# A finish_reason of "length" means the summarizer hit its output token cap mid-generation: the text
# present is PARTIAL. Persisting a partial summary as the compaction checkpoint silently truncates
# the conversation's memory — the cut-off text replaces the real middle turns AND is fed back into
# every subsequent iterative update prompt, compounding the loss across compactions. Treat it as a
# failure so it routes through the same main-model fallback + abort machinery as other degraded
# responses instead of becoming a checkpoint. (Ported from earendil-works/pi#7048.)
# A length stop means the merged rolling summary is partial — persisting it would silently drop the
# tail of the merge and feed the cut-off text into every later micro-compact pass. Leave the
# exchange unabsorbed instead; a later pass retries it. (Same class as _generate_summary's guard;
# pi#7048.)
if _response_finish_reason(response) == "length": if _response_finish_reason(response) == "length":
raise RuntimeError( raise RuntimeError(
f"Context compression summary was truncated ({_TRUNCATED_SUMMARY_MARKER}): generation hit the output " f"Context compression summary was truncated ({_TRUNCATED_SUMMARY_MARKER}): generation hit the output "
@@ -3046,6 +3257,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
# bypass_cooldown: provider-proven overflow gets ONE real attempt while armed. # bypass_cooldown: provider-proven overflow gets ONE real attempt while armed.
if prompt_started_at < self._summary_failure_cooldown_until and not bypass_cooldown: if prompt_started_at < self._summary_failure_cooldown_until and not bypass_cooldown:
logger.debug( logger.debug(
# See #100661.
"Skipping context summary during cooldown (%.0fs remaining)", "Skipping context summary during cooldown (%.0fs remaining)",
self._summary_failure_cooldown_until - prompt_started_at, self._summary_failure_cooldown_until - prompt_started_at,
) )
@@ -3076,6 +3288,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
# The summarizer may echo secrets verbatim; redact the output too. # The summarizer may echo secrets verbatim; redact the output too.
summary = _redact_compaction_text(content.strip()) summary = _redact_compaction_text(content.strip())
# Restore any [SKILL_PRUNED] marker the summarizer paraphrased away. # Restore any [SKILL_PRUNED] marker the summarizer paraphrased away.
# See #32106.
summary = _reinject_pruned_skill_markers(summary, _pruned_skill_names) summary = _reinject_pruned_skill_markers(summary, _pruned_skill_names)
summary = self._ground_historical_task_snapshot(summary, turns_to_summarize) summary = self._ground_historical_task_snapshot(summary, turns_to_summarize)
summary = self._augment_summary_lean(summary, turns_to_summarize) summary = self._augment_summary_lean(summary, turns_to_summarize)
@@ -3229,6 +3442,12 @@ Write only the summary body. Do not include any preamble or prefix."""
"""Classify a summary-call failure; retry once on the main model (returning its result) or arm a cooldown (None).""" """Classify a summary-call failure; retry once on the main model (returning its result) or arm a cooldown (None)."""
# Only a genuine no-provider RuntimeError gets the long cooldown; empty/invalid-response # Only a genuine no-provider RuntimeError gets the long cooldown; empty/invalid-response
# RuntimeErrors are transient and must get the main-model retry below first. # RuntimeErrors are transient and must get the main-model retry below first.
# ``call_llm`` raises ``RuntimeError`` for two very different cases: 1. 2. An empty/invalid response
# from a configured provider (``_validate_llm_response`` empty-``choices``/``None``, or our
# empty-``content`` guard above) — a transient/proxy fault that should fall back to the main model
# first, exactly like the transport errors handled below. Only (1) belongs in the long no-provider
# cooldown; (2) and every other exception flow into the generic fallback logic so they get a
# main-model retry before any cooldown. (#11978, #11914)
if isinstance(e, RuntimeError) and "no llm provider configured" in str(e).lower(): if isinstance(e, RuntimeError) and "no llm provider configured" in str(e).lower():
self._record_compression_failure_cooldown(_SUMMARY_FAILURE_COOLDOWN_SECONDS, "no auxiliary LLM provider configured") self._record_compression_failure_cooldown(_SUMMARY_FAILURE_COOLDOWN_SECONDS, "no auxiliary LLM provider configured")
self._last_summary_error = "no auxiliary LLM provider configured" self._last_summary_error = "no auxiliary LLM provider configured"
@@ -3269,6 +3488,11 @@ Write only the summary body. Do not include any preamble or prefix."""
# Terminal network/empty-content failure after any fallback: flag so compress() ABORTS # Terminal network/empty-content failure after any fallback: flag so compress() ABORTS
# and preserves the session; independent of abort_on_summary_failure. # and preserves the session; independent of abort_on_summary_failure.
if kind.streaming_closed: if kind.streaming_closed:
# A terminal connection/network failure or empty-content response from a degraded provider (we
# reach this branch only after any main-model fallback has already been tried or is
# unavailable). Flag it so compress() ABORTS and preserves the session unchanged instead of
# destroying the middle window for a placeholder marker — retrying once the provider recovers is
# strictly better than dropping context (#29559, #25585, #94448).
self._last_summary_network_failure = True self._last_summary_network_failure = True
elif kind.truncated: elif kind.truncated:
self._last_summary_truncated_failure = True self._last_summary_truncated_failure = True
@@ -3494,6 +3718,8 @@ Write only the summary body. Do not include any preamble or prefix."""
"""Find handoff summaries inside a compression window.""" """Find handoff summaries inside a compression window."""
n = len(messages) n = len(messages)
# Clamp: callers may pass end = len(messages)+1. # Clamp: callers may pass end = len(messages)+1.
# Defensive: clamp bounds so a caller passing an out-of-range end (e.g. tail-cut returning
# len(messages)+1 when head_end >= n) cannot trigger IndexError. (#75588)
start = max(0, min(start, n)) start = max(0, min(start, n))
end = max(start, min(end, n)) end = max(start, min(end, n))
return [ return [
@@ -3578,7 +3804,12 @@ Write only the summary body. Do not include any preamble or prefix."""
@staticmethod @staticmethod
def _tool_call_id_variants(tc) -> set: def _tool_call_id_variants(tc) -> set:
"""Return every id variant a result might reference *tc* by (forwards to message_sanitization).""" """Return every id variant a result might reference *tc* by (forwards to message_sanitization).
Thin forwarder — the policy owner is ``agent.message_sanitization.tool_call_id_variants``, which
also expands ``response_item_id`` and composite ``call|item`` bridge spellings (#63000), so the
compressor's pairing tolerance matches the pre-call sanitizer's exactly and the two can never drift.
"""
from agent.message_sanitization import tool_call_id_variants from agent.message_sanitization import tool_call_id_variants
return set(tool_call_id_variants(tc)) return set(tool_call_id_variants(tc))
@@ -3648,7 +3879,11 @@ Write only the summary body. Do not include any preamble or prefix."""
return self.protect_first_n return self.protect_first_n
def _protect_head_size(self, messages: List[Dict[str, Any]]) -> int: def _protect_head_size(self, messages: List[Dict[str, Any]]) -> int:
"""Head messages to protect: the system prompt (if present) plus the decaying ``protect_first_n`` extra rows.""" """Head messages to protect: the system prompt (if present) plus the decaying ``protect_first_n`` extra rows.
The ``protect_first_n`` portion DECAYS after the first compression (see _effective_protect_first_n)
so early user turns don't fossilize across repeated compactions (#11996).
"""
head = 1 if messages and messages[0].get("role") == "system" else 0 head = 1 if messages and messages[0].get("role") == "system" else 0
return head + self._effective_protect_first_n(messages) return head + self._effective_protect_first_n(messages)
@@ -3762,6 +3997,16 @@ Write only the summary body. Do not include any preamble or prefix."""
from agent.conversation_compression import _is_real_user_message from agent.conversation_compression import _is_real_user_message
last_user_idx = -1 last_user_idx = -1
# Find the newest user message that carries at least one image part. We anchor on image-bearing user
# messages (not all user messages) so a plain text follow-up after a big-image turn still strips the
# old image — matching the problem kilocode#9434 set out to solve.
# Newest tool message carrying an image. Tool-result images (``vision_analyze``,
# screenshot-returning tools) accumulate on their own timeline and the user anchor never protects
# the stale ones: a session whose only image-bearing user message is the FIRST one leaves ``anchor
# <= 0`` and strips nothing at all, so twenty tool results keep multi-MB of base64 in every request
# body until the provider answers 413 -- and the 413 handler's recovery compaction lands right back
# here and frees nothing, which is the wedge in #89938. Keep the newest tool image, since that is
# the one the model is reasoning about, and drop every older one wherever it sits.
for i in range(len(messages) - 1, -1, -1): for i in range(len(messages) - 1, -1, -1):
msg = messages[i] msg = messages[i]
# _is_real_user_message also rejects metadata-flagged scaffolding # _is_real_user_message also rejects metadata-flagged scaffolding
@@ -3898,7 +4143,16 @@ Write only the summary body. Do not include any preamble or prefix."""
def _ensure_last_n_user_messages_in_tail( def _ensure_last_n_user_messages_in_tail(
self, messages: List[Dict[str, Any]], cut_idx: int, head_end: int, n: int, self, messages: List[Dict[str, Any]], cut_idx: int, head_end: int, n: int,
) -> int: ) -> int:
"""Keep the last N actionable user messages in the tail; n <= 1 delegates to the single-message method.""" """Keep the last N actionable user messages in the tail; n <= 1 delegates to the single-message method.
Only REAL actionable user turns count toward N — the collector uses the same
``_is_actionable_user_turn`` / ``_is_synthetic_compression_user_turn`` pair as
``_find_last_user_message_idx``, so blank platform echoes, compaction handoffs, continuation
markers, and todo-snapshot rows never consume a slot (#69291 bug class).
A user message is already a clean boundary — there is no tool_call/result group that spans across
it, so ``_align_boundary_backward`` is intentionally NOT called. Calling it can pull the cut past
the user message into the preceding assistant(tool_calls)→tool group and split it (#22566).
"""
if n <= 1: if n <= 1:
return self._ensure_last_user_message_in_tail(messages, cut_idx, head_end) return self._ensure_last_user_message_in_tail(messages, cut_idx, head_end)
@@ -3957,6 +4211,8 @@ Write only the summary body. Do not include any preamble or prefix."""
cut_idx = self._align_boundary_backward(messages, cut_idx) cut_idx = self._align_boundary_backward(messages, cut_idx)
# Latest user message must stay in the tail (active task). Latest assistant reply must stay too; # Latest user message must stay in the tail (active task). Latest assistant reply must stay too;
# anchors only walk backward, so chaining is monotonic. # anchors only walk backward, so chaining is monotonic.
# Ensure the most recent user message is always in the tail so the active task is never lost to
# compression (fixes #10896).
cut_idx = self._ensure_last_user_message_in_tail(messages, cut_idx, head_end) cut_idx = self._ensure_last_user_message_in_tail(messages, cut_idx, head_end)
cut_idx = self._ensure_last_assistant_message_in_tail(messages, cut_idx, head_end) cut_idx = self._ensure_last_assistant_message_in_tail(messages, cut_idx, head_end)
@@ -4275,6 +4531,10 @@ Write only the summary body. Do not include any preamble or prefix."""
self.compression_count += 1 self.compression_count += 1
# Replace historical image payloads with placeholders; multi-MB base64 blobs otherwise # Replace historical image payloads with placeholders; multi-MB base64 blobs otherwise
# exceed body limits. # exceed body limits.
# Replace image parts in all compressed messages before the newest image-bearing user turn with a
# short text placeholder. Without this, tail messages keep their original multi-MB base-64 image
# payloads forever, which can push every subsequent API request past the provider's body-size limit
# and wedge the session. Port of Kilo-Org/kilocode#9434.
compressed = _strip_historical_media(compressed) compressed = _strip_historical_media(compressed)
# Like-for-like savings: current_tokens includes system prompt/tool schemas, new_estimate is # Like-for-like savings: current_tokens includes system prompt/tool schemas, new_estimate is
@@ -4300,6 +4560,10 @@ Write only the summary body. Do not include any preamble or prefix."""
# Compaction frees the biggest allocation: hand pages back to the OS (glibc/config-gated, # Compaction frees the biggest allocation: hand pages back to the OS (glibc/config-gated,
# rate-limited, #70782). debug, not warning: compression must never fail because of a trim. # rate-limited, #70782). debug, not warning: compression must never fail because of a trim.
try: try:
# A successful compaction just freed the largest allocation a long session ever drops (the
# compressed-away message dicts), which makes this the natural point to hand allocator pages
# back to the OS. #76905's trim lifecycle covers the gateway/TUI housekeeping loops but not the
# CLI compression path, so RSS keeps the pre-compaction high-water mark until exit. (#70782)
from hermes_cli.mem_trim import trim_memory from hermes_cli.mem_trim import trim_memory
trim_memory(reason="post-compression") trim_memory(reason="post-compression")
except Exception as exc: except Exception as exc:
@@ -4317,7 +4581,19 @@ Write only the summary body. Do not include any preamble or prefix."""
) -> List[Dict[str, Any]]: ) -> List[Dict[str, Any]]:
"""Summarize the middle turns: prune tool results and blank echoes (survives an abort), protect head and a """Summarize the middle turns: prune tool results and blank echoes (survives an abort), protect head and a
token-budget tail, summarize, clean orphaned tool pairs. ``force`` clears the failure cooldown and bypasses token-budget tail, summarize, clean orphaned tool pairs. ``force`` clears the failure cooldown and bypasses
the feasibility skip; ``bypass_cooldown`` runs the summary LLM without clearing the cooldown.""" the feasibility skip; ``bypass_cooldown`` runs the summary LLM without clearing the cooldown.
Args: focus_topic: Optional focus string for guided compression. When provided, the summariser will
prioritise preserving information related to this topic and be more aggressive about compressing
everything else. Inspired by Claude Code's ``/compact``. force: If True, clear any active
summary-failure cooldown before running so a manual ``/compress`` can retry immediately after an
auto-compression abort, and bypass the pre-LLM feasibility skip so an explicit user request always
exercises the full summary path. Auto-compress callers pass False. memory_context: Optional
provider-supplied context to preserve in the summary prompt. Whitespace-only values are ignored.
bypass_cooldown: If True, run the summary LLM even while the summary-failure cooldown is armed,
WITHOUT clearing it (#100661). Set by provider-proven overflow recovery, which is already bounded by
the caller's attempt budget.
"""
telemetry = self._begin_compress_attempt(current_tokens, force) telemetry = self._begin_compress_attempt(current_tokens, force)
n_messages = len(messages) n_messages = len(messages)
# Only need head + 3 tail messages minimum (token budget decides the real tail size) # Only need head + 3 tail messages minimum (token budget decides the real tail size)
+12
View File
@@ -62,6 +62,8 @@ class ContextEngine(ABC):
# Compaction parameters (read by run_agent.py for preflight). protect_first_n counts # Compaction parameters (read by run_agent.py for preflight). protect_first_n counts
# non-system head messages kept verbatim IN ADDITION to the always-protected system # non-system head messages kept verbatim IN ADDITION to the always-protected system
# prompt (3 keeps the historical head shape). # prompt (3 keeps the historical head shape).
# These control the preflight compression check. Subclasses may override via __init__ or property;
# defaults are sensible for most engines. See #13754.
threshold_percent: float = 0.75 threshold_percent: float = 0.75
protect_first_n: int = 3 protect_first_n: int = 3
protect_last_n: int = 6 protect_last_n: int = 6
@@ -183,6 +185,16 @@ class ContextEngine(ABC):
def on_session_reset(self) -> None: def on_session_reset(self) -> None:
"""/new or /reset: reset per-session state (default: counters and token tracking).""" """/new or /reset: reset per-session state (default: counters and token tracking)."""
# Reset cross-call calibration state captured under the PREVIOUS model. These fields encode "the
# provider proved this prompt fit" / "preflight can be deferred" decisions that are only valid for
# the model that produced them. Carrying them across a switch to a smaller-context model would let
# should_defer_preflight_to_real_usage() suppress a preflight compression the new model actually
# needs — the exact oversized-send-after-switch failure in #23767. The new model's first response
# repopulates them via update_from_response(). Setting last_prompt_tokens to 0 (NOT -1) is
# deliberate: 0 is the documented "no real usage yet -> use the rough estimate" state, so the post-
# response should_compress path falls back to estimate_request_tokens_rough rather than skipping
# compression. -1 is a different sentinel (#36718, "compression just ran, await real usage") and
# must not be set here.
self.last_prompt_tokens = 0 self.last_prompt_tokens = 0
self.last_completion_tokens = 0 self.last_completion_tokens = 0
self.last_total_tokens = 0 self.last_total_tokens = 0
+2
View File
@@ -20,6 +20,8 @@ from hermes_cli.sizefmt import format_bytes
# ── Plugin context-reference provider API ──────────────────────────────────── # ── Plugin context-reference provider API ────────────────────────────────────
# --------------------------------------------------------------------------- Plugin context-reference
# provider API (Issue #26193) ---------------------------------------------------------------------------
BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"}) BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"})
_context_reference_providers: dict[str, "ContextReferenceProvider"] = {} _context_reference_providers: dict[str, "ContextReferenceProvider"] = {}
+251 -8
View File
@@ -53,6 +53,8 @@ _TERMINAL_COMPRESSION_PROVENANCES = frozenset(
# Split failures are usually transient lease/DB conditions, so use the FIRST # Split failures are usually transient lease/DB conditions, so use the FIRST
# timeout-ladder rung (60s), not the 600s summary-provider cooldown. # timeout-ladder rung (60s), not the 600s summary-provider cooldown.
# Cooldown armed when a compression SPLIT fails (session_split_failed / rotation rollback, #97948 symptom
# B).
_SPLIT_FAILURE_COOLDOWN_SECONDS = 60 _SPLIT_FAILURE_COOLDOWN_SECONDS = 60
# Marker tui_gateway/server.py::_status_update matches to tag kind="compacting" for drivers' "Summarizing…" UI. Keep # Marker tui_gateway/server.py::_status_update matches to tag kind="compacting" for drivers' "Summarizing…" UI. Keep
@@ -108,6 +110,11 @@ COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE = (
# FAILURE-class notice: compression blocked, so the session grows until the provider limit kills it. Must stay visible # FAILURE-class notice: compression blocked, so the session grows until the provider limit kills it. Must stay visible
# on gateways: never add it to ROUTINE_COMPRESSION_STATUS_SAMPLES or _TELEGRAM_NOISY_STATUS_RE. # on gateways: never add it to ROUTINE_COMPRESSION_STATUS_SAMPLES or _TELEGRAM_NOISY_STATUS_RE.
# FAILURE-CLASS notice — a deliberate carve-out from routine-compression silence (#16775 class): the context
# is over the compression threshold but compression is blocked (summary-LLM cooldown / anti-thrash breaker),
# so the session will keep growing until the hard provider token limit kills it. Do NOT add it to
# ROUTINE_COMPRESSION_STATUS_SAMPLES or the gateway noise regex (_TELEGRAM_NOISY_STATUS_RE); it is pinned
# un-swallowed in tests/gateway/test_telegram_noise_filter.py::VISIBLE_COMPRESSION_MESSAGES.
CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = ( CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = (
"⚠ Context is over the compression threshold (~{tokens:,} tokens >= {threshold:,}) " "⚠ Context is over the compression threshold (~{tokens:,} tokens >= {threshold:,}) "
"but compression is currently blocked ({reason}). The model may stop responding. Run /new to start a fresh " "but compression is currently blocked ({reason}). The model may stop responding. Run /new to start a fresh "
@@ -188,6 +195,19 @@ def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]:
# Attempt ownership: stall-fallback detaches a timed-out worker and reuses the compressor, so its late unwind could # Attempt ownership: stall-fallback detaches a timed-out worker and reuses the compressor, so its late unwind could
# restore a stale snapshot or clear the fallback's cancel check. Generation guards ATTRIBUTE writes; fence, COMMITs. # restore a stale snapshot or clear the fallback's cancel check. Generation guards ATTRIBUTE writes; fence, COMMITs.
# --------------------------------------------------------------------------- Attempt ownership (#96634
# follow-up). The stall-fallback path deliberately DETACHES a timed-out primary worker (fence cancel wins;
# the future stays on the shared pool) and immediately starts a fallback attempt against the SAME
# ContextCompressor. Two races follow from that overlap: 1. The late primary's unwind still calls
# _restore_compressor_attempt_state with the PRIMARY's pre-attempt snapshot. Landing after the fallback's
# commit, it rolls _previous_summary / cooldown / provenance / telemetry back to pre-primary values —
# silently discarding fallback-owned state. 2. _compression_cancelled_check is one shared attribute: the
# late primary's ``finally`` clears the callback the fallback just installed, so the fallback's F4
# cancellation consult reads None. Both are fixed with a monotonic per-compressor attempt generation,
# claimed under one module lock. Restores and callback set/clear are keyed to the claiming generation and
# no-op when a newer attempt owns the compressor. The commit fence still owns COMMIT admission; the
# generation owns compressor-ATTRIBUTE writes — two different boundaries.
# ---------------------------------------------------------------------------
_COMPRESSOR_ATTEMPT_LOCK = threading.Lock() _COMPRESSOR_ATTEMPT_LOCK = threading.Lock()
@@ -268,7 +288,11 @@ def _restore_compressor_attempt_state(
) -> None: ) -> None:
"""Restore the per-attempt snapshot after a pre-commit hard cancel. """Restore the per-attempt snapshot after a pre-commit hard cancel.
A restore stamped with a stale ``attempt_generation`` no-ops so a timed-out primary's late unwind cannot A restore stamped with a stale ``attempt_generation`` no-ops so a timed-out primary's late unwind cannot
roll back state owned by the fallback attempt.""" roll back state owned by the fallback attempt.
``attempt_generation`` (when provided) is the claim the calling attempt took via
:func:`_claim_compressor_attempt`. See #96634.
"""
if attempt_generation is not None and not _compressor_attempt_is_current(compressor, attempt_generation): if attempt_generation is not None and not _compressor_attempt_is_current(compressor, attempt_generation):
logger.warning( logger.warning(
"Skipping stale compressor attempt-state restore: attempt " "Skipping stale compressor attempt-state restore: attempt "
@@ -348,10 +372,25 @@ class CompressionCommitFence:
self._cancelled = False self._cancelled = False
self._commit_started = False self._commit_started = False
# Readable WITHOUT the lock (begin_commit holds it until finish_commit): hosts see a hung commit. # Readable WITHOUT the lock (begin_commit holds it until finish_commit): hosts see a hung commit.
# Lock-free commit-phase marker (#76354 review F1). ``begin_commit`` RETAINS ``self._lock`` until
# ``finish_commit``, so any host-side observation that needs the lock (``try_cancel_before_commit``)
# blocks/space-outs for the whole commit. This Event is set inside ``begin_commit`` while the lock
# is held but is READABLE WITHOUT the lock, so a host can observe "a commit was admitted and may be
# in flight" even while the commit itself is hung — which is exactly when the overrun warning must
# be able to fire.
self._commit_phase = threading.Event() self._commit_phase = threading.Event()
# Set on ANY host unwind without the fence lock so FUTURE commits are blocked; bool store is atomic. # Set on ANY host unwind without the fence lock so FUTURE commits are blocked; bool store is atomic.
# Lock-free admission revocation (#76354 review F2). Set by :meth:`revoke_commit_admission` on ANY
# host unwind (KeyboardInterrupt, cancellation, unexpected exception) without touching the fence
# lock, so a host that cannot afford to block behind an in-flight commit can still guarantee no
# FUTURE commit is admitted.
self._admission_revoked = False self._admission_revoked = False
# Holder-scoped release published by the worker once it owns the durable lock (no ABA on a NEW holder). # Holder-scoped release published by the worker once it owns the durable lock (no ABA on a NEW holder).
# Holder-qualified durable-lock release hook (#76354 review F4; transplanted from PR #71569 by
# @ciabata-git). The worker publishes an idempotent, holder-scoped release callable once it owns the
# durable compression lock; a timed-out host invokes it to free the lease without racing a NEW
# holder (DB release is holder-qualified, so a stale release can never delete a replacement's row —
# no ABA).
self._lock_release_guard = threading.Lock() self._lock_release_guard = threading.Lock()
self._cancelled_lock_release: Optional[Callable[[], None]] = None self._cancelled_lock_release: Optional[Callable[[], None]] = None
self._cancelled_lock_release_requested = False self._cancelled_lock_release_requested = False
@@ -389,7 +428,14 @@ class CompressionCommitFence:
@property @property
def deadline_monotonic(self) -> float | None: def deadline_monotonic(self) -> float | None:
"""Armed deadline (absolute monotonic); the worker's stream consumer stops when the host stops waiting.""" """Armed deadline (absolute monotonic); the worker's stream consumer stops when the host stops waiting.
:meth:`set_total_ceiling_seconds` documents this deadline as "shared by the host and worker", but
until #99692 only the host could read it — ``deadline_exceeded`` answers "is it past?" for a caller
that is already polling, which is useless to a worker blocked inside a provider stream. Publishing
the instant itself lets the worker's stream consumer stop at exactly the moment the host stops
waiting (see ``auxiliary_client.aux_stream_deadline``).
"""
return self._deadline return self._deadline
def seconds_since_progress(self) -> float: def seconds_since_progress(self) -> float:
@@ -454,7 +500,14 @@ class CompressionCommitFence:
self._retain_cancelled_lock_until_worker_done = True self._retain_cancelled_lock_until_worker_done = True
def mark_commit_watermark_fenced(self) -> None: def mark_commit_watermark_fenced(self) -> None:
"""Record a watermark-bounded commit (later rows survive as tail); a detached worker may keep admission.""" """Record a watermark-bounded commit (later rows survive as tail); a detached worker may keep admission.
Called by the compression worker right after it captures ``get_active_message_watermark()`` under
the durable compression lock (#75316/#87484). A watermark-fenced commit archives ONLY rows at or
below the watermark; rows appended later — e.g. the user turn the host released at the turn-hold
boundary (#97963) — are cloned as live concurrent tail. That is exactly the property a host needs
before letting a detached worker keep its commit admission.
"""
self._commit_watermark_fenced = True self._commit_watermark_fenced = True
@property @property
@@ -481,6 +534,11 @@ class CompressionCommitFence:
# ── Holder-qualified durable-lease cancellation: release is DELETE WHERE # ── Holder-qualified durable-lease cancellation: release is DELETE WHERE
# holder = ?, so a stale release can never free a NEW holder's lease (no ABA). # holder = ?, so a stale release can never free a NEW holder's lease (no ABA).
# ── Holder-qualified durable-lease cancellation (#76354 F4) ────────── Transplanted from PR #71569
# (@ciabata-git): the worker publishes an idempotent, holder-scoped release hook once it owns the
# durable compression lock, and the host invokes it after winning cancellation. ABA safety comes from
# SessionDB.release_compression_lock being holder-qualified (DELETE ... WHERE holder = ?), so a stale
# release can never free a NEW holder's lease.
def begin_lock_setup(self) -> bool: def begin_lock_setup(self) -> bool:
"""Hold the fence across lock acquisition + release-hook publication so a timeout cannot win between.""" """Hold the fence across lock acquisition + release-hook publication so a timeout cannot win between."""
self._lock.acquire() self._lock.acquire()
@@ -524,6 +582,9 @@ DEFAULT_CONTEXT_TIMEOUT_SECONDS = 120.0
DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0 DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0
# Unlike explicit_interrupt, a /stop after the stall window arms the durable backoff (no automatic re-entry). # Unlike explicit_interrupt, a /stop after the stall window arms the durable backoff (no automatic re-entry).
# Distinct from ``explicit_interrupt``: a /stop that arrived after the summary stream had already crossed
# the no-progress stall window (#96775). Ordinary early /stop stays cooldown-neutral; this class arms the
# durable backoff so the next automatic turn does not re-enter the same stalled strategy.
STALL_INTERRUPTED_FAILURE_CLASS = "stall_interrupted" STALL_INTERRUPTED_FAILURE_CLASS = "stall_interrupted"
# Daemon pool so a fence-cancelled hung worker cannot block interpreter exit; never shut down per call. # Daemon pool so a fence-cancelled hung worker cannot block interpreter exit; never shut down per call.
@@ -535,6 +596,8 @@ _COMMIT_OVERRUN_WAIT_SLICE_SECONDS = 30.0
# A worker exiting within the grace proves no provider call is in flight, so its lease may be released even # A worker exiting within the grace proves no provider call is in flight, so its lease may be released even
# on the total-ceiling path; one that doesn't exit is orphaned behind the poison fence and keeps its lease. # on the total-ceiling path; one that doesn't exit is orphaned behind the poison fence and keeps its lease.
# Bounded grace given to a fence-cancelled compression worker to actually exit before the host moves on
# (#97488).
_CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS = 5.0 _CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS = 5.0
@@ -561,6 +624,16 @@ def _join_cancelled_worker(future: Any, grace_seconds: float) -> bool:
# The executor queue is unbounded and a queued job would run stale, so admission is capped at the worker # The executor queue is unbounded and a queued job would run stale, so admission is capped at the worker
# count (fail fast, continue uncompressed). Slots free via done-callback; a never-returning worker loses one. # count (fail fast, continue uncompressed). Slots free via done-callback; a never-returning worker loses one.
# Bounded admission for the shared compress-timeout pool (#76354 review F6). The stdlib executor queue is
# unbounded: with all four workers wedged in hung summaries, a fifth compression would queue silently, wait
# out its whole timeout without ever starting, and remain eligible to run as a stale job whenever a worker
# recovered. Admission is therefore capped at the worker count — when every worker slot is occupied (running
# OR admitted-not-started) submission FAILS FAST and the caller continues without compression. Recovery
# contract when all workers are wedged: new compressions fail fast (no queue growth, conversation continues
# uncompressed, a warning is logged each attempt); wedged workers are fence-cancelled so they cannot publish
# anything when they eventually return, and each recovery frees its admission slot via the future
# done-callback, restoring normal service. If a worker NEVER returns, its slot is lost for the process
# lifetime — bounded, observable degradation instead of an unbounded stale-job queue.
_COMPRESS_EXECUTOR_MAX_WORKERS = 4 _COMPRESS_EXECUTOR_MAX_WORKERS = 4
_compress_admission_lock = threading.Lock() _compress_admission_lock = threading.Lock()
_compress_admitted_count = 0 _compress_admitted_count = 0
@@ -632,7 +705,12 @@ def compression_attempt_stalled(
) -> bool: ) -> bool:
"""Return whether a pre-commit cancel landed after the stall window. """Return whether a pre-commit cancel landed after the stall window.
An early ``/stop`` stays cooldown-neutral; an interrupt after the inactivity budget counts as a stall so An early ``/stop`` stays cooldown-neutral; an interrupt after the inactivity budget counts as a stall so
the next automatic turn does not blindly retry.""" the next automatic turn does not blindly retry.
When the fence (or, without a fence, the attempt clock) has already sat idle for the configured
compression inactivity budget, the interrupt is a stalled attempt — the same condition the host timeout
uses — and the next automatic turn must not blindly retry that strategy (#96775).
"""
idle = idle_timeout_seconds idle = idle_timeout_seconds
if idle is None: if idle is None:
idle, _ceiling = resolve_context_compression_timeouts() idle, _ceiling = resolve_context_compression_timeouts()
@@ -674,6 +752,8 @@ def _record_stall_interrupted_backoff(
if not compression_attempt_stalled(commit_fence=commit_fence, started_at=started_at): if not compression_attempt_stalled(commit_fence=commit_fence, started_at=started_at):
return False return False
compressor = getattr(agent, "context_compressor", None) compressor = getattr(agent, "context_compressor", None)
# Same timeout cooldown ladder as summary-LLM timeouts (#62452): avoid re-burning the full idle budget
# every turn.
record = getattr(compressor, "record_timeout_failure", None) record = getattr(compressor, "record_timeout_failure", None)
if not callable(record): if not callable(record):
return False return False
@@ -741,7 +821,17 @@ def _retry_compression_on_fallback_chain(
"""Re-run an aborted compression once with the summary route pinned. """Re-run an aborted compression once with the summary route pinned.
Returns ``(messages, system_prompt)`` on real compression, else ``None`` and the caller degrades as Returns ``(messages, system_prompt)`` on real compression, else ``None`` and the caller degrades as
before. The entry's ``timeout`` sets the idle window. Re-runs the whole worker, so pre-compression before. The entry's ``timeout`` sets the idle window. Re-runs the whole worker, so pre-compression
callbacks must be idempotent.""" callbacks must be idempotent.
The retry is bounded the same way the primary was: silence for one idle window ends it, while a fallback
that is streaming keeps its ceiling. The entry's own ``timeout`` (when declared) sets that idle window,
so a fallback tuned for a slower-but-healthy backend is not held to a deadline the stalled primary
defined (#62452 semantics, applied to the stall path).
Known limitation (accepted, #96634 review): the retry re-runs the COMPLETE worker, which repeats
memory/plugin pre-compression callbacks. Built-in callbacks are idempotent (re-reads and overwrites of
attempt-scoped state); third-party plugin callbacks are advised to be. Splitting the worker to resume
mid-pipeline would couple this path to every host's callback ordering — deliberately out of scope.
"""
# An explicit stop is not a stalled route. The retry worker would abort on # An explicit stop is not a stalled route. The retry worker would abort on
# the same event anyway, but starting one at all makes /stop look ignored. # the same event anyway, but starting one at all makes /stop look ignored.
hard_cancel = getattr(telemetry_agent, "_hard_interrupt_requested", None) hard_cancel = getattr(telemetry_agent, "_hard_interrupt_requested", None)
@@ -930,6 +1020,9 @@ def run_compress_context_with_progress_timeout(
executor = _get_compress_timeout_executor() executor = _get_compress_timeout_executor()
# Refuse rather than queue when the pool is full: a queued job would wait out # Refuse rather than queue when the pool is full: a queued job would wait out
# its budget unstarted and run stale later. Skip compression this cycle. # its budget unstarted and run stale later. Skip compression this cycle.
# A queued job would silently wait out its whole budget without starting and stay eligible to run as a
# stale cancelled job when a worker recovers. Fail fast: continue without compression this cycle. See
# #76354.
if not _try_admit_compression_job(): if not _try_admit_compression_job():
logger.warning( logger.warning(
"Context compression pool saturated (%d workers busy) — refusing new compression this cycle and continuing without " "Context compression pool saturated (%d workers busy) — refusing new compression this cycle and continuing without "
@@ -980,6 +1073,14 @@ def run_compress_context_with_progress_timeout(
# cancel() is a no-op for a running worker (fence handles that path). # cancel() is a no-op for a running worker (fence handles that path).
future.cancel() future.cancel()
total_exhausted = time.monotonic() - wait_started >= ceiling or fence.deadline_exceeded total_exhausted = time.monotonic() - wait_started >= ceiling or fence.deadline_exceeded
# #97488 teardown (total-ceiling path only): give the cancelled worker a bounded grace to actually
# exit before this host moves on. The worker checks the poison fence between provider phases, so a
# cooperative worker exits quickly; an uninterruptible provider call is orphaned behind the fence
# after the grace elapses (its late result is discarded and cannot touch session state). The
# idle-stall path intentionally skips the join: its worker is by definition silent/hung, the
# stall-fallback retry below needs a prompt host return (pinned by the #76354 S3 latency contract),
# and the fence poison + attempt-generation supersession already protect state against its late
# unwind.
if total_exhausted: if total_exhausted:
# A total-ceiling candidate may be unwinding a healthy provider call; keep its # A total-ceiling candidate may be unwinding a healthy provider call; keep its
# lease until it exits so no other attempt overlaps the unchanged source. # lease until it exits so no other attempt overlaps the unchanged source.
@@ -999,6 +1100,9 @@ def run_compress_context_with_progress_timeout(
handled_exit = True handled_exit = True
_release_cancelled_worker(future, fence, total_exhausted=total_exhausted, ceiling=ceiling) _release_cancelled_worker(future, fence, total_exhausted=total_exhausted, ceiling=ceiling)
waited = time.monotonic() - wait_started waited = time.monotonic() - wait_started
# #76354 S3 analogue for this wait: charge the idle budget from the LAST PROGRESS event, not from
# the start of this wait slice. Waiting a full ``idle`` after progress that landed early in the
# previous slice would allow silence to approach 2x the budget.
since_progress = fence.seconds_since_progress() since_progress = fence.seconds_since_progress()
# Lease is free, so run the fallback BEFORE on_timeout: that callback records # Lease is free, so run the fallback BEFORE on_timeout: that callback records
# the summary-failure cooldown, which would no-op the retry's summary call. # the summary-failure cooldown, which would no-op the retry's summary call.
@@ -1222,7 +1326,15 @@ def compression_blocked_transiently(agent: Any) -> bool:
"""Type-pinned read of the transient-block signal. """Type-pinned read of the transient-block signal.
Set when an automatic pass no-ops on a TRANSIENT guard (summary-failure cooldown or structural backoff). Set when an automatic pass no-ops on a TRANSIENT guard (summary-failure cooldown or structural backoff).
Consumers must defer, not count it toward ``compression_exhausted``, or an overflow auto-reset wipes a Consumers must defer, not count it toward ``compression_exhausted``, or an overflow auto-reset wipes a
session that was merely cooling down. The permanent ``ineffective`` breaker never sets it.""" session that was merely cooling down. The permanent ``ineffective`` breaker never sets it.
See #97488.
Consumers (the overflow-recovery loops in ``conversation_loop``) must treat such a no-op as a temporary
defer, NOT as evidence the session is incompressible: counting it toward ``compression_exhausted`` lets
a real upstream ``context_length_exceeded`` auto-reset (wipe) a session whose compression was merely
cooling down (#97488). The permanent ``ineffective`` breaker intentionally does NOT set this signal — a
genuinely incompressible session must still be able to exhaust.
"""
_sig = getattr(agent, "_compression_blocked_transient", None) _sig = getattr(agent, "_compression_blocked_transient", None)
return isinstance(_sig, str) and bool(_sig) return isinstance(_sig, str) and bool(_sig)
@@ -1265,7 +1377,14 @@ def _adopt_live_compression_child(
) -> Optional[List[Dict[str, Any]]]: ) -> Optional[List[Dict[str, Any]]]:
"""Move a stale compression contender onto the live continuation tip. """Move a stale compression contender onto the live continuation tip.
Resolve and load first, then mutate the agent, so ambiguous lineage or an unreadable handoff fails closed. Resolve and load first, then mutate the agent, so ambiguous lineage or an unreadable handoff fails closed.
Uses the transitive ``get_compression_tip`` walk; a tip is adopted only while its row is still live.""" Uses the transitive ``get_compression_tip`` walk; a tip is adopted only while its row is still live.
Resolution uses the canonical transitive walk ``get_compression_tip`` so a lineage with >=2 compression
hops (root -> mid -> tip) recovers to the live tip — the depth-1 ``find_live_compression_child`` lookup
this used to call finds no live *direct* child in that shape and skipped recovery (#82001). The tip walk
returns the input id when no continuation exists, and a resolved tip is adopted only while its row is
still live — both cases fail closed exactly as before.
"""
resolver = getattr(type(session_db), "get_compression_tip", None) resolver = getattr(type(session_db), "get_compression_tip", None)
row_getter = getattr(type(session_db), "get_session", None) row_getter = getattr(type(session_db), "get_session", None)
loader = getattr(type(session_db), "get_messages_as_conversation", None) loader = getattr(type(session_db), "get_messages_as_conversation", None)
@@ -1591,6 +1710,12 @@ def _lower_threshold_to_aux_context(
safe_pct = int((aux_context / main_ctx) * 100) if main_ctx else 50 safe_pct = int((aux_context / main_ctx) * 100) if main_ctx else 50
# Mirror the compressor's threshold math (percent floor, output reservation, 64K floor): a suggestion it # Mirror the compressor's threshold math (percent floor, output reservation, 64K floor): a suggestion it
# would override is silently ignored and this warning reappears every session. External engines: keep it plain. # would override is silently ignored and this warning reappears every session. External engines: keep it plain.
# The "lower the threshold" suggestion must survive the built-in trigger recomputation (#67422):
# _effective_threshold_percent() raises sub-75% values back up for main windows under 512K, and
# _compute_threshold_tokens() further applies the output-token reservation, the 64K floor, and the
# degenerate-window guard. Recommending a value those would override is silently ignored and this
# warning would reappear every session — so mirror the compressor's own math and only offer the option
# when the recomputed trigger actually fits the auxiliary model's context.
from agent.context_compressor import ContextCompressor as _CC from agent.context_compressor import ContextCompressor as _CC
recomputed_threshold = None recomputed_threshold = None
if main_ctx and isinstance(compressor, _CC): if main_ctx and isinstance(compressor, _CC):
@@ -1979,6 +2104,12 @@ def _ensure_compressed_has_user_turn(original_messages: list, compressed: list)
"""Preserve human intent, not merely a synthetic user-role placeholder.""" """Preserve human intent, not merely a synthetic user-role placeholder."""
if any(_is_real_user_message(message) for message in compressed) or _compressed_has_busy_steer(compressed): if any(_is_real_user_message(message) for message in compressed) or _compressed_has_busy_steer(compressed):
return "already_present" return "already_present"
# Post-commit contract (#98450, mirrors _sync_micro_compact_to_db): archive_and_compact just durably
# wrote every dict in `compressed` as the new active set, but compress() returned marker-swept COPIES
# (_strip_persistence_markers, #57491). These exact dict instances become the live message list the
# caller keeps, so without the stamp the next _persist_session → _flush_messages_to_session_db_unlocked
# walk treats the whole compacted transcript as unpersisted and re-INSERTs it — the live set doubles on
# every compaction (~58K → ~512K tokens in production).
from agent.context_compressor import ( from agent.context_compressor import (
_INFLIGHT_REPLAY_MERGED_KEY, COMPRESSION_CONTINUATION_USER_CONTENT, _fresh_compaction_message_copy, _INFLIGHT_REPLAY_MERGED_KEY, COMPRESSION_CONTINUATION_USER_CONTENT, _fresh_compaction_message_copy,
) )
@@ -1987,6 +2118,9 @@ def _ensure_compressed_has_user_turn(original_messages: list, compressed: list)
return "already_present" return "already_present"
# One reversed scan over BOTH kinds: scanning steer then user would let an older # One reversed scan over BOTH kinds: scanning steer then user would let an older
# consumed steer outrank a newer real user request and replay it. # consumed steer outrank a newer real user request and replay it.
# One reversed positional scan: the anchor is whichever intent-bearing row is LAST in the original
# transcript — a real ``role=user`` turn or a steer marker riding inside a ``role=tool`` result. See
# #100053.
for message in reversed(original_messages): for message in reversed(original_messages):
if _is_real_user_message(message): if _is_real_user_message(message):
return _insert_real_user_anchor(compressed, _fresh_compaction_message_copy(message)) return _insert_real_user_anchor(compressed, _fresh_compaction_message_copy(message))
@@ -2421,6 +2555,13 @@ def _adopt_grown_durable_parent(agent: Any, lease: _CompressionLease, messages:
return None return None
# In-memory carries this turn's un-persisted user tail; flush it via the normal # In-memory carries this turn's un-persisted user tail; flush it via the normal
# rotation-boundary path before adopting, else skip adoption (would drop input). # rotation-boundary path before adopting, else skip adoption (would drop input).
# The in-memory transcript carries the CURRENT turn's un-persisted user tail (anchored by
# _persist_user_message_idx) that the durable snapshot read above does not contain yet. Flush that tail
# through the normal rotation-boundary path (conversation_history = the already-durable prefix, #68196
# boundary) BEFORE adopting, then re-read the durable parent so the adopted snapshot includes the live
# input. If the flush fails (or the anchor is unknown), skip adoption entirely: replacing the in-memory
# transcript with a snapshot that lacks the user's input would silently drop it from the summarized and
# rotated history (#adopt-live-tail).
_preflush_idx = getattr(agent, "_persist_user_message_idx", None) _preflush_idx = getattr(agent, "_persist_user_message_idx", None)
# No un-persisted tail means the transcript is fully durable: adopting the longer parent cannot drop input. # No un-persisted tail means the transcript is fully durable: adopting the longer parent cannot drop input.
_preflush_ok = True _preflush_ok = True
@@ -2523,6 +2664,9 @@ def _run_summary_dispatch(
# A LATE successful summary must not undo the host's timeout cooldown: the # A LATE successful summary must not undo the host's timeout cooldown: the
# compressor checks cancellation before clearing; removed in finally (no leak). # compressor checks cancellation before clearing; removed in finally (no leak).
if commit_fence is not None: if commit_fence is not None:
# Install a cancellation check the compressor consults BEFORE clearing the failure cooldown; removed
# in the finally below so it cannot leak into later attempts (e.g. a manual /compress force-clear).
# See #76354.
_install_compression_cancelled_check( _install_compression_cancelled_check(
agent.context_compressor, lambda: commit_fence.is_cancelled, attempt_generation agent.context_compressor, lambda: commit_fence.is_cancelled, attempt_generation
) )
@@ -2594,11 +2738,20 @@ def _fold_todo_snapshot(agent: Any, compressed: list) -> None:
if todo_snapshot: if todo_snapshot:
# If this boundary pruned skill bodies, the policy behind the todos is gone: # If this boundary pruned skill bodies, the policy behind the todos is gone:
# add a reload notice after TODO_INJECTION_HEADER so both strip together. # add a reload notice after TODO_INJECTION_HEADER so both strip together.
# Retention parity (#84718): the snapshot below re-injects the imperative verbatim. If this same
# boundary pruned skill bodies to [SKILL_PRUNED: ...] markers, the policy that governed those tasks
# is gone — couple a reload instruction to the snapshot so the imperative never crosses the boundary
# alone.
_reload_notice = _pruned_skill_reload_notice(compressed) _reload_notice = _pruned_skill_reload_notice(compressed)
if _reload_notice: if _reload_notice:
todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}" todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}"
# Fold the snapshot into a trailing REAL user msg (no synthetic user/user pair); # Fold the snapshot into a trailing REAL user msg (no synthetic user/user pair);
# strip old snapshots first. Scaffolding tails must not absorb it (provenance). # strip old snapshots first. Scaffolding tails must not absorb it (provenance).
# Any snapshot merged at an earlier boundary is stripped first so repeated compactions refresh
# rather than accumulate todo state (#26981). Scaffolding tails (continuation marker, summary
# handoff, a bare stale snapshot row) must never absorb the snapshot: merging would upgrade them to
# "real user" evidence and break zero-user provenance (#69292), so those keep the flagged standalone
# append and the real-user preservation pass continues to see todo scaffolding, not human intent.
from agent.context_compressor import _append_text_to_content from agent.context_compressor import _append_text_to_content
merged = False merged = False
_tail = compressed[-1] if compressed and isinstance(compressed[-1], dict) else None _tail = compressed[-1] if compressed and isinstance(compressed[-1], dict) else None
@@ -2630,6 +2783,12 @@ def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str:
# Refresh tool schemas at the commit boundary: forever-sessions never restart, # Refresh tool schemas at the commit boundary: forever-sessions never restart,
# so config reaches agent.tools here. Keep list identity if byte-equal (cache). # so config reaches agent.tools here. Keep list identity if byte-equal (cache).
try: try:
# Refresh dynamic tool schemas at the same admitted-commit boundary that rebuilds the system prompt
# (maintainer-directed, #95681 arc): forever-sessions (Bot Mode chats, gateway channels) never
# restart, so compaction is the ONLY point where a config change — image model swap, delegation
# depth, code_execution mode — can reach agent.tools. The prompt cache is already broken here, so
# the refresh is free; when nothing changed the snapshot is byte-equal and we keep the existing list
# object (identity matters to provider-side tool-block caching on some backends).
_refresh_agent_tool_definitions(agent) _refresh_agent_tool_definitions(agent)
except Exception: # noqa: BLE001 except Exception: # noqa: BLE001
logger.warning( logger.warning(
@@ -2638,6 +2797,13 @@ def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str:
# ALWAYS rebuild the prompt here: keeping old bytes meant prompt-builder changes # ALWAYS rebuild the prompt here: keeping old bytes meant prompt-builder changes
# never reached long sessions. Equal bytes keep KV; preserve object identity. # never reached long sessions. Equal bytes keep KV; preserve object identity.
# ALWAYS rebuild the prompt at the admitted-commit boundary (maintainer-directed, #95681 arc). The
# previous "keep-prompt" containment branch put the OLD bytes back whenever the reloaded memory blocks
# were already embedded — which meant prompt-builder changes (guidance diets, new blocks, renames) NEVER
# reached a long-lived session. The cache argument for keeping bytes was hollow: when nothing changed,
# the rebuild is byte-identical and local KV prefixes survive on equality; when something changed, the
# cache was stale by definition and propagation is the point. Preserve OBJECT identity on byte-equality
# for backends that key on it.
rebuilt_system_prompt = agent._build_system_prompt(system_message) rebuilt_system_prompt = agent._build_system_prompt(system_message)
if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt: if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt:
new_system_prompt = agent._cached_system_prompt = cached_system_prompt new_system_prompt = agent._cached_system_prompt = cached_system_prompt
@@ -2662,6 +2828,13 @@ def _salvage_or_refuse_grown_transcript(
Compares like-for-like rough estimates; on growth tries one mechanical salvage pass, else treats the Compares like-for-like rough estimates; on growth tries one mechanical salvage pass, else treats the
attempt as a refused no-op. Returns ``(compressed, None)`` to proceed or ``(None, prompt)`` when refused attempt as a refused no-op. Returns ``(compressed, None)`` to proceed or ``(None, prompt)`` when refused
(caller releases the lease).""" (caller releases the lease)."""
# Anti-growth guard at the COMMIT SITE: never persist a compression that makes the transcript larger
# (observed: 379K -> 687K when the generated summary plus retained reasoning exceeded what it replaced).
# Compare like-for-like (both rough estimates of the same message shape) so an "actual vs estimate"
# measurement mismatch cannot produce a false verdict. The gateway has a rotation-path-only guard
# (#83339), but in-place compaction commits inside this method via archive_and_compact — before the
# gateway can inspect the result — so the guard must live here to protect both paths. On growth, treat
# the attempt as a no-op: the original transcript stays untouched and durable.
_rough_in = estimate_messages_tokens_rough(messages) _rough_in = estimate_messages_tokens_rough(messages)
_rough_out = estimate_messages_tokens_rough(compressed) _rough_out = estimate_messages_tokens_rough(compressed)
if _rough_out > _rough_in: if _rough_out > _rough_in:
@@ -2699,6 +2872,9 @@ def _salvage_or_refuse_grown_transcript(
# Count the refusal as an ineffective-compaction strike so the anti-thrash # Count the refusal as an ineffective-compaction strike so the anti-thrash
# breaker latches; otherwise auto-compress retries the same summary every turn. # breaker latches; otherwise auto-compress retries the same summary every turn.
with _swallow('could not record rejected-compaction strike', exc_info=True): with _swallow('could not record rejected-compaction strike', exc_info=True):
# Without this, the unchanged transcript stays over the compression threshold and automatic
# compression retries the identical summary request on every turn (#88568). Manual /compress
# keeps bypassing the latch (force=True skips the guards).
agent.context_compressor.record_rejected_compaction() agent.context_compressor.record_rejected_compaction()
_restore_prune_rearm_tokens(agent.context_compressor, attempt_snapshot) _restore_prune_rearm_tokens(agent.context_compressor, attempt_snapshot)
return None, _existing_sp return None, _existing_sp
@@ -2726,6 +2902,9 @@ def _carry_session_state_to_child(agent: Any, old_session_id: str, old_title: An
transfer clears the ancestor's row, then restored so an inherited auto-title stays upgradeable. transfer clears the ancestor's row, then restored so an inherited auto-title stays upgradeable.
""" """
with _swallow('Could not migrate goal on compression: %s'): with _swallow('Could not migrate goal on compression: %s'):
# Carry a persistent /goal onto the continuation session. Compression mints a fresh child id;
# load_goal does a flat per-session lookup with no parent walk, so without this an active goal
# silently dies at the boundary (#33618).
from hermes_cli.goals import migrate_goal_to_session from hermes_cli.goals import migrate_goal_to_session
migrate_goal_to_session(old_session_id, agent.session_id, reason="compression") migrate_goal_to_session(old_session_id, agent.session_id, reason="compression")
with _swallow('Could not migrate heartbeat on compression: %s'): with _swallow('Could not migrate heartbeat on compression: %s'):
@@ -2980,6 +3159,10 @@ def _candidate_rejected(
# Compare semantic state, not identity: engines may return an equal copy or # Compare semantic state, not identity: engines may return an equal copy or
# mutate the live list. ``==`` first (subclass __eq__), then marker-insensitive. # mutate the live list. ``==`` first (subclass __eq__), then marker-insensitive.
# Neither case may rotate or rewrite the session. The raw ``==`` leg runs FIRST so a list subclass
# returned by an engine keeps its ``__eq__`` semantics (tests seam on this); the marker-insensitive leg
# (#92231) then covers the cold-resume shape where the stamped snapshot differs from the marker-swept
# compress() output only by ``_db_persisted``.
if compressed == messages_before_compression or ( if compressed == messages_before_compression or (
_strip_marker_for_comparison(compressed) == _strip_marker_for_comparison(messages_before_compression) _strip_marker_for_comparison(compressed) == _strip_marker_for_comparison(messages_before_compression)
): ):
@@ -3095,6 +3278,19 @@ def _commit_compaction(
agent._last_flushed_db_idx = 0 agent._last_flushed_db_idx = 0
else: else:
# Bind old_session_id first: it is the rollback key in the handler below. # Bind old_session_id first: it is the rollback key in the handler below.
# ── Rotation (legacy): end this session, fork a continuation ─ Flush any un-persisted
# current-turn messages to the OLD session before ending it, so they survive in the
# preserved parent transcript (#47202). (In-place skips this — see above.) Pass the
# already-durable prefix as conversation_history so the flush skips it by identity (#68196).
# Preflight compression runs BEFORE the normal turn flush has stamped the cold-resumed
# history dicts with _DB_PERSISTED_MARKER, so without a boundary
# _flush_messages_to_session_db treats every restored row as new and re-appends the whole
# transcript to the parent. turn_context anchors _persist_user_message_idx at the
# current-turn user message before preflight runs, so messages[:idx] is exactly the
# persisted prefix; only the current turn's new messages get written. Bound to
# old_session_id, hoisted above the flush: the ``except`` handler below keys its in-memory
# rollback off this name, so anything that fails from here on rolls the transcript back
# instead of leaving the failed attempt's compacted snapshot in place.
old_session_id = agent.session_id old_session_id = agent.session_id
_publish_rotated_compaction( _publish_rotated_compaction(
agent, messages, compressed, new_system_prompt=new_system_prompt, lease=lease, agent, messages, compressed, new_system_prompt=new_system_prompt, lease=lease,
@@ -3117,6 +3313,23 @@ def _commit_compaction(
): ):
if rotation_rollback: if rotation_rollback:
old_session_id = None old_session_id = None
# In-place sibling of the rotation rollback above (#99477). archive_and_compact() is atomic,
# so a raise before it returned means EVERY pre-compaction row is still ``active = 1`` in
# state.db — nothing was archived and the compacted set was never inserted. But
# ``compressed`` is the marker-swept output of compress() (_strip_persistence_markers,
# #57491) and the post-commit ``stamp_db_persisted_markers`` never ran, so handing it back
# makes the next append-only flush treat the whole compacted transcript as new and INSERT it
# ON TOP of the rows it was supposed to replace. The active set then holds the summary AND
# the turns it summarized; the next resume reloads both, the token count goes UP, preflight
# fires again, and each failed attempt appends another copy of the protected head + tail
# (#99477: ~15 real turns stored as 3,814 rows, the first user message repeated 893 times).
# Gate on ``split_status`` rather than ``compacted_in_place``: it is assigned on the
# statement immediately after the atomic commit returns, so a committed compaction can never
# be rolled back into a live/durable mismatch of the opposite sign. The deepcopy carries
# each row's _DB_PERSISTED_MARKER from the pre-compression snapshot, so the restored
# transcript is correctly skipped by the flush, and replacing every dict breaks
# _db_flush_scan_prefix identity (same reasoning as the rotation branch — no explicit clear
# needed).
messages[:] = copy.deepcopy(messages_before_compression) messages[:] = copy.deepcopy(messages_before_compression)
compressed = messages compressed = messages
made_progress = False made_progress = False
@@ -3134,6 +3347,7 @@ def _commit_compaction(
# Arm the failure cooldown so the next turn can't rerun the doomed compression; # Arm the failure cooldown so the next turn can't rerun the doomed compression;
# try/except so a stub compressor can't mask the original error in this handler. # try/except so a stub compressor can't mask the original error in this handler.
with _swallow('could not record split-failure cooldown', exc_info=True): with _swallow('could not record split-failure cooldown', exc_info=True):
# See #97948.
agent.context_compressor._record_compression_failure_cooldown( agent.context_compressor._record_compression_failure_cooldown(
_SPLIT_FAILURE_COOLDOWN_SECONDS, f"session_split_failed: {e}" _SPLIT_FAILURE_COOLDOWN_SECONDS, f"session_split_failed: {e}"
) )
@@ -3270,6 +3484,13 @@ def _begin_compression_attempt(agent: Any, *, force: bool, defer_notification: b
agent._last_compression_attempt_recorded = True agent._last_compression_attempt_recorded = True
agent._last_compression_attempt_in_place = None agent._last_compression_attempt_in_place = None
agent._compression_skipped_due_to_lock = None agent._compression_skipped_due_to_lock = None
# Clear the lock-skip signal at the VERY TOP, before the codex route and the breaker gates below can
# early-return (per-attempt state rule, #58630/#69853). A stale ``True``/holder value from a prior
# lock-skip must never make a later breaker/codex no-op look like lock contention to the automatic-path
# consumers (compression_deferred, #49874) — the second clear before lock acquisition below stays for
# the same reason it was added in #69870 and is simply idempotent now.
# Transient-block signal (#97488): cleared with the same per-attempt rule; set by the breaker gates
# below when a TRANSIENT guard (cooldown / structural backoff) no-ops this pass.
agent._compression_blocked_transient = None agent._compression_blocked_transient = None
started_at = time.monotonic() started_at = time.monotonic()
attempt_id = uuid.uuid4().hex attempt_id = uuid.uuid4().hex
@@ -3327,7 +3548,24 @@ def compress_context(
"""Compress conversation context and split the session in SQLite. """Compress conversation context and split the session in SQLite.
``force`` (manual /compress) clears the summary-failure cooldown; ``bypass_cooldown`` (provider-proven ``force`` (manual /compress) clears the summary-failure cooldown; ``bypass_cooldown`` (provider-proven
overflow) skips it once, breakers still apply. ``commit_fence`` stops a timed-out worker mutating session overflow) skips it once, breakers still apply. ``commit_fence`` stops a timed-out worker mutating session
state. Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split.""" state. Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split.
Args: agent: The owning :class:`AIAgent`. messages: Current message history (will be summarised).
system_message: Current system prompt; used when compression needs a rebuilt cached prompt.
approx_tokens: Pre-compression token estimate, logged for ops. task_id: Tool task scope (used for
clearing file-read dedup state). focus_topic: Optional focus string for guided compression — the
summariser will prioritise preserving information related to this topic. Inspired by Claude Code's
``/compact <focus>``. force: If True, bypass any active summary-failure cooldown. Set by the manual
``/compress`` slash command so users can retry immediately after an auto-compress abort. Auto-compress
callers use the default ``False``. bypass_cooldown: If True, the automatic breaker gates ignore ONLY the
summary-failure cooldown for this attempt (#100661). Set by the provider-proven overflow recovery path:
the provider already rejected the request, so deferring until the cooldown lapses wedges the session.
Unlike ``force`` it does not clear the cooldown, and the ineffective/structural breakers still apply; a
failed attempt records its cooldown normally. defer_context_engine_notification: Delay the existing
context-engine hook until a manual host commits its outer history transaction. commit_fence: Optional
cooperative fence for executor callers that may time out. It prevents a late worker from mutating
session state after its caller has moved on.
"""
attempt = _begin_compression_attempt(agent, force=force, defer_notification=defer_context_engine_notification) attempt = _begin_compression_attempt(agent, force=force, defer_notification=defer_context_engine_notification)
# Codex owns the real thread; route compaction to its own compact (config # Codex owns the real thread; route compaction to its own compact (config
@@ -3395,6 +3633,11 @@ def compress_context(
# Interrupts/redirects must not tear a summary in half. Use the explicit stop # Interrupts/redirects must not tear a summary in half. Use the explicit stop
# Event (message fields race) + fence timeout so pool slots free promptly. # Event (message fields race) + fence timeout so pool slots free promptly.
# Explicit stop surfaces set a separate Event atomically; never infer cause from the racy message
# fields. A host timeout also cancels the attempt's commit fence. Feed BOTH into the protected
# auxiliary-call seam so the compression owner unwinds promptly while an isolated provider stream
# finishes or closes in its daemon worker. Otherwise four timed-out streams retain all four shared
# compression-pool slots until the auxiliary stream's longer absolute ceiling expires. See #23975.
_hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None) _hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None)
phase = _run_summary_phase( phase = _run_summary_phase(
agent, messages, lease=lease, in_place=in_place, checkpoint_required=checkpoint_required, agent, messages, lease=lease, in_place=in_place, checkpoint_required=checkpoint_required,
+42 -1
View File
@@ -54,6 +54,14 @@ from hermes_logging import set_session_context
# patch them here, so they must stay bound in this namespace. # patch them here, so they must stay bound in this namespace.
from agent.conversation_compression import conversation_history_after_compression # noqa: F401 from agent.conversation_compression import conversation_history_after_compression # noqa: F401
from agent.model_metadata import ( # noqa: F401 from agent.model_metadata import ( # noqa: F401
# ----------------------------------------------------------------- Session hygiene: auto-compress
# pathologically large transcripts Long-lived gateway sessions can accumulate enough history that every
# new message rehydrates an oversized transcript, causing repeated truncation/context failures. Detect
# this early and compress proactively — before the agent even starts. (#628) Token source priority: 1.
# Actual API-reported prompt_tokens from the last turn (stored in session_entry.last_prompt_tokens) 2.
# Rough char-based estimate (str(msg)//4). Overestimates by 30-50% on code/JSON-heavy sessions, but that
# just means hygiene fires a bit early — safe and harmless.
# -----------------------------------------------------------------
estimate_messages_tokens_rough, estimate_messages_tokens_rough,
estimate_request_tokens_rough, estimate_request_tokens_rough,
save_context_length, save_context_length,
@@ -91,7 +99,13 @@ def _midturn_request_pressure_tokens(
"""Token figure the mid-turn pre-API compression guard compares: the pruned """Token figure the mid-turn pre-API compression guard compares: the pruned
native-Responses estimate when native compaction eligibility is proven (the generic native-Responses estimate when native compaction eligibility is proven (the generic
estimate overstates the wire on compacted sessions, #96995), else messages+tools. estimate overstates the wire on compacted sessions, #96995), else messages+tools.
The system prompt is counted exactly once.""" The system prompt is counted exactly once.
When the upcoming request is eligible for native Responses compaction the transport will
checkpoint-prune the payload before sending, so the generic durable-history estimate overstates the wire
by orders of magnitude on a compacted session and fires a 600s local compression the main request never
needed (#96995).
"""
try: try:
from agent.codex_responses_adapter import estimate_native_responses_preflight_tokens from agent.codex_responses_adapter import estimate_native_responses_preflight_tokens
native = estimate_native_responses_preflight_tokens( native = estimate_native_responses_preflight_tokens(
@@ -182,11 +196,18 @@ def _should_skip_model_call_for_reference_handoff(
# Fallback final_response for the sole-handoff skip (#80622); finalize_turn appends it as a # Fallback final_response for the sole-handoff skip (#80622); finalize_turn appends it as a
# fresh assistant row, so it must not replay the last assistant text. # fresh assistant row, so it must not replay the last assistant text.
# Deliberately NOT a replay of the last assistant text: finalize_turn's non-assistant-tail chokepoint
# (#43849) appends final_response as a fresh assistant row, so recovering the previous turn's prose here
# would duplicate it in the durable transcript AND re-deliver it to the user as if it were this turn's
# answer. A short status is honest and idempotent.
_HANDOFF_SKIP_FINAL_RESPONSE = ( _HANDOFF_SKIP_FINAL_RESPONSE = (
"Context was compacted. The previous response is complete — awaiting your next message." "Context was compacted. The previous response is complete — awaiting your next message."
) )
# Terminal final_response when compression timed out while the request was still oversized (#98722). # Terminal final_response when compression timed out while the request was still oversized (#98722).
# Terminal final_response for a turn ended because context compression hit its host progress-aware timeout
# while the request was still oversized (#98722, salvaged from #98741). Sending the unchanged request would
# only bounce off the provider's overflow error and re-enter compression in the same turn.
_COMPRESSION_TIMEOUT_FINAL_RESPONSE = ( _COMPRESSION_TIMEOUT_FINAL_RESPONSE = (
"Context compression timed out without reducing this conversation. No messages were " "Context compression timed out without reducing this conversation. No messages were "
"dropped. Start a fresh session with /new, or check auxiliary.compression before retrying /compress." "dropped. Start a fresh session with /new, or check auxiliary.compression before retrying /compress."
@@ -228,6 +249,12 @@ def _is_interpreter_shutdown_error(exc: Exception) -> bool:
"""True for a fatal interpreter-shutdown RuntimeError. The RuntimeError type gate """True for a fatal interpreter-shutdown RuntimeError. The RuntimeError type gate
stays here: a ValueError carrying similar text must not match (#93269).""" stays here: a ValueError carrying similar text must not match (#93269)."""
if isinstance(exc, RuntimeError): if isinstance(exc, RuntimeError):
# ── Interpreter finalization: abandon immediately ── The process is exiting (TUI quit, SIGTERM,
# one-shot done) while this turn — typically the post-turn review fork's daemon thread — is
# mid-flight. Retries, credential rotation, and fallbacks are all futile ("cannot schedule new
# futures..."), and the buffered ⚠️/❌ retry trace spams the shell after the TUI already exited. End
# the turn with a single log line: no print, no traceback, no debug dump, no retry. Same class as
# cron delivery (#55924/#58720) and concurrent tool submission — shared predicate.
from tools.interpreter_shutdown import interpreter_shutting_down from tools.interpreter_shutdown import interpreter_shutting_down
return interpreter_shutting_down(exc) return interpreter_shutting_down(exc)
return False return False
@@ -296,6 +323,12 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text
# Transcript shows the user's own words; the provider replays the scaffolded form. # Transcript shows the user's own words; the provider replays the scaffolded form.
append_message(messages, {"role": "user", "content": text, "api_content": correction}) append_message(messages, {"role": "user", "content": text, "api_content": correction})
# Stateful scrubber for <memory-context> spans split across stream deltas (#5719). sanitize_context()
# alone can't survive chunk boundaries because the block regex needs both tags in one string.
# Stateful scrubber for reasoning/thinking tags in streamed deltas (#17924). Replaces the per-delta
# _strip_think_blocks regex that destroyed downstream state (e.g. MiniMax-M2.7 streaming '<think>' as
# delta1 and 'Let me check' as delta2 — the regex erased delta1, so downstream state machines never
# learned a block was open and leaked delta2 as content).
agent._current_streamed_assistant_text = "" agent._current_streamed_assistant_text = ""
agent._stream_needs_break = True agent._stream_needs_break = True
@@ -991,6 +1024,14 @@ def _provider_overflow_exhausted_result(
"remains over threshold at ~%s tokens.", "remains over threshold at ~%s tokens.",
agent.log_prefix, max_compression_attempts, f"{request_pressure_tokens:,}", agent.log_prefix, max_compression_attempts, f"{request_pressure_tokens:,}",
) )
# Host progress-aware timeout (#98722, salvaged from #98741): the provider proved the request does not
# fit, but this recovery pass spent the full wait budget without a committed summary. Re-sending the
# unchanged request would bounce off the same overflow error and re-enter compression in the same turn.
# End the turn with the typed recovery contract instead — transcript intact, no further doomed provider
# sends.
# Prior <3 retries (or an earlier successful tool batch) leave a tool-result tail. Closing it here
# matches interrupt aborts (#48879 / #52592) so the next user turn is not tool→user for strict
# providers.
agent._persist_session(messages, conversation_history) agent._persist_session(messages, conversation_history)
return _partial_turn_result( return _partial_turn_result(
"Context length exceeded: compression could not reduce the rebuilt request below the safe threshold.", "Context length exceeded: compression could not reduce the rebuilt request below the safe threshold.",
+3
View File
@@ -119,6 +119,7 @@ def _build_subprocess_env() -> dict[str, str]:
# Copilot ACP drives a model and needs LLM provider credentials; the central helper still # Copilot ACP drives a model and needs LLM provider credentials; the central helper still
# strips Tier-1 secrets (bot tokens, GitHub auth, infra). # strips Tier-1 secrets (bot tokens, GitHub auth, infra).
# See #29157.
env = hermes_subprocess_env(inherit_credentials=True) env = hermes_subprocess_env(inherit_credentials=True)
env["HOME"] = _resolve_home_dir() env["HOME"] = _resolve_home_dir()
apply_subprocess_home_env(env) apply_subprocess_home_env(env)
@@ -303,6 +304,8 @@ class CopilotACPClient:
try: try:
from hermes_cli._subprocess_compat import windows_hide_flags # hide the Windows console flash (#56747); pipes intact for the ACP wire from hermes_cli._subprocess_compat import windows_hide_flags # hide the Windows console flash (#56747); pipes intact for the ACP wire
# Hide the console the CLI child would otherwise flash on Windows (#56747). Hide-only — stdio
# pipes stay intact for the ACP wire.
proc = subprocess.Popen( proc = subprocess.Popen(
[self._acp_command] + self._acp_args, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, [self._acp_command] + self._acp_args, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
text=True, encoding='utf-8', errors='replace', bufsize=1, cwd=self._acp_cwd, env=_build_subprocess_env(), text=True, encoding='utf-8', errors='replace', bufsize=1, cwd=self._acp_cwd, env=_build_subprocess_env(),
+11
View File
@@ -146,6 +146,14 @@ FAILURE_REASON_BILLING_UNVERIFIED = "billing_unverified"
# every model call; on Windows several processes share one rotating log behind # every model call; on Windows several processes share one rotating log behind
# a cross-process lock, and per-selection logging stormed that lock, pegged a # a cross-process lock, and per-selection logging stormed that lock, pegged a
# core, and stalled the event loop (Desktop backend readiness timeouts). # core, and stalled the event loop (Desktop backend readiness timeouts).
# Credential selection runs on a hot path (every model call, plus auxiliary tasks like
# compression/moa/titles), so when a pool is empty or fully exhausted the un-throttled log fires on *every*
# selection. On Windows several Hermes processes share one rotating log guarded by concurrent-log-handler's
# cross-process lock; that per-selection volume storms the lock (``RuntimeError: Cannot acquire lock after
# 20 attempts``), pegs a core, and stalls the asyncio event loop long enough to fail the Desktop backend
# readiness handshake ("Timed out connecting to Hermes backend after 15000ms"). Logging the condition at
# most once per window preserves the signal while removing the storm — same class of fix as the warn-once
# dedup in #58265.
NO_AVAILABLE_ENTRIES_LOG_THROTTLE_SECONDS = 60.0 NO_AVAILABLE_ENTRIES_LOG_THROTTLE_SECONDS = 60.0
# Pool key prefix for custom OpenAI-compatible endpoints: all share # Pool key prefix for custom OpenAI-compatible endpoints: all share
@@ -731,6 +739,8 @@ def _write_through_provider_state_to_global_root(
a failed write-through degrades to root-stale and must never break the a failed write-through degrades to root-stale and must never break the
profile's own successful save. Mirrors profile's own successful save. Mirrors
``hermes_cli.auth._write_through_xai_oauth_to_global_root``. ``hermes_cli.auth._write_through_xai_oauth_to_global_root``.
See #48415.
""" """
try: try:
global_path = _guarded_global_root(auth_mod._global_auth_file_path()) global_path = _guarded_global_root(auth_mod._global_auth_file_path())
@@ -2717,6 +2727,7 @@ def _seed_custom_pool(pool_key: str, entries: List[PooledCredential]) -> Tuple[b
# The pool may be keyed under the durable ``providers.<key>`` # The pool may be keyed under the durable ``providers.<key>``
# slug or legacy ``custom:<name>``; accept any candidate, or # slug or legacy ``custom:<name>``; accept any candidate, or
# seeding is skipped when the pool holds the other identity. # seeding is skipped when the pool holds the other identity.
# Check if this model's base_url matches our custom provider. See #100413.
matched_keys = { matched_keys = {
str(key).strip().lower() for key in custom_provider_pool_key_candidates(model_base_url) str(key).strip().lower() for key in custom_provider_pool_key_candidates(model_base_url)
} }
+1
View File
@@ -33,6 +33,7 @@ DEFAULT_KEEP = 5
# is the backup dir itself; .git is repository metadata — rolling it back breaks git tracking, and snapshots that include it grow # is the backup dir itself; .git is repository metadata — rolling it back breaks git tracking, and snapshots that include it grow
# with the full history (once backups are committed back, each snapshot contains the prior ones: 38MB of skills inflated to 24GB # with the full history (once backups are committed back, each snapshot contains the prior ones: 38MB of skills inflated to 24GB
# in weeks). The tar filter in ``snapshot_skills`` applies the same set to nested paths, so a nested ``.git`` is skipped too. # in weeks). The tar filter in ``snapshot_skills`` applies the same set to nested paths, so a nested ``.git`` is skipped too.
# See #91449.
_EXCLUDE_TOP_LEVEL = {".curator_backups", ".hub", ".git"} _EXCLUDE_TOP_LEVEL = {".curator_backups", ".hub", ".git"}
# Snapshot id: UTC ISO with colons replaced by dashes (Windows-safe filename); optional ``-NN`` suffix for same-second snapshots. # Snapshot id: UTC ISO with colons replaced by dashes (Windows-safe filename); optional ``-NN`` suffix for same-second snapshots.
+12 -1
View File
@@ -41,6 +41,8 @@ _LOOP_BLOCKED_DUMP_GRACE_S = 5.0
# ``Event.wait`` is a C-level block: KeyboardInterrupt / SetAsyncExc only land when the # ``Event.wait`` is a C-level block: KeyboardInterrupt / SetAsyncExc only land when the
# thread returns to Python, so the sync wait is sliced to observe /stop or SIGINT promptly. # thread returns to Python, so the sync wait is sliced to observe /stop or SIGINT promptly.
# Slice the wait so a /stop or SIGINT during a bounded sync call is observed within this window rather than
# at the full deadline (#94285, tools/test_local_interrupt_cleanup).
_BOUNDED_SYNC_WAIT_SLICE_S = 0.2 _BOUNDED_SYNC_WAIT_SLICE_S = 0.2
@@ -165,6 +167,12 @@ def resolve_timeout(key: str, *, default: Optional[float], env_var: Optional[str
# second timer dumps all thread stacks when the loop provably failed to process the expiry. # second timer dumps all thread stacks when the loop provably failed to process the expiry.
# --------------------------------------------------------------------------- Bounded execution — async
# flavor. Generalizes plugins/platforms/telegram/adapter.py:_await_with_thread_deadline (the #63309 fix):
# the deadline is driven by a daemon threading.Timer so a blocked event loop cannot disable it, and a second
# timer dumps all thread stacks when the loop provably failed to process the expiry — the one piece of
# information loop-blocked hangs otherwise never surface.
# ---------------------------------------------------------------------------
def _consume_abandoned(task: "asyncio.Future[Any]") -> None: def _consume_abandoned(task: "asyncio.Future[Any]") -> None:
"""Observe an abandoned task's outcome so it never logs 'never retrieved'.""" """Observe an abandoned task's outcome so it never logs 'never retrieved'."""
try: try:
@@ -282,7 +290,10 @@ def run_bounded_sync(
"""Run ``fn`` in a daemon worker thread under a wall-clock deadline; exceptions re-raise in """Run ``fn`` in a daemon worker thread under a wall-clock deadline; exceptions re-raise in
the caller. On expiry the worker is **abandoned** (every timeout leaks one daemon thread, so the caller. On expiry the worker is **abandoned** (every timeout leaks one daemon thread, so
do NOT use per-item in hot loops) and ``on_timeout`` runs best-effort in the caller's thread. do NOT use per-item in hot loops) and ``on_timeout`` runs best-effort in the caller's thread.
The worker runs under ``contextvars.copy_context()`` so secret scope / session id survive.""" The worker runs under ``contextvars.copy_context()`` so secret scope / session id survive.
See #94285.
"""
timeout_s = clamp_timeout(timeout) timeout_s = clamp_timeout(timeout)
start = time.monotonic() start = time.monotonic()
if timeout_s is None: if timeout_s is None:
+15
View File
@@ -140,6 +140,9 @@ _USAGE_LIMIT_TRANSIENT_SIGNALS = (
# Anthropic's "request_too_large" type without one). # Anthropic's "request_too_large" type without one).
_PAYLOAD_TOO_LARGE_PATTERNS = ( _PAYLOAD_TOO_LARGE_PATTERNS = (
"request entity too large", "payload too large", "error code: 413", "request_too_large", "request entity too large", "payload too large", "error code: 413", "request_too_large",
# Normally arrives with an HTTP 413 status (handled by the status path), but aggregators/proxies can
# re-wrap it into a plain message with no status attribute — route it to the same compression recovery.
# (port of anomalyco/opencode#37848)
"request exceeds the maximum size", "request exceeds the maximum size",
) )
@@ -181,6 +184,8 @@ _CONTEXT_OVERFLOW_PATTERNS = (
"超过最大长度", "上下文长度", "超过最大长度", "上下文长度",
"tokens in request more than max tokens allowed", "tokens in request more than max tokens allowed",
"input is too long", "max input token", "input token", "exceeds the maximum number of input tokens", "input is too long", "max input token", "input token", "exceeds the maximum number of input tokens",
# Together/Fireworks-style: "Input length 131393 exceeds the maximum allowed input length of 131040
# tokens." No other pattern in this list matches that wording. (port of anomalyco/opencode#37848)
"maximum allowed input length", "maximum allowed input length",
) )
@@ -548,6 +553,16 @@ def _by_transport(c: _Ctx) -> Optional[Verdict]:
if any(p in msg for p in _SERVER_DISCONNECT_PATTERNS) and not c.status_code: if any(p in msg for p in _SERVER_DISCONNECT_PATTERNS) and not c.status_code:
# Reasoning models: far more likely the gateway idle-killed a long # Reasoning models: far more likely the gateway idle-killed a long
# thinking stream — never compress on a phantom overflow (#52310). # thinking stream — never compress on a phantom overflow (#52310).
# Reasoning-model override: a transport disconnect on a reasoning model is much more likely the
# upstream proxy idle-killing a long thinking stream than a true context overflow — even on large
# sessions. The default disconnect+large-session routing below would otherwise send the user into
# the compression branch (should_compress=True) and silently delete conversation history on a
# phantom context-length error. Reasoning models have multi-minute thinking phases that routinely
# exceed the cloud gateway's idle window (NVIDIA NIM ~120s — first-party repro at
# NVIDIA/NemoClaw#4846; OpenAI worker / Anthropic stream-idle similar). The per-reasoning-model
# stale-timeout floor in agent/reasoning_timeouts.py raises the stale-detector threshold to tolerate
# long thinking, so a true transport-layer failure here is recoverable via the retry path — not via
# context compression. Reclassify as timeout. (Part 1 of Fixes #52310.)
from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor
if get_reasoning_stale_timeout_floor(c.model) is not None: if get_reasoning_stale_timeout_floor(c.model) is not None:
return _V_TIMEOUT return _V_TIMEOUT
+26 -1
View File
@@ -84,7 +84,13 @@ def bare_gemini_model_id(model: str) -> str:
def gemini_requires_tool_call_ids(model: str) -> bool: def gemini_requires_tool_call_ids(model: str) -> bool:
"""Gemini 3+ needs explicit functionCall/functionResponse ids so replayed parallel tool calls """Gemini 3+ needs explicit functionCall/functionResponse ids so replayed parallel tool calls
pair with their responses; 2.x rejects the field.""" pair with their responses; 2.x rejects the field.
Gemini 3+ models require explicit tool call IDs in replayed history — without them, multi-tool turns can
be rejected or mismatched. Older Gemini models (2.x) reject unexpected ``id`` fields, so this is gated
on the major version. Mirrors earendil-works/pi#7494 (their fix for the same class of bug in the
google-shared converter).
"""
match = re.match(r"gemini-(\d+)", bare_gemini_model_id(model).lower()) match = re.match(r"gemini-(\d+)", bare_gemini_model_id(model).lower())
return match is not None and int(match.group(1)) >= 3 return match is not None and int(match.group(1)) >= 3
@@ -242,6 +248,10 @@ def _translate_tool_result_to_gemini(
parsed = json.loads(content) if content.strip().startswith(("{", "[")) else None parsed = json.loads(content) if content.strip().startswith(("{", "[")) else None
except json.JSONDecodeError: except json.JSONDecodeError:
parsed = None parsed = None
# Gemini 3 resolves JSON-Schema ``$ref`` pointers inside a functionResponse.response payload and rejects
# unknown references with HTTP 400 INVALID_ARGUMENT ("referenced name '#/$defs/...' does not match a
# display_name"; see vercel/ai#14369). A tool result that is itself a JSON Schema (e.g. tool_describe
# output for an MCP tool) must therefore be forwarded as opaque text, not as a structured response.
structured = isinstance(parsed, dict) and not _looks_like_json_schema(parsed) structured = isinstance(parsed, dict) and not _looks_like_json_schema(parsed)
function_response: Dict[str, Any] = {"name": name, "response": parsed if structured else {"output": content}} function_response: Dict[str, Any] = {"name": name, "response": parsed if structured else {"output": content}}
if include_ids and tool_call_id: if include_ids and tool_call_id:
@@ -264,6 +274,17 @@ def _merge_alternating(contents: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
functionResponse + functionResponse still merge); 3) the split pair stays API-valid via an interposed functionResponse + functionResponse still merge); 3) the split pair stays API-valid via an interposed
placeholder model turn.""" placeholder model turn."""
merged: List[Dict[str, Any]] = [] merged: List[Dict[str, Any]] = []
# Compatibility contract for native Gemini generateContent: 1) Same-role adjacent contents still merge
# in general (strict user/model alternation for ordinary text turns and parallel tool-result grouping;
# consecutive same-role contents are rejected with HTTP 400 "Please ensure that multiturn requests
# alternate between user and model"). 2) Exception: do NOT fuse a human user text turn into a preceding
# user content that only carries functionResponse parts (or vice versa). Gemini 3 accepts that fold with
# HTTP 200 but then reads the trailing text as a continuation of the tool result — it returns an empty
# candidate or "finishes the user's sentence" instead of answering (same defect gemini-cli fixed in
# google-gemini/gemini-cli#28700). 3) Because rule 1's HTTP 400 makes two consecutive user contents
# unsafe to emit (#55125 — the reason this merge exists), the split pair is kept API-valid by
# interposing a placeholder model turn between the functionResponse content and the human text content,
# mirroring gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER repair.
for content in contents: for content in contents:
prev = merged[-1] if merged else None prev = merged[-1] if merged else None
same_role = prev is not None and prev["role"] == content["role"] same_role = prev is not None and prev["role"] == content["role"]
@@ -677,6 +698,10 @@ class AsyncGeminiNativeClient:
self.api_key, self.base_url = sync_client.api_key, sync_client.base_url self.api_key, self.base_url = sync_client.api_key, sync_client.base_url
self.chat = SimpleNamespace(completions=SimpleNamespace(create=self._create_chat_completion)) self.chat = SimpleNamespace(completions=SimpleNamespace(create=self._create_chat_completion))
# Expose the underlying sync client as _real_client so the auxiliary cache's eviction-by-leaf-client
# helper (#23482) can find and drop this async entry when the sync GeminiNativeClient is poisoned.
# GeminiNativeClient is itself the leaf (no OpenAI client beneath it), so we point at the sync_client
# directly.
async def _create_chat_completion(self, **kwargs: Any) -> Any: async def _create_chat_completion(self, **kwargs: Any) -> Any:
result = await asyncio.to_thread(self._sync.chat.completions.create, **kwargs) result = await asyncio.to_thread(self._sync.chat.completions.create, **kwargs)
return self._async_stream(result) if kwargs.get("stream") else result return self._async_stream(result) if kwargs.get("stream") else result
+10 -1
View File
@@ -212,7 +212,11 @@ def _resolve_inference_base_url(cfg: Optional[Dict[str, Any]], provider: str) ->
def _resolve_inference_api_key(cfg: Optional[Dict[str, Any]], provider: str) -> str: def _resolve_inference_api_key(cfg: Optional[Dict[str, Any]], provider: str) -> str:
"""Best-effort API key, resolved like :func:`_resolve_inference_base_url` so it """Best-effort API key, resolved like :func:`_resolve_inference_base_url` so it
matches the base URL actually probed; otherwise the local server-type probe hits matches the base URL actually probed; otherwise the local server-type probe hits
a keyed remote endpoint without Authorization and sprays 401s on every image turn.""" a keyed remote endpoint without Authorization and sprays 401s on every image turn.
Mirrors :func:`_resolve_inference_base_url`'s resolution order (runtime value, then ``model.api_key``,
then the providers blocks) so the key matches the base URL actually being probed. See #89863.
"""
return _resolve_inference_value(cfg, provider, "api_key", runtime_ok=lambda _: True) return _resolve_inference_value(cfg, provider, "api_key", runtime_ok=lambda _: True)
@@ -270,6 +274,11 @@ def _probe_models_dev(provider: str, model: str, cfg: Optional[Dict[str, Any]])
The fetch is cached (4h TTL) and backoff-limited.""" The fetch is cached (4h TTL) and backoff-limited."""
from agent.models_dev import get_model_capabilities from agent.models_dev import get_model_capabilities
# allow_network=True on purpose: vision-capability lookup runs when an image actually needs routing (not
# per turn), and the #31179 text-only-main guard depends on catalog data — a cold cache returning
# "unknown" would fall back to attempting the call and reintroduce the bug. This preserves the
# historical network-on-cold-cache behavior for this one path; the fetch is cached (4h TTL) and
# backoff-limited after failures.
caps = get_model_capabilities(provider, model, allow_network=True) caps = get_model_capabilities(provider, model, allow_network=True)
return None if caps is None else bool(caps.supports_vision) return None if caps is None else bool(caps.supports_vision)
+7 -1
View File
@@ -16,7 +16,11 @@ _SKILL_TOOLS = {"skill_view", "skill_manage"}
def _fmt_est_cost(est_cost: float) -> str: def _fmt_est_cost(est_cost: float) -> str:
"""Shared label helper so sub-cent totals render at 4dp, not "~$0.00".""" """Shared label helper so sub-cent totals render at 4dp, not "~$0.00".
Routes through ``format_cost_label`` so sub-cent aggregates render at 4dp instead of collapsing to
"~$0.00" (#79220 bug class — the same dishonesty this module's cost buckets exist to fix, #77223).
"""
return format_cost_label(Decimal(str(est_cost))) return format_cost_label(Decimal(str(est_cost)))
@@ -450,6 +454,8 @@ class InsightsEngine:
@staticmethod @staticmethod
def _cost_lines(o: Dict, templates: tuple) -> List[str]: def _cost_lines(o: Dict, templates: tuple) -> List[str]:
"""One formatted line per non-zero cost bucket (estimated, included, unknown).""" """One formatted line per non-zero cost bucket (estimated, included, unknown)."""
# Cost breakdown — surface the three buckets so subscription-included and unknown-cost sessions are
# visible instead of silently collapsing to $0. See #77223.
est_cost = o.get("estimated_cost", 0.0) est_cost = o.get("estimated_cost", 0.0)
values = (_fmt_est_cost(est_cost) if est_cost > 0 else "", o.get("included_cost_sessions", 0), o.get("unknown_cost_sessions", 0)) values = (_fmt_est_cost(est_cost) if est_cost > 0 else "", o.get("included_cost_sessions", 0), o.get("unknown_cost_sessions", 0))
return [tpl.format(v) for tpl, v in zip(templates, values) if v] return [tpl.format(v) for tpl, v in zip(templates, values) if v]
+5
View File
@@ -111,6 +111,11 @@ def delete_node(node_id: str) -> dict[str, Any]:
def _delete_skill(name: str) -> dict[str, Any]: def _delete_skill(name: str) -> dict[str, Any]:
from tools import skill_usage from tools import skill_usage
# Pin must be respected by autonomous maintenance. The curator already skips pinned skills from every
# auto-transition; the background review fork is the same kind of autonomous, no-user-present actor, so
# it must not write to a pinned skill either (issue #25839). This is stricter than the foreground
# ``_pinned_guard`` (which only blocks deletion) precisely because there is no user in the loop to
# consent to an edit here.
if skill_usage.get_record(name).get("pinned"): if skill_usage.get_record(name).get("pinned"):
return {"ok": False, "message": f"'{name}' is pinned — unpin it first (hermes curator unpin {name})"} return {"ok": False, "message": f"'{name}' is pinned — unpin it first (hermes curator unpin {name})"}
ok, message = skill_usage.archive_skill(name) ok, message = skill_usage.archive_skill(name)
+14
View File
@@ -127,6 +127,7 @@ def inject_memory_provider_tools(agent: Any) -> int:
if not memory_provider_tools_exposed(agent): if not memory_provider_tools_exposed(agent):
# Say so once: a silent 0 leaves the provider looking "half on" with no clue which # Say so once: a silent 0 leaves the provider looking "half on" with no clue which
# config key (platform_toolsets / disabled_toolsets) gated it. # config key (platform_toolsets / disabled_toolsets) gated it.
# See #81014.
_providers = [p for p in getattr(memory_manager, "providers", None) or [] _providers = [p for p in getattr(memory_manager, "providers", None) or []
if getattr(p, "name", "") != "builtin"] if getattr(p, "name", "") != "builtin"]
if _providers: if _providers:
@@ -344,6 +345,9 @@ class MemoryManager:
# Core tool names are reserved: built-ins always win at agent init, so a shadowing # Core tool names are reserved: built-ins always win at agent init, so a shadowing
# provider tool would linger in ``_tool_to_provider`` and hijack dispatch. # provider tool would linger in ``_tool_to_provider`` and hijack dispatch.
# ``clarify``, ``delegate_task``). Reject it here, at the door, so it never enters the routing table
# at all — matching the built-ins-always-win invariant used by the TTS/browser/search provider
# registries. See #40466.
from toolsets import _HERMES_CORE_TOOLS from toolsets import _HERMES_CORE_TOOLS
for raw_schema in provider.get_tool_schemas(): for raw_schema in provider.get_tool_schemas():
@@ -592,6 +596,16 @@ class MemoryManager:
``on_session_end`` (LLM-bound, seconds) must run strictly BEFORE ``on_session_switch`` rebinds ``on_session_end`` (LLM-bound, seconds) must run strictly BEFORE ``on_session_switch`` rebinds
provider state; an ad-hoc thread raced the inline switch and misattributed transcripts. provider state; an ad-hoc thread raced the inline switch and misattributed transcripts.
Running extraction inline blocked the /new command for the whole LLM round-trip (#16454); running it
on an ad-hoc thread raced the inline switch — providers key off internal state, so a late
``on_session_end`` ran against post-switch bindings (transcript misattributed to the new session id,
double-ingest of the old turn buffer, new-session buffers cleared).
Submitting BOTH hooks as one task on the manager's single background worker gives both properties at
a single chokepoint: the caller returns immediately, and the worker's FIFO order serializes
end→switch against every other provider write (per-turn ``sync_all``, prefetches), which already
share the same worker. If the executor is unavailable, ``_submit_background`` degrades to inline
execution — the pre-#16454 synchronous behavior, slow but correct.
""" """
if not self._providers: if not self._providers:
return return
+53
View File
@@ -253,6 +253,7 @@ _IMAGE_REJECTION_PHRASES = (
"does not support images", "does not support image input", "does not support multimodal", "does not support images", "does not support image input", "does not support multimodal",
"does not support vision", "model does not support image", "does not support vision", "model does not support image",
# DashScope-style gateways reject non-text blocks with this generic body. # DashScope-style gateways reject non-text blocks with this generic body.
# Some OpenAI-compatible endpoints (e.g. (issue #57948)
"unexpected item type in content", "unexpected item type in content",
# ChatGPT-account Codex backend rejects data:image URLs in input_image; keyed on the # ChatGPT-account Codex backend rejects data:image URLs in input_image; keyed on the
# field-path apostrophe so other URL errors don't false-trip. Second: its wording for # field-path apostrophe so other URL errors don't false-trip. Second: its wording for
@@ -262,8 +263,17 @@ _IMAGE_REJECTION_PHRASES = (
"unknown variant `image_url`, expected `text`", "unknown variant image_url, expected text", "unknown variant `image_url`, expected `text`", "unknown variant image_url, expected text",
# OpenRouter HTTP 404 when no upstream endpoint accepts image input (passes the 4xx # OpenRouter HTTP 404 when no upstream endpoint accepts image input (passes the 4xx
# gate; without this the gateway queue wedges behind the stuck turn). # gate; without this the gateway queue wedges behind the stuck turn).
# Without this phrase the agent never strips the images, the retry loop re-sends the same rejected
# request until exhaustion, and the gateway leaves every subsequent message queued behind the stuck turn
# — the P1 in issue #21160.
"no endpoints found that support image input", "no endpoints found that support image input",
# Kimi/Moonshot et al. reject truncated/corrupt image bytes baked into history. # Kimi/Moonshot et al. reject truncated/corrupt image bytes baked into history.
# Kimi / Moonshot / other OpenAI-compatible Chinese providers reject truncated or corrupt image bytes
# with HTTP 400 "Invalid request: prepare image failed ... failed to decode image: invalid or
# unsupported image format". Like the Codex case above, the bad bytes are baked into immutable
# conversation history and re-sent on every retry, wedging the session. Strip the images so the turn
# recovers instead of exhausting retries. (issue #76884; complements the proactive full-decode
# validation in tools/vision_tools._normalize_to_supported_image)
"failed to decode image", "failed to decode image",
) )
@@ -305,6 +315,17 @@ def _tc_set(tc: Any, key: str, value: Any) -> None:
tc.__setitem__(key, value) if isinstance(tc, dict) else setattr(tc, key, value) tc.__setitem__(key, value) if isinstance(tc, dict) else setattr(tc, key, value)
# --------------------------------------------------------------------------- call_id policy — single owner
# (audit F4, incident chain I4) ---------------------------------------------------------------------------
# Three forked policy sites converged here: * agent/codex_responses_adapter.py `_deterministic_call_id` —
# hash synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2). *
# run_agent.AIAgent._get_tool_call_id_static — `call_id or id` coalescing for dicts and SDK objects. *
# run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with deterministic `_d<n>` suffixes
# (#58327 loss class). NOT consolidated (different scheme on purpose):
# agent/transports/codex_event_projector._deterministic_call_id maps codex app-server ITEM ids
# (`codex_<type>_<item_id>`), not chat tool-call content; merging the two would change ids and invalidate
# prompt caches. HARD INVARIANT: everything here must stay deterministic (never uuid4) and byte-identical
# for existing inputs — these ids feed prompt-cache prefixes.
def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str: def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Deterministic call_id fallback when the API omits one (random ids would break caching).""" """Deterministic call_id fallback when the API omits one (random ids would break caching)."""
seed = f"{fn_name}:{arguments}:{index}" seed = f"{fn_name}:{arguments}:{index}"
@@ -392,6 +413,19 @@ def uniquify_tool_call_ids(tool_calls: list) -> list:
# empty-string pads → " ". Strict side (400/422 "Extra inputs are not permitted"): everyone # empty-string pads → " ". Strict side (400/422 "Extra inputs are not permitted"): everyone
# else — Mistral, Cerebras, Groq, SambaNova, … Strip the key entirely, even a one-space pad. # else — Mistral, Cerebras, Groq, SambaNova, … Strip the key entirely, even a one-space pad.
# --------------------------------------------------------------------------- reasoning_content policy —
# single owner (audit F4) --------------------------------------------------------------------------- The
# strip-vs-repad decision was previously forked across the wire files in separate incident commits
# (2b3a4f0af8 strip for strict providers, b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi
# pad). The POLICY — which provider direction gets which treatment — lives here as one rule table + apply
# functions; adapters keep only SYNTAX mapping (e.g. anthropic_adapter turning reasoning_content into a
# thinking block). Direction table: require-side (echo-back enforced; replays 400 without the field): kimi
# — provider kimi-coding/kimi-coding-cn, or host api.kimi.com / moonshot.ai / moonshot.cn. Host-driven on
# purpose: aggregators re-exporting kimi models reject the echo. deepseek — provider "deepseek", model
# contains "deepseek", or host api.deepseek.com (#15250; V4 rejects empty-string pads, hence the " "
# single-space pad, #17341). mimo — provider "xiaomi", model contains "mimo", or host *.xiaomimimo.com.
# strict side (field rejected with 400/422 "Extra inputs are not permitted"): everyone else — Mistral,
# Cerebras, Groq, SambaNova, … (#45655). Strip the key entirely, even a single-space pad.
_REASONING_ECHO_RULES: tuple = ( _REASONING_ECHO_RULES: tuple = (
# (family, exact providers (raw), exact providers (lowered), model substrings (lowered), hosts) # (family, exact providers (raw), exact providers (lowered), model substrings (lowered), hosts)
("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (), ("api.kimi.com", "moonshot.ai", "moonshot.cn")), ("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (), ("api.kimi.com", "moonshot.ai", "moonshot.cn")),
@@ -449,6 +483,15 @@ def apply_reasoning_content_policy(source_msg: dict, api_msg: dict, needs_thinki
api_msg.pop("reasoning_content", None) api_msg.pop("reasoning_content", None)
return return
existing, reasoning = source_msg.get("reasoning_content"), source_msg.get("reasoning") existing, reasoning = source_msg.get("reasoning_content"), source_msg.get("reasoning")
# 1. Explicit reasoning_content already set. When the active provider enforces the thinking-mode
# echo-back (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their own space-placeholder
# written at creation time and any valid reasoning from the same provider. Sessions persisted BEFORE
# #17341 have empty-string placeholders pinned at creation time; DeepSeek V4 Pro rejects those with
# HTTP 400, so upgrade "" → " " on replay. When the active provider does NOT enforce echo-back, strip
# the field entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq, SambaNova, …)
# reject ANY reasoning_content key in input messages with HTTP 400/422 ("Extra inputs are not
# permitted"), even an empty string or a single-space pad. Stripping here covers the rebuild path;
# ``reapply_reasoning_echo`` covers the already-built api_messages path. Refs #45655.
if isinstance(existing, str): if isinstance(existing, str):
# Explicit value: preserve verbatim, upgrading legacy "" to " " (DeepSeek V4 400s on ""). # Explicit value: preserve verbatim, upgrading legacy "" to " " (DeepSeek V4 400s on "").
api_msg["reasoning_content"] = existing or " " api_msg["reasoning_content"] = existing or " "
@@ -475,6 +518,16 @@ def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int:
for api_msg in api_messages: for api_msg in api_messages:
if api_msg.get("role") != "assistant": if api_msg.get("role") != "assistant":
continue continue
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content' for providers that use the
# internal 'reasoning' key. This must happen before the unconditional empty-string fallback so
# genuine reasoning content is not overwritten (#15812 regression in PR #15478). Only promote for
# providers that enforce echo-back — strict providers reject the field (refs #45655).
# 4. DeepSeek / Kimi thinking mode: all assistant messages need reasoning_content. Inject a single
# space to satisfy the provider's requirement when no explicit reasoning content is present.
# Covers both tool-call turns (already-poisoned history with no reasoning at all) and plain text
# turns. Space (not "") because DeepSeek V4 Pro tightened validation and rejects empty string with
# HTTP 400 ("The reasoning content in the thinking mode must be passed back to the API"). Refs
# #17341.
if needs_thinking_pad: if needs_thinking_pad:
if not api_msg.get("reasoning_content"): if not api_msg.get("reasoning_content"):
apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad) apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad)
+9
View File
@@ -180,6 +180,11 @@ class MicroCompactionMixin:
# in-place pop on a live dict would be identity-skipped by the bounded flush scan; # in-place pop on a live dict would be identity-skipped by the bounded flush scan;
# flag the finalizer. # flag the finalizer.
entry.pop(_cc()._DB_PERSISTED_MARKER, None) entry.pop(_cc()._DB_PERSISTED_MARKER, None)
# Sibling of the finalize_turn pop site (#75170): this pop also strips the marker from a LIVE
# dict in place, so the bounded flush-scan cursor would identity-skip the rewritten marker and
# the defragged summary would never reach state.db. The compressor holds no agent reference, so
# raise a flag the finalizer consumes to invalidate agent._db_flush_scan_prefix. (The pop sites
# at module scope — fresh copies in strip-marker helpers — break identity and need no flag.)
self._flush_scan_cursor_invalidated = True self._flush_scan_cursor_invalidated = True
logger.info( logger.info(
"Micro-compaction defrag: rolling summary re-summarized (%d -> %d chars)", "Micro-compaction defrag: rolling summary re-summarized (%d -> %d chars)",
@@ -313,6 +318,9 @@ class MicroCompactionMixin:
self._micro_compact_tokens_saved_total -= delta or 0 self._micro_compact_tokens_saved_total -= delta or 0
self._micro_compact_passes += 1 self._micro_compact_passes += 1
# Cached reads only: the lazy properties can fire a synchronous /models probe. # Cached reads only: the lazy properties can fire a synchronous /models probe.
# The ``threshold_tokens`` / ``context_length`` properties resolve lazily and can fire a
# synchronous /models probe on first access (#32221) — telemetry must never be the thing that
# blocks a turn. Unresolved simply reports null.
threshold = self._threshold_tokens threshold = self._threshold_tokens
has_occupancy = threshold and tokens_after is not None and threshold > 0 has_occupancy = threshold and tokens_after is not None and threshold > 0
occupancy = round(tokens_after / threshold * 100, 1) if has_occupancy else None occupancy = round(tokens_after / threshold * 100, 1) if has_occupancy else None
@@ -343,6 +351,7 @@ class MicroCompactionMixin:
# Every row except the marker is a carried-forward original: archive rewind-style. # Every row except the marker is a carried-forward original: archive rewind-style.
session_db.archive_and_compact(session_id, compacted_messages, tail_count=max(0, len(compacted_messages) - 1)) session_db.archive_and_compact(session_id, compacted_messages, tail_count=max(0, len(compacted_messages) - 1))
# Shared post-commit stamp site with batch commit and proactive prune. # Shared post-commit stamp site with batch commit and proactive prune.
# See #98450.
_cc().stamp_db_persisted_markers(compacted_messages) _cc().stamp_db_persisted_markers(compacted_messages)
except Exception: except Exception:
logger.info( logger.info(
+57
View File
@@ -30,6 +30,19 @@ logger = logging.getLogger(__name__)
# Privacy filter (moa.privacy_filter: '' | display | full): PII classes agent.redact # Privacy filter (moa.privacy_filter: '' | display | full): PII classes agent.redact
# leaves alone. The phone pattern requires explicit delimiters so line numbers, # leaves alone. The phone pattern requires explicit delimiters so line numbers,
# dates, times, SHAs, IPs and versions never match. # dates, times, SHAs, IPs and versions never match.
# Advisor (reference) outputs can echo PII from the conversation — emails, phone numbers, credentials pasted
# by the user — into surfaces the user may not expect: the labelled reference blocks rendered in the UI,
# saved MoA trace files, and (in `full` mode) the guidance block injected into the aggregator prompt (issue
# #59959). Secret/credential shapes (API-key prefixes, JWTs, private keys, DB connection strings, E.164
# phone numbers) are handled by the repo's central redactor, ``agent.redact .redact_sensitive_text`` — the
# MoA filter never re-implements those. The two patterns below cover the PII classes the central redactor
# deliberately leaves alone for log/tool output (emails and formatted phone numbers). Pattern safety:
# advisory text is frequently code-review-shaped — line numbers, timestamps, git SHAs, IDs, IP addresses. A
# bare 10-digit match would mangle all of those, so the phone pattern requires clearly delimited formatting:
# a parenthesized area code and/or explicit `-`/`.` separators between groups ((555) 123-4567, 555-123-4567,
# 555.123.4567, +1 555-123-4567). Undelimited digit runs (5551234567), dates (2026-07-12), times (12:34:56),
# hex IDs, and dotted quads never match. International numbers in E.164 form (+14155551234) are already
# masked by the central redactor.
_MOA_EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b") _MOA_EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")
_MOA_PHONE_RE = re.compile( _MOA_PHONE_RE = re.compile(
r"(?<![\w.+-])" # no leading word char / dot / + / - (kills IPs, IDs, versions) r"(?<![\w.+-])" # no leading word char / dot / + / - (kills IPs, IDs, versions)
@@ -99,6 +112,9 @@ def _redact_trace_accounting(acct: Any) -> Any:
# Cold-start caches: preset and per-(provider, model) runtime are immutable for a turn. # Cold-start caches: preset and per-(provider, model) runtime are immutable for a turn.
# A MoA preset switch used to re-resolve the full config + preset + every slot's provider runtime on EACH
# create() call (once per tool-loop iteration), serially before the parallel fan-out could start — adding
# 5-30s of "frozen" latency on complex presets (#66793).
_preset_cache_lock = threading.Lock() _preset_cache_lock = threading.Lock()
_preset_cache: dict[tuple, Any] = {} _preset_cache: dict[tuple, Any] = {}
@@ -286,6 +302,9 @@ def _maybe_apply_moa_cache_control(
legacy system-and-3 fallback is used. ``cache_disabled`` is stamped onto the legacy system-and-3 fallback is used. ``cache_disabled`` is stamped onto the
stub so ``cache_ttl: off`` is honored; ``cache_ttl`` is clamped per destination. stub so ``cache_ttl: off`` is honored; ``cache_ttl`` is clamped per destination.
Returns the messages unchanged on any error. Returns the messages unchanged on any error.
``cache_disabled`` (or the live config when omitted) is stamped onto the policy stub so
``prompt_caching.cache_ttl: off`` is not bypassed by the blank-agent pattern (#76085).
""" """
try: try:
from agent.agent_runtime_helpers import anthropic_prompt_cache_policy, blank_cache_policy_stub from agent.agent_runtime_helpers import anthropic_prompt_cache_policy, blank_cache_policy_stub
@@ -353,6 +372,10 @@ def _run_reference(
# Trim to THIS model's window (advisors may be smaller than the aggregator); the # Trim to THIS model's window (advisors may be smaller than the aggregator); the
# advisory view is append-only across iterations, so cache_control lets # advisory view is append-only across iterations, so cache_control lets
# iteration N+1 replay N's cached prefix. # iteration N+1 replay N's cached prefix.
# Reference models may have a smaller window than the aggregator (e.g. kimi-k2.7-code @ 262K
# advising a glm-5.2 @ 1M conversation); without this trim the provider returns a hard HTTP 400
# which the except below silently converts to a [failed: …] note (issue #60345). Estimated AFTER the
# advisory system prompt is prepended so its tokens count against the budget too.
trimmed = _trim_messages_for_reference( trimmed = _trim_messages_for_reference(
messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache, messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache,
) )
@@ -421,6 +444,11 @@ def _trim_messages_for_reference(
body and the trailing user turn plus one preceding turn (even if still over body and the trailing user turn plus one preceding turn (even if still over
budget). ``context_length_cache`` memoizes the window per (provider, model); budget). ``context_length_cache`` memoizes the window per (provider, model);
unresolvable windows leave messages unchanged. unresolvable windows leave messages unchanged.
Reference models may have a smaller context window than the aggregator or the main conversation. Without
this trim, a reference whose window is exceeded gets a hard HTTP 400 from the provider, which
``_run_reference``'s try/except silently converts to a ``[failed: …]`` note — the MoA turn silently
degrades to fewer references (issue #60345).
""" """
if not messages or not slot.get("model"): if not messages or not slot.get("model"):
return messages return messages
@@ -480,6 +508,13 @@ def _settle_interrupted(
for future, idx in futures.items(): for future, idx in futures.items():
if results[idx] is not None: if results[idx] is not None:
continue continue
# #38922: a slow confirmation does NOT necessarily mean the send failed — but we must distinguish
# two cases via future.cancel()'s return value: cancel() == False -> the coroutine was already
# running on the gateway loop when the timeout fired; the request is in flight on the wire and
# cannot be un-sent. Re-sending via standalone would be a guaranteed DUPLICATE, so treat it as
# delivered (assume-delivered). cancel() == True -> the scheduled callback never started executing
# (loop wedged/backlogged for the full 60s), so nothing was sent. We MUST fall through to the
# standalone path or the message is silently dropped (worse than a duplicate).
cancelled = future.cancel() cancelled = future.cancel()
if not cancelled and future.done(): if not cancelled and future.done():
results[idx] = future.result() results[idx] = future.result()
@@ -753,6 +788,12 @@ def aggregate_moa_context(
Failures become model-specific notes instead of aborting the loop. Failures become model-specific notes instead of aborting the loop.
``reference_max_tokens`` caps ONLY the fan-out (capping the aggregator truncated ``reference_max_tokens`` caps ONLY the fan-out (capping the aggregator truncated
long syntheses). ``agent`` makes the fan-out interruptible. long syntheses). ``agent`` makes the fan-out interruptible.
``reference_max_tokens`` applies ONLY to the reference fan-out — the aggregator's own synthesis call is
never capped, so it always uses its model's own maximum. ``call_llm`` omits the parameter entirely when
it is ``None`` (see its docstring), which also sidesteps providers that reject ``max_tokens`` outright.
A hardcoded cap on the aggregator call previously truncated long aggregator syntheses (#53580) — passing
``reference_max_tokens`` to both calls here would silently reintroduce that regression.
""" """
reference_models = [slot for slot in reference_models if slot.get("enabled", True)] reference_models = [slot for slot in reference_models if slot.get("enabled", True)]
reference_outputs = _run_references_parallel( reference_outputs = _run_references_parallel(
@@ -1015,6 +1056,9 @@ class MoAChatCompletions:
planning_messages = peel_reference_guidance(agg_messages, str(guidance)) if guidance else agg_messages planning_messages = peel_reference_guidance(agg_messages, str(guidance)) if guidance else agg_messages
# Tri-state cache_disabled: facades built via __new__ have no _agent; forcing # Tri-state cache_disabled: facades built via __new__ have no _agent; forcing
# False would suppress the planner's config fallback. # False would suppress the planner's config fallback.
# plan_cache_sections_for_destination never mutates its inputs and always returns request-local
# copies, so the prepared state stays canonical. Tri-state: only pass a bool when a live agent
# snapshot exists. See #76085.
_agent = getattr(self, "_agent", None) _agent = getattr(self, "_agent", None)
cache_disabled, cache_ttl = _agent_cache_opts(_agent) cache_disabled, cache_ttl = _agent_cache_opts(_agent)
# Agent TTL + stable system prefix so MoA does not regress 1h → 5m. # Agent TTL + stable system prefix so MoA does not regress 1h → 5m.
@@ -1090,6 +1134,19 @@ class MoAChatCompletions:
view changes. "every_n:<N>": iteration 1 of a turn, then every Nth; in-between view changes. "every_n:<N>": iteration 1 of a turn, then every Nth; in-between
iterations return the pinned last on-cadence key (HIT: no calls, no re-emit). iterations return the pinned last on-cadence key (HIT: no calls, no re-emit).
""" """
# "user_turn" (default — cheapest cadence, #67199): advisors run ONCE per user turn; subsequent tool
# iterations reuse that turn's advice and the aggregator acts alone (the original MoA shape:
# synthesize at the start, then let the acting model work). Implemented by hashing only the prefix
# up to the LAST USER message so mid-turn growth doesn't change the signature — iteration 2+ becomes
# a cache HIT. "per_iteration": advisors re-run whenever the advisory view changes — i.e. every tool
# iteration, since the view grows with each tool result; advice tracks live task state at the cost
# of multiplying advisor latency/spend by tool-loop depth. "every_n:<N>" (N >= 2): the middle ground
# (issue #63393 — advisor fan-out multiplies latency/cost by the tool-iteration count). Advisors run
# on iteration 1 of a user turn and then every Nth tool iteration; the iterations in between REUSE
# the cached guidance from the last on-cadence run (same mechanism as user_turn's cache HIT — the
# aggregator still gets advice every iteration, it's just not refreshed against the very latest tool
# results). The iteration counter is scoped per user turn and resets on a new user message, so every
# turn starts with fresh advice.
fanout_mode = str(preset.get("fanout") or "user_turn").strip().lower() fanout_mode = str(preset.get("fanout") or "user_turn").strip().lower()
every_n = 0 every_n = 0
if fanout_mode.startswith("every_n:"): if fanout_mode.startswith("every_n:"):
+39 -1
View File
@@ -112,6 +112,11 @@ _ENDPOINT_MODEL_CACHE_TTL = 300
# server swap on the same port is re-detected; None gets the short TTL so a # server swap on the same port is re-detected; None gets the short TTL so a
# transient failure recovers in minutes without re-running the waterfall each turn. # transient failure recovers in minutes without re-running the waterfall each turn.
_ENDPOINT_PROBE_TTL_SECONDS = 3600.0 _ENDPOINT_PROBE_TTL_SECONDS = 3600.0
# A failed probe verdict (server_type is None — no known endpoint answered) is cached for a much shorter
# window: the in-memory entry exists only to keep one image-bearing turn from re-running the 5-request
# waterfall on every subsequent turn (#89863 — a keyed remote endpoint answered 401 to each leg and the None
# verdict was never cached, so every turn re-probed). Short TTL keeps a transient failure (server starting
# up, key being fixed) recoverable within minutes instead of pinning "undetected" for an hour.
_ENDPOINT_PROBE_FAILURE_TTL_SECONDS = 300.0 _ENDPOINT_PROBE_FAILURE_TTL_SECONDS = 300.0
_endpoint_probe_path_cache: Dict[str, tuple] = {} _endpoint_probe_path_cache: Dict[str, tuple] = {}
# Routable-but-dead endpoints (corp LAN off-VPN) blackhole TCP: once ANY probe paid # Routable-but-dead endpoints (corp LAN off-VPN) blackhole TCP: once ANY probe paid
@@ -674,6 +679,9 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
return result return result
# Cache the negative verdict in memory only (never on disk — a failure is often transient: server starting,
# key being fixed) so the very next turn does not re-run the whole waterfall against an endpoint that just
# answered nothing (#89863).
def _iter_nested_dicts(value: Any): def _iter_nested_dicts(value: Any):
if isinstance(value, dict): if isinstance(value, dict):
yield value yield value
@@ -778,6 +786,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any
try: try:
_ensure_requests() _ensure_requests()
# (connect, read) tuple: a flat timeout lets urllib3 block per retry stage through proxies that 403 CONNECT. # (connect, read) tuple: a flat timeout lets urllib3 block per retry stage through proxies that 403 CONNECT.
# See #46620.
response = requests.get(OPENROUTER_MODELS_URL, timeout=(5, 10), verify=_resolve_requests_verify()) response = requests.get(OPENROUTER_MODELS_URL, timeout=(5, 10), verify=_resolve_requests_verify())
response.raise_for_status() response.raise_for_status()
cache = {} cache = {}
@@ -1186,6 +1195,8 @@ def _ollama_show(server_url: str, api_key: str, bare_model: str, timeout: float
def _is_ollama_server(base_url: str, api_key: str) -> bool: def _is_ollama_server(base_url: str, api_key: str) -> bool:
try: try:
# Forward the API key: a remote API-keyed endpoint answers the probe waterfall with 401s without it,
# and an unauthorized probe can never produce a positive verdict (#89863).
return detect_local_server_type(base_url, api_key=api_key) == "ollama" return detect_local_server_type(base_url, api_key=api_key) == "ollama"
except Exception: except Exception:
return False return False
@@ -1447,7 +1458,11 @@ _CODEX_900K_SNAPSHOT_RE = re.compile(r"^\d{4}-\d{2}-\d{2}$")
def _bare_codex_slug(model: Optional[str]) -> str: def _bare_codex_slug(model: Optional[str]) -> str:
"""Lowercased slug without ``vendor/`` (display/auxiliary callers pass ``openai/gpt-5.6-sol-900k``).""" """Lowercased slug without ``vendor/`` (display/auxiliary callers pass ``openai/gpt-5.6-sol-900k``).
Display/auxiliary callers pass ids like ``openai/gpt-5.6-sol-900k``; the main-agent path normalizes the
namespace away earlier, but this resolver must accept both shapes (#92797 review).
"""
return (model or "").strip().lower().rsplit("/", 1)[-1] return (model or "").strip().lower().rsplit("/", 1)[-1]
@@ -1567,6 +1582,10 @@ def _resolve_codex_oauth_context_length_with_source(model: str, access_token: st
return bumped, source return bumped, source
return ctx, source return ctx, source
# The Codex catalog only knows the base slug (no -900k, no vendor/). # The Codex catalog only knows the base slug (no -900k, no vendor/).
# ``-900k`` variants are Hermes picker aliases — the Codex catalog only knows the base slug, so resolve
# against the stripped id. Also drop any ``vendor/`` namespace (``openai/gpt-5.6-sol-900k``): the
# main-agent path normalizes it away before reaching here, but display/auxiliary callers pass it through
# (#92797 review).
lookup_bare = _bare_codex_slug(strip_codex_context_variant_suffix(model_bare)) lookup_bare = _bare_codex_slug(strip_codex_context_variant_suffix(model_bare))
if access_token: if access_token:
live, fresh_probe = _fetch_codex_oauth_context_lengths_with_source(access_token) live, fresh_probe = _fetch_codex_oauth_context_lengths_with_source(access_token)
@@ -1650,6 +1669,10 @@ def _validate_cached_context_length(model: str, base_url: str, cached: int, is_b
_invalidate_cached_context_length(model, base_url) _invalidate_cached_context_length(model, base_url)
return bedrock_ctx return bedrock_ctx
return cached return cached
# For local endpoints, run the probe that respects configured Modelfile context values first.
# _query_local_context_length prefers num_ctx from Modelfile, while _query_ollama_api_show returns the
# GGUF training max first which can be larger and would create a false-safe window for compression
# (#63122). Non-local endpoints preserve the existing GGUF-first behavior.
if is_local_endpoint(base_url): if is_local_endpoint(base_url):
return _reconcile_local_cached_context_length(model, base_url, cached, api_key=api_key) return _reconcile_local_cached_context_length(model, base_url, cached, api_key=api_key)
return cached return cached
@@ -1740,12 +1763,17 @@ def _config_override_context_length(model: str, base_url: str, provider: str, cu
"""Steps 0b-0c: config-only overrides (never touch the network). 0b: EXPLICIT model_overrides """Steps 0b-0c: config-only overrides (never touch the network). 0b: EXPLICIT model_overrides
only — fill-gap _default entries apply inside lookup_models_dev_context once the catalog has only — fill-gap _default entries apply inside lookup_models_dev_context once the catalog has
missed, so a _default can never preempt custom_providers or live probes. 0c: custom_providers.""" missed, so a _default can never preempt custom_providers or live probes. 0c: custom_providers."""
# This is the supported self-unblock path for models with wrong context in models.dev (#84482) and for
# custom/local models (#8731).
if provider and model: if provider and model:
with contextlib.suppress(Exception): # fall through to other resolution paths with contextlib.suppress(Exception): # fall through to other resolution paths
from agent.models_dev import _override_context_window from agent.models_dev import _override_context_window
mo_ctx = _override_context_window(provider, model) mo_ctx = _override_context_window(provider, model)
if mo_ctx is not None and mo_ctx > 0: if mo_ctx is not None and mo_ctx > 0:
return mo_ctx return mo_ctx
# 0c. custom_providers per-model override — check before any probe. This closes the gap where /model
# switch and display paths used to fall back to 128K despite the user having a per-model context_length
# set. See #15779.
if custom_providers and base_url and model: if custom_providers and base_url and model:
with contextlib.suppress(Exception): # fall through to probing with contextlib.suppress(Exception): # fall through to probing
from hermes_cli.config import get_custom_provider_context_length from hermes_cli.config import get_custom_provider_context_length
@@ -1981,6 +2009,16 @@ def _strip_stale_thinking_for_estimate(messages: List[Dict[str, Any]]) -> List[D
# pinned (strong ref in the entry, so the id can't be reused and immutability makes id-equality # pinned (strong ref in the entry, so the id can't be reused and immutability makes id-equality
# value-equality); numbers/bools/None by value; dicts/lists structurally in key order (``str(shadow)`` # value-equality); numbers/bools/None by value; dicts/lists structurally in key order (``str(shadow)``
# depends on it); any other type aborts the memo. api_messages shallow-copies dicts but shares the strings. # depends on it); any other type aborts the memo. api_messages shallow-copies dicts but shares the strings.
# ``estimate_messages_tokens_rough`` is called on the full history every loop iteration (conversation_loop
# preflight), repeatedly during compaction telemetry, and inside an O(n^2) shrink loop in moa_loop. The
# per-message helpers are pure functions of the message's value, so a memo keyed on a fingerprint that
# uniquely determines the value is exactly equivalent. Fingerprint design (soundness argument): While the
# entry lives, that id cannot be reused by another object, so id-equality implies object-equality — strings
# are immutable, so value-equality too (no #50372-style aliasing). Equal fingerprints therefore imply
# deep-equal messages built from identical immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical
# estimate. Because the api_messages build shallow-copies history dicts each iteration, the copies share the
# same content strings — so unchanged history messages hit the memo even though the outer dicts are fresh
# objects every turn.
_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {} _MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {}
_MSG_TOKENS_CACHE_MAX = 4096 _MSG_TOKENS_CACHE_MAX = 4096
+20 -3
View File
@@ -498,7 +498,10 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
"""Context window in tokens for provider+model, or None if not found. An EXPLICIT ``model_overrides`` """Context window in tokens for provider+model, or None if not found. An EXPLICIT ``model_overrides``
entry wins over the catalog; ``_default`` fills the gap only when the catalog has no answer (the entry wins over the catalog; ``_default`` fills the gap only when the catalog has no answer (the
self-unblock path for wrong/missing context in models.dev). Catalog entries with context=0 are self-unblock path for wrong/missing context in models.dev). Catalog entries with context=0 are
skipped in favour of later candidates. ``allow_network`` defaults to False — runs every turn.""" skipped in favour of later candidates. ``allow_network`` defaults to False — runs every turn.
See #84482.
"""
override_ctx = _override_context_window(provider, model) override_ctx = _override_context_window(provider, model)
if override_ctx is not None: if override_ctx is not None:
return override_ctx return override_ctx
@@ -513,6 +516,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
# catalog. ``<provider>._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to # catalog. ``<provider>._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to
# models the catalog does not know and never displace catalog data. Provider keys accept the Hermes # models the catalog does not know and never displace catalog data. Provider keys accept the Hermes
# or models.dev id; model ids match exactly, then case-insensitively (mirroring catalog lookup). # or models.dev id; model ids match exactly, then case-insensitively (mirroring catalog lookup).
# Resolution semantics: 1. 2. See #84482, #8731.
_OVERRIDE_WARNED_KEYS: set = set() _OVERRIDE_WARNED_KEYS: set = set()
# Safe defaults for models absent from the catalog (tools on, vision/reasoning off, 200K context); # Safe defaults for models absent from the catalog (tools on, vision/reasoning off, 200K context);
# shared by get_model_capabilities and get_model_info so the two unknown-model paths agree. # shared by get_model_capabilities and get_model_info so the two unknown-model paths agree.
@@ -589,6 +593,7 @@ def _override_context_window(provider: str, model: str) -> Optional[int]:
return _override_int(ov, "context_window") if ov is not None else None return _override_int(ov, "context_window") if ov is not None else None
# Catalog miss — a _default override may fill the gap (#84482).
def _default_override_context(provider: str) -> Optional[int]: def _default_override_context(provider: str) -> Optional[int]:
"""Fill-gap context from a ``_default`` override, for catalog misses.""" """Fill-gap context from a ``_default`` override, for catalog misses."""
default = _default_model_override(provider) default = _default_model_override(provider)
@@ -655,7 +660,13 @@ def _entry_supports_vision(entry: Dict[str, Any]) -> bool:
def get_model_capabilities(provider: str, model: str, *, allow_network: bool = False) -> Optional[ModelCapabilities]: def get_model_capabilities(provider: str, model: str, *, allow_network: bool = False) -> Optional[ModelCapabilities]:
"""Capability metadata from the models.dev cache, or None if unresolvable. EXPLICIT ``model_overrides`` """Capability metadata from the models.dev cache, or None if unresolvable. EXPLICIT ``model_overrides``
patch catalog fields; ``_default`` fills the gap only for models the catalog does not know. Unspecified patch catalog fields; ``_default`` fills the gap only for models the catalog does not know. Unspecified
fields fall through to the catalog, or to safe defaults. ``allow_network`` defaults to False (hot path).""" fields fall through to the catalog, or to safe defaults. ``allow_network`` defaults to False (hot path).
EXPLICIT ``model_overrides`` entries (per-provider+model) win over catalog values for the fields they
set. ``_default`` entries fill the gap only for models the catalog does not know — the supported
self-unblock path for custom/local models (#8731) and for models with wrong metadata in models.dev
(#84482).
"""
models = _get_provider_models(provider, allow_network=allow_network) models = _get_provider_models(provider, allow_network=allow_network)
entry = _find_model_entry(models, model) if models is not None else None entry = _find_model_entry(models, model) if models is not None else None
raw = _apply_overrides(provider, model, entry) raw = _apply_overrides(provider, model, entry)
@@ -751,7 +762,13 @@ def get_provider_info(provider_id: str, *, allow_network: bool = True) -> Option
def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = False) -> Optional[ModelInfo]: def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = False) -> Optional[ModelInfo]:
"""Full model metadata by Hermes or models.dev provider ID (exact match, then case-insensitive), or """Full model metadata by Hermes or models.dev provider ID (exact match, then case-insensitive), or
None if not found. EXPLICIT ``model_overrides`` patch known catalog models; ``_default`` fills the gap None if not found. EXPLICIT ``model_overrides`` patch known catalog models; ``_default`` fills the gap
only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths.""" only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths.
``model_overrides`` entries use the SAME canonical schema as every other consumer (``context_window``,
``max_output_tokens``, ``supports_*``, ``model_family``) — they are translated into the catalog shape at
this boundary, and sub-dicts (``limit``, ``modalities``) are merged rather than clobbered. See #84482,
#8731.
"""
mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id) mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id)
models = _registry_models(mdev_id, allow_network=allow_network) models = _registry_models(mdev_id, allow_network=allow_network)
mid, entry = next(_iter_model_entries(models, model_id, suffix_fallback=False), (model_id, None)) if models is not None else (model_id, None) mid, entry = next(_iter_model_entries(models, model_id, suffix_fallback=False), (model_id, None)) if models is not None else (model_id, None)
+22
View File
@@ -206,6 +206,18 @@ def prune_pre_checkpoint_items(
that is itself a canonical summary carrier is read from the SOURCE and retained as a that is itself a canonical summary carrier is read from the SOURCE and retained as a
synthesized ``role="assistant"`` message. synthesized ``role="assistant"`` message.
- ``enable_summary_retention`` is a test override, not a config surface. - ``enable_summary_retention`` is a test override, not a config surface.
The server drops every input item that precedes a replayed ``compaction`` item (live-verified Aug 2026),
so sending pre-checkpoint history is dead weight AND silently erases the user's plaintext asks —
including any local-compression summary the agent already produced, which previously vanished here
because it carries ``role="assistant"``, not ``"user"`` (#90975).
A summary is never byte/character-sliced: Hermes summaries carry structural framing (handoff prefix, end
marker, merge-into-tail delimiters) that a blind slice can corrupt, so one that doesn't fit whole is
dropped instead. A summary already retained once (identical text) is never duplicated, so repeated
checkpoints stay idempotent. - ``enable_summary_retention`` is a function-level override (used by tests
and callers that need the pre-#90975 behavior back); it is not wired to a user-facing config surface.
Without ``item_sources`` (default), retention only sees what survived conversion, matching pre-#90976
behavior (#90976).
""" """
if not isinstance(items, list) or not items: if not isinstance(items, list) or not items:
return items return items
@@ -241,6 +253,9 @@ def prune_pre_checkpoint_items(
continue continue
# Source-based detection sees past a lossy conversion; it only fires # Source-based detection sees past a lossy conversion; it only fires
# when the source itself is a provenance-tagged summary carrier. # when the source itself is a provenance-tagged summary carrier.
# Canonical source-based summary detection: reads the ORIGINAL chat message's own content, so it
# sees past a lossy conversion (a typed `function_call_output` wrapper, or a stale exact-replay
# message) that erased the summary from `item` itself (#90976).
if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source): if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source):
text = flatten_message_text(source.get("content")) text = flatten_message_text(source.get("content"))
_src_role = source.get("role") _src_role = source.get("role")
@@ -291,6 +306,13 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool:
Drives one-shot recovery (strip, disable for the session, retry), so matching is narrow: Drives one-shot recovery (strip, disable for the session, retry), so matching is narrow:
a transient 5xx that merely ECHOES the request must not downgrade native compaction. a transient 5xx that merely ECHOES the request must not downgrade native compaction.
Requires ``status_code`` 400 (or unknown) AND the field name with rejection language. Requires ``status_code`` 400 (or unknown) AND the field name with rejection language.
See #82777.
* ``status_code`` is 400 (or unknown/None — some transports surface only a message string; field-name
matching alone is then the best available signal, preserving pre-#82777 behavior for them), and * the
error text names ``context_management`` / ``compact_threshold`` alongside rejection language ("unknown",
"unsupported", "invalid", "unexpected", "not permitted"...). A bare field-name echo without rejection
language does not match.
""" """
text = str(error or "").lower() text = str(error or "").lower()
if "context_management" not in text and "compact_threshold" not in text: if "context_management" not in text and "compact_threshold" not in text:
+12
View File
@@ -56,6 +56,18 @@ def opencode_session_headers(
from agent.transports.codex import _cache_scope_from_session_id from agent.transports.codex import _cache_scope_from_session_id
key = _cache_scope_from_session_id( key = _cache_scope_from_session_id(
# Top-level session_id → OpenRouter's sticky routing key. Per their prompt-caching docs it is
# used directly as the routing key instead of hashing the opening messages, and it activates
# stickiness on the first successful request rather than only after a cache hit. Resolve it from
# the declared routing scope first (set only by a host that names its own conversation, #96811),
# then the ambient conversation contextvar, with the explicit argument as fallback. The gap this
# closes is the auxiliary call sites — compression, title generation, vision, web_extract,
# session_search, MoA slots — which funnel through ``agent.auxiliary_client``. That module has
# no session handle and passes no ``session_id``, so those calls sent NO sticky key at all and
# each routed independently of the conversation it belonged to (#70820). Mirrors the Nous Portal
# profile, which resolves the same way (f2f4df064d). The ambient value is the session-lineage
# ROOT, so it also stays stable for installs that opt out of the default ``compression.in_place:
# true`` and across delegate-subagent trees.
get_affinity_scope() or get_conversation_context() or session_id get_affinity_scope() or get_conversation_context() or session_id
) )
except Exception: except Exception:
+7 -1
View File
@@ -133,7 +133,12 @@ def flush(timeout: float = 5.0) -> bool:
def re_register_config_hooks() -> None: def re_register_config_hooks() -> None:
"""Re-register outbound webhooks after a plugin force-reload cleared ``_hooks``. Only the """Re-register outbound webhooks after a plugin force-reload cleared ``_hooks``. Only the
current home's idempotence keys are cleared so a force-reload in one profile cannot current home's idempotence keys are cleared so a force-reload in one profile cannot
invalidate another profile's still-live registration.""" invalidate another profile's still-live registration.
Mirrors ``agent.shell_hooks.re_register_config_hooks``: config-owned outbound-webhook callbacks live in
the same ``_hooks`` dict that ``PluginManager.discover_and_load(force=True)`` clears via ``unload()``,
so without this the force-reloaded profile's outbound webhooks go silently inert (#92682 review).
"""
from hermes_cli.config import load_config from hermes_cli.config import load_config
_forget_home_registrations(_registered, _registered_lock) _forget_home_registrations(_registered, _registered_lock)
register_from_config(load_config()) register_from_config(load_config())
@@ -235,6 +240,7 @@ def _serialize_payload(event: str, kwargs: Dict[str, Any], delivery_id: str) ->
(also the ``X-Hermes-Delivery`` header) and ``timestamp`` live inside the HMAC-signed (also the ``X-Hermes-Delivery`` header) and ``timestamp`` live inside the HMAC-signed
body, so they double as replay protection.""" body, so they double as replay protection."""
# Profile resolved at fire time so a multiplexed gateway's receivers can tell which profile emitted. # Profile resolved at fire time so a multiplexed gateway's receivers can tell which profile emitted.
# See #92674.
from hermes_cli.profiles import get_active_profile_name from hermes_cli.profiles import get_active_profile_name
payload = { payload = {
"hook_event_name": event, "profile": get_active_profile_name(), **_payload_fields(kwargs), "hook_event_name": event, "profile": get_active_profile_name(), **_payload_fields(kwargs),
+4 -1
View File
@@ -58,7 +58,10 @@ Interaction style:
def build_plan_prompt(task: str = "") -> str: def build_plan_prompt(task: str = "") -> str:
"""Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context.""" """Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context.
See #36821.
"""
task = (task or "").strip() task = (task or "").strip()
task_block = f"Task to plan:\n{task}\n" if task else ( task_block = f"Task to plan:\n{task}\n" if task else (
"No explicit task was given with /plan — infer the task from the " "No explicit task was given with /plan — infer the task from the "
+5 -1
View File
@@ -211,7 +211,11 @@ def _check_task(policy: _TrustPolicy, *, plugin_id: str, requested_task: Optiona
registered itself → allowed; a built-in key → only with ``allow_task_override``; registered itself → allowed; a built-in key → only with ``allow_task_override``;
anything else raises + logs. Never silently downgraded to ``auto``: that would anything else raises + logs. Never silently downgraded to ``auto``: that would
mask the misconfiguration and could route to a main model the user steered mask the misconfiguration and could route to a main model the user steered
elsewhere on purpose.""" elsewhere on purpose.
A foreign/unknown key raises :class:`PluginLlmTrustError` and logs a warning naming the offending plugin
and key. See #64174, #64182.
"""
task = (requested_task or "").strip() task = (requested_task or "").strip()
if not task or task.lower() == "auto": if not task or task.lower() == "auto":
return None return None
+10
View File
@@ -209,6 +209,11 @@ def enable_happy_eyeballs_on_client(client) -> None:
For callers that build clients inline (Codex OAuth/device-login). Proxy-backed For callers that build clients inline (Codex OAuth/device-login). Proxy-backed
pools are skipped (TCP connect goes to the proxy host); async clients need pools are skipped (TCP connect goes to the proxy host); async clients need
nothing (anyio already races per RFC 8305). Best-effort. nothing (anyio already races per RFC 8305). Best-effort.
Proxy-backed transports (``httpcore.HTTPProxy`` / SOCKS pools) are left untouched: with a proxy in play
the TCP connect goes to the proxy host, which is out of scope for the direct-transport racing added in
#94388. Async clients are also left untouched — httpcore's async backend already performs RFC 8305
racing natively via ``anyio.connect_tcp(happy_eyeballs_delay=0.25)``.
""" """
try: try:
import httpcore import httpcore
@@ -312,6 +317,8 @@ def _shared_transport_cls():
mounted object absorbs that close while the shared pool keeps serving other clients. mounted object absorbs that close while the shared pool keeps serving other clients.
``handle_request`` stamps the owning view into ``request.extensions`` so socket-abort ``handle_request`` stamps the owning view into ``request.extensions`` so socket-abort
sweeps target only this client's in-flight connections on the shared pool. sweeps target only this client's in-flight connections on the shared pool.
See #10933.
""" """
__slots__ = ("_inner", "_closed") __slots__ = ("_inner", "_closed")
@@ -391,6 +398,9 @@ def build_keepalive_http_client(base_url: str = "", *, async_mode: bool = False,
``HTTPTransport`` through a ``_SharedTransport`` view, so N delegated children share one ``HTTPTransport`` through a ``_SharedTransport`` view, so N delegated children share one
connection pool + SSL context. Async clients are never shared: an httpcore async pool is connection pool + SSL context. Async clients are never shared: an httpcore async pool is
bound to the event loop that first used it. Proxy-backed clients keep httpx's own transport. bound to the event loop that first used it. Proxy-backed clients keep httpx's own transport.
See #12952, #54049.
See #10933.
""" """
try: try:
import httpx import httpx
+86 -1
View File
@@ -150,6 +150,10 @@ HERMES_AGENT_HELP_GUIDANCE_NO_SKILLS = (
) )
# Memory guidance (#95681, consolidated): ONE block from ONE builder. The opening frame adapts to which
# stores config enables; everything else is written exactly once. Leads with the positive posture (save
# proactively, replace when full) — the routing rules come after, as refinements, not as the headline. WHAT
# belongs in memory is the memory tool schema's job and is never re-taught here.
def build_memory_guidance(memory_enabled: bool = True, profile_enabled: bool = True) -> str: def build_memory_guidance(memory_enabled: bool = True, profile_enabled: bool = True) -> str:
"""ONE memory-guidance block whose opening frame adapts to the enabled store(s); "" when both are off. """ONE memory-guidance block whose opening frame adapts to the enabled store(s); "" when both are off.
@@ -199,6 +203,17 @@ SESSION_SEARCH_GUIDANCE = (
# ("After completing a complex task (5+ tool calls)... save the approach as a skill...") on subscription OAuth # ("After completing a complex task (5+ tool calls)... save the approach as a skill...") on subscription OAuth
# credentials, surfacing as a billing-shaped HTTP 400. If you rewrite it, re-verify with a subscription OAuth # credentials, surfacing as a billing-shaped HTTP 400. If you rewrite it, re-verify with a subscription OAuth
# token — sk-ant-api keys do not hit the filter. The safety-rule heading is referenced by tests and compaction summaries. # token — sk-ant-api keys do not hit the filter. The safety-rule heading is referenced by tests and compaction summaries.
# Anthropic's server-side content filter rejects the previous phrasing ("After completing a complex task (5+
# tool calls), fixing a tricky error, or discovering a non-trivial workflow, save the approach as a skill
# with skill_manage so you can reuse it next time.") on subscription OAuth credentials, and surfaces that
# rejection as a billing-shaped HTTP 400 ("You're out of extra usage"), which sends users to buy quota they
# do not need. Bisected against the live API: that sentence alone reproduces the 400 and removing it alone
# clears it; size and the system[0] identity gate were both ruled out. The reword is empirically validated,
# not understood — if you rewrite this sentence, re-verify against a subscription OAuth token, not an
# sk-ant-api… key, which does not hit the filter. Dieted (#95681, maintainer-directed): the record-it /
# patch-it coaching that used to open this block duplicated the ## Skills section (which teaches both "offer
# to save as a skill" and "fix it with skill_manage(action='patch')") and skill_manage's own schema. Only
# the compaction-pruning contract lives here — nothing else teaches it.
SKILLS_GUIDANCE = ( SKILLS_GUIDANCE = (
"When you work out a non-trivial workflow, record it with skill_manage for future reuse.\n\n" "When you work out a non-trivial workflow, record it with skill_manage for future reuse.\n\n"
"## Skill Safety Rule\n" "## Skill Safety Rule\n"
@@ -308,6 +323,14 @@ TOOL_USE_ENFORCEMENT_MODELS = ("gpt", "codex", "gemini", "gemma", "grok", "glm",
# traces showed the same failure modes; Muse Spark stops after a chat-only turn on defaults). Gemini/Gemma get # traces showed the same failure modes; Muse Spark stops after a chat-only turn on defaults). Gemini/Gemma get
# GOOGLE_MODEL_OPERATIONAL_GUIDANCE instead; Claude does not exhibit these modes. Any model can opt in via # GOOGLE_MODEL_OPERATIONAL_GUIDANCE instead; Claude does not exhibit these modes. Any model can opt in via
# config.yaml (`true` or a substring list). # config.yaml (`true` or a substring list).
# Model name substrings whose sessions receive OPENAI_MODEL_EXECUTION_GUIDANCE (execution discipline: tool
# persistence, mandatory tool use for arithmetic, external-write read-back, count reconciliation, literal
# preservation, verification-gated completion) when agent.execution_guidance is "auto". gpt/codex/grok are
# the historical set; deepseek/kimi/qwen/glm/minimax/ mimo/mistral were added after Composio agentic-eval
# traces showed the same failure modes on those families (financial math in prose, no read-back after
# external writes, identifier "repair", completeness claims despite count mismatches). GLM's
# tool-calls-as-plain-text stall (#53847) and MiMo (#41874) are covered here too. Gemini/Gemma are excluded
# — they get the more specific GOOGLE_MODEL_OPERATIONAL_GUIDANCE block instead.
EXECUTION_GUIDANCE_MODELS = ( EXECUTION_GUIDANCE_MODELS = (
"gpt", "codex", "grok", "gpt", "codex", "grok",
"deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral", "muse", "deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral", "muse",
@@ -329,6 +352,20 @@ TASK_COMPLETION_GUIDANCE = (
# Universal parallel-tool-call guidance (ALL models): the runtime already executes independent calls # Universal parallel-tool-call guidance (ALL models): the runtime already executes independent calls
# concurrently. Supersedes the former Google-only bullet so no model receives the steer twice. # concurrently. Supersedes the former Google-only bullet so no model receives the steer twice.
# Why this matters for cost: every assistant turn resends the entire accumulated conversation (and, on
# cache-friendly providers, re-reads the cached prefix and pays for the newly-appended turn). A model that
# issues one tool call per turn multiplies the number of round-trips — and therefore the resent context —
# for any task that needs several independent reads, searches, or safe lookups. Batching independent calls
# into a single assistant response collapses N turns into one, cutting both latency and the resent-context
# cost that compounds over a long conversation. The hermes-agent runtime already executes a batch of tool
# calls concurrently when they are independent (read-only tools always; path-scoped file ops when their
# targets don't overlap — see run_agent._execute_tool_calls / tool_dispatch_helpers). The missing piece was
# telling the *model* to emit those calls together in the first place. Until now the only batching steer in
# the prompt lived in GOOGLE_MODEL_OPERATIONAL_GUIDANCE — Gemini/Gemma got it, every other model got
# nothing. Short on purpose — shipped in the cached system prompt to every user, every session. Token cost
# is paid once at install and amortised across all sessions via prefix caching. Keep it tight. Ported from
# cline/cline#11514 ("encourage parallel tool calls"), adapted from Cline's TypeScript tool-surface guidance
# to hermes-agent's Python prompt-assembly architecture.
PARALLEL_TOOL_CALL_GUIDANCE = ( PARALLEL_TOOL_CALL_GUIDANCE = (
"# Parallel tool calls\n" "# Parallel tool calls\n"
"When you need several pieces of information that don't depend on each other, request them together in a " "When you need several pieces of information that don't depend on each other, request them together in a "
@@ -342,6 +379,15 @@ PARALLEL_TOOL_CALL_GUIDANCE = (
# Execution-discipline guidance for models that abandon partial results, skip prerequisite lookups, answer # Execution-discipline guidance for models that abandon partial results, skip prerequisite lookups, answer
# from memory, or declare "done" unverified. Body is family-agnostic (OPENAI_ prefix reflects origin). # from memory, or declare "done" unverified. Body is family-agnostic (OPENAI_ prefix reflects origin).
# Injection gate: system_prompt.py via config.yaml ``agent.execution_guidance`` (auto/true/false/list). # Injection gate: system_prompt.py via config.yaml ``agent.execution_guidance`` (auto/true/false/list).
# OpenAI GPT/Codex-specific execution guidance. Addresses known failure modes where GPT models abandon work
# on partial results, skip prerequisite lookups, hallucinate instead of using tools, and declare "done"
# without verification. Inspired by patterns from OpenAI's GPT-5.4 prompting guide & OpenClaw PR #38953.
# Also applied to xAI Grok — same failure modes in practice (claims completion without tool calls, suggests
# workarounds instead of using existing tools, replies with plans/suggestions instead of executing). As of
# the Composio agentic-eval follow-up, the block is no longer fenced to gpt/codex/grok: eval traces showed
# DeepSeek/Kimi doing financial math in prose, skipping read-back verification after external writes,
# "repairing" malformed identifiers, and claiming completeness despite count mismatches — exactly the
# failure modes this block targets.
OPENAI_MODEL_EXECUTION_GUIDANCE = ( OPENAI_MODEL_EXECUTION_GUIDANCE = (
"# Execution discipline\n" "# Execution discipline\n"
"<tool_persistence>\n" "<tool_persistence>\n"
@@ -462,6 +508,13 @@ def format_steer_marker(steer_text: str) -> str:
STEER_CHANNEL_NOTE = ( STEER_CHANNEL_NOTE = (
# Only what the marker cannot say about itself: it is the ONLY trusted shape and carries full user authority. # Only what the marker cannot say about itself: it is the ONLY trusted shape and carries full user authority.
# Dieted (#95681, maintainer-directed). History: #40240 added this note when the marker was bare and
# models refused steers as prompt injection (screenshot-verified). The marker has since become
# self-describing — it declares its own provenance ("a direct message from the user...") and its own
# replay rule ("not a new delivery when replayed from conversation history") at delivery time — so the
# prompt-side briefing keeps only what the marker cannot say about itself: it is the ONLY trusted shape
# (anti-lookalike), and it carries full user authority. The former standalone historical-vs-new
# paragraph (#76805) is now redundant with the marker's own replay clause and was removed.
"## Mid-turn user steering\n" "## Mid-turn user steering\n"
"Mid-turn, the user can steer you: Hermes appends their message to the end of a tool result, wrapped exactly as:\n" "Mid-turn, the user can steer you: Hermes appends their message to the end of a tool result, wrapped exactly as:\n"
f"{STEER_MARKER_OPEN}\n<their message>\n{STEER_MARKER_CLOSE}\n" f"{STEER_MARKER_OPEN}\n<their message>\n{STEER_MARKER_CLOSE}\n"
@@ -681,6 +734,12 @@ PLATFORM_HINTS = {
# Telegram rich-messages extension — injected only with # Telegram rich-messages extension — injected only with
# ``platforms.telegram.extra.rich_messages: true`` (gateway.* or top-level). # ``platforms.telegram.extra.rich_messages: true`` (gateway.* or top-level).
# NOTE: a "webui" hint lived here until 2026-08-29. It was a ghost (verified in the all-platform hint audit,
# PR #97873): no code path constructs platform="webui" — the dashboard chat resolves to 'desktop' or 'tui'
# (tui_gateway/server.py:_resolve_session_platform), and the browser chat tab is an xterm.js PTY hosting the
# TUI, not an HTML chat renderer. Its content (tables/LaTeX/Mermaid, MEDIA: rich previews incl. Excalidraw)
# described a renderer that does not exist anywhere in web/. If a real WebUI chat surface ships, write a
# hint from its actual renderer — do not resurrect this text.
TELEGRAM_RICH_MESSAGES_HINT = ( TELEGRAM_RICH_MESSAGES_HINT = (
"Telegram now supports rich Markdown, so lean into it: whenever it makes the answer clearer or easier to scan, " "Telegram now supports rich Markdown, so lean into it: whenever it makes the answer clearer or easier to scan, "
"actively reach for real Markdown tables (pipe `| col | col |` syntax), bullet and numbered lists, task lists (`- " "actively reach for real Markdown tables (pipe `| col | col |` syntax), bullet and numbered lists, task lists (`- "
@@ -737,7 +796,13 @@ def _plugin_backend_is_remote(backend: str) -> bool:
def _windows_marketing_version() -> str: def _windows_marketing_version() -> str:
""""10"/"11" (``platform.release()`` says 10 for both; 11 is build >= 22000).""" """"10"/"11" (``platform.release()`` says 10 for both; 11 is build >= 22000).
``platform.release()`` reports the kernel version, which is ``10`` for BOTH Windows 10 and Windows 11 —
the prompt then claims "Windows (10)" on Windows 11 hosts and misleads the model about the OS (#51755).
Windows 11 is distinguished by build number: >= 22000 is 11. Falls back to ``platform.release()`` on any
lookup failure.
"""
try: try:
return "11" if sys.getwindowsversion().build >= 22000 else "10" # type: ignore[attr-defined] return "11" if sys.getwindowsversion().build >= 22000 else "10" # type: ignore[attr-defined]
except Exception: except Exception:
@@ -974,6 +1039,9 @@ def drain_truncation_warnings() -> list:
# Skills index (two-layer cache: in-process LRU, then disk snapshot). # Skills index (two-layer cache: in-process LRU, then disk snapshot).
# One entry per profile × platform (key carries skills_dir); a multiplexing gateway needs more than a handful. # One entry per profile × platform (key carries skills_dir); a multiplexing gateway needs more than a handful.
# Sized for multi-profile processes: since #86313 the cache key carries a per-profile skills_dir (one entry
# per profile × platform), so the old cap of 8 could thrash on a gateway multiplexing default + several bots
# (each miss = full os.walk manifest rebuild). ~32 costs low single-digit MB worst case.
_SKILLS_PROMPT_CACHE_MAX = 32 _SKILLS_PROMPT_CACHE_MAX = 32
_SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict() _SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict()
_SKILLS_PROMPT_CACHE_LOCK = threading.Lock() _SKILLS_PROMPT_CACHE_LOCK = threading.Lock()
@@ -1361,6 +1429,11 @@ def load_soul_md(context_length: Optional[int] = None, home_override: "Path | No
Callers must pass ``skip_soul=True`` to ``build_context_files_prompt`` so it isn't injected twice. Callers must pass ``skip_soul=True`` to ``build_context_files_prompt`` so it isn't injected twice.
``home_override`` pins the profile home (a thread that lost the HERMES_HOME ContextVar reads the wrong one). ``home_override`` pins the profile home (a thread that lost the HERMES_HOME ContextVar reads the wrong one).
``home_override`` scopes the read to an explicit profile home (the agent knows its own home from its
session_db path). Without it, resolution is ambient — which on a thread that lost the HERMES_HOME
ContextVar falls back to the launch home and reads the wrong profile's SOUL.md (#50233, same class as
the skills-index leak fixed in #86313).
""" """
try: try:
from hermes_cli.config import ensure_hermes_home from hermes_cli.config import ensure_hermes_home
@@ -1423,6 +1496,14 @@ def _load_agents_md(cwd_path: Path, context_length: Optional[int] = None) -> str
Per directory the first of ``AGENTS.override.md`` / ``AGENTS.md`` / ``agents.md`` wins (a gitignored Per directory the first of ``AGENTS.override.md`` / ``AGENTS.md`` / ``agents.md`` wins (a gitignored
personal override shadows the committed file); identical content seen again down the chain is skipped. personal override shadows the committed file); identical content seen again down the chain is skipped.
Each directory on the chain (see ``_agents_md_directory_chain``) contributes its ``AGENTS.override.md``
/ ``AGENTS.md`` / ``agents.md`` (first name wins per directory) as its own provenance-labelled section.
``AGENTS.override.md`` wins over ``AGENTS.md`` so a developer can keep a personal, typically-gitignored
override next to the committed project instructions without editing the tracked file (same convention as
earendil-works/pi#7681). Identical content encountered again further down the chain (copied or symlinked
files) is deduplicated. With a single match — the common case, and always the case outside a git repo —
output is identical to the historical single-file behavior.
""" """
cwd_resolved = cwd_path.resolve() cwd_resolved = cwd_path.resolve()
sections: list[str] = [] sections: list[str] = []
@@ -1483,6 +1564,10 @@ def build_context_files_prompt(
cwd_path = Path(cwd if cwd is not None else os.getcwd()).resolve() cwd_path = Path(cwd if cwd is not None else os.getcwd()).resolve()
# A FALLBACK-picked cwd inside the Hermes install tree must not gain system-prompt authority (the desktop # A FALLBACK-picked cwd inside the Hermes install tree must not gain system-prompt authority (the desktop
# default would load this repo's contributor AGENTS.md). An explicit cwd is honored verbatim. # default would load this repo's contributor AGENTS.md). An explicit cwd is honored verbatim.
# An explicitly configured cwd is honored verbatim — the Hermes tree is a legitimate workspace when the
# user deliberately points a session at it — and CLI-style surfaces pass
# allow_install_tree_fallback=True because their launch dir IS the user's shell cwd (developing Hermes
# in-tree). See #64590.
from agent.runtime_cwd import _is_install_tree from agent.runtime_cwd import _is_install_tree
if cwd is None and not allow_install_tree_fallback and _is_install_tree(cwd_path): if cwd is None and not allow_install_tree_fallback and _is_install_tree(cwd_path):
logger.warning( logger.warning(
+15 -1
View File
@@ -73,6 +73,8 @@ def _conversation_generation(session_key: str, source: str, session_db: Any) ->
The declared key survives ``/new``, so hashing it alone would reuse one scope across The declared key survives ``/new``, so hashing it alone would reuse one scope across
conversations. The counter advances with each reset boundary, independent of prunable conversations. The counter advances with each reset boundary, independent of prunable
rows and wall-clock; compression does not advance it. rows and wall-clock; compression does not advance it.
The declared key names a chat and deliberately survives `/new` and policy resets. See #79017, #86733.
""" """
reader = getattr(session_db, "latest_conversation_boundary", None) reader = getattr(session_db, "latest_conversation_boundary", None)
if not callable(reader): if not callable(reader):
@@ -99,6 +101,11 @@ def declared_conversation_scope(agent: Any) -> Optional[str]:
if sid and db is not None: if sid and db is not None:
try: try:
# One read for both halves of the row identity (fork verdict + source). # One read for both halves of the row identity (fork verdict + source).
# One read for both halves of the row's identity: the fork verdict and the source the peer
# queries match on live on the same ``sessions`` row, and asking for them separately read it
# twice per resolution (@teknium1 on #98811). A SessionDB without the combined view keeps the
# original call, so nothing that predates it — including the doubles that certify the
# fail-closed contract below — changes behaviour.
identity = getattr(db, "declared_scope_identity", None) identity = getattr(db, "declared_scope_identity", None)
if callable(identity): if callable(identity):
is_fork, row_source = identity(sid) is_fork, row_source = identity(sid)
@@ -158,7 +165,14 @@ def declared_conversation_scope_safe(agent: Any) -> Optional[str]:
def resolve_prompt_cache_scope_safe(agent: Any) -> Optional[str]: def resolve_prompt_cache_scope_safe(agent: Any) -> Optional[str]:
"""Never-raising variant of :func:`resolve_prompt_cache_scope` (None = use the physical id). """Never-raising variant of :func:`resolve_prompt_cache_scope` (None = use the physical id).
At turn_context an exception inside ``set_runtime_main(...)`` would skip the whole binding.""" At turn_context an exception inside ``set_runtime_main(...)`` would skip the whole binding.
Returns None on any failure (or when there is no scope). Consumers treat None/empty as "fall back to the
physical session_id", so a resolution failure degrades to pre-#79017 behavior instead of blocking the
caller — important at turn_context's call site, where an exception raised inside the
``set_runtime_main(...)`` argument list would otherwise skip the whole runtime binding, not just the
cache scope.
"""
try: try:
return resolve_prompt_cache_scope(agent) or None return resolve_prompt_cache_scope(agent) or None
except Exception: except Exception:
+6
View File
@@ -280,6 +280,12 @@ def apply_anthropic_cache_control(
marker and the remaining two go to the latest cacheable non-system messages; otherwise marker and the remaining two go to the latest cacheable non-system messages; otherwise
the legacy system-and-3 layout applies. Idempotent: pre-existing markers are stripped from the legacy system-and-3 layout applies. Idempotent: pre-existing markers are stripped from
a per-message copy first. Returns a shallow list copy with deep copies of modified messages. a per-message copy first. Returns a shallow list copy with deep copies of modified messages.
Idempotent: pre-existing ``cache_control`` markers are stripped from a per-message copy before new ones
are placed, so calling this twice (or handing it messages a prior call already marked) can never
accumulate past 4 markers. Only messages that already carry a marker pay the copy cost — a shallow
top-level copy suffices because :func:`strip_anthropic_cache_control` is copy-on-write on content parts
— and the rest of the copy-on-write contract is unchanged (#90971).
""" """
if not api_messages: if not api_messages:
return api_messages return api_messages
+7
View File
@@ -71,6 +71,13 @@ _BEARER_PROVIDERS: Dict[str, Tuple[str, ...]] = {
# two require-rules on one host would reject each other's requests; the sandbox gets the token # two require-rules on one host would reject each other's requests; the sandbox gets the token
# under every name. Authorization is also matched for Anthropic/Azure (SDKs may send Bearer); # under every name. Authorization is also matched for Anthropic/Azure (SDKs may send Bearer);
# Gemini's ``?key=<token>`` style is covered by match_query. # Gemini's ``?key=<token>`` style is covered by match_query.
# Providers whose API authenticates with a NON-Authorization header. iron-proxy v0.39's
# ``secrets.replace.match_headers`` targets arbitrary header names (case-insensitive; confirmed by the
# iron-proxy author on PR #30179 and verified in the pinned v0.39.0 source — ``swapHeaders`` +
# ``parseHeaderMatchers``), so these are first-class swapped providers, not "uncovered". ``aliases`` are
# interchangeable env-var names for the SAME upstream credential (Hermes' auth.py keys Google on both
# GEMINI_API_KEY and GOOGLE_API_KEY). The sandbox receives the minted token under the canonical name AND
# every alias so SDKs reading either work.
_HEADER_AUTH_PROVIDERS: Dict[str, Dict[str, Tuple[str, ...]]] = { _HEADER_AUTH_PROVIDERS: Dict[str, Dict[str, Tuple[str, ...]]] = {
"ANTHROPIC_API_KEY": {"hosts": ("api.anthropic.com",), "match_headers": ("x-api-key", "Authorization"), "aliases": ()}, "ANTHROPIC_API_KEY": {"hosts": ("api.anthropic.com",), "match_headers": ("x-api-key", "Authorization"), "aliases": ()},
"AZURE_OPENAI_API_KEY": {"hosts": ("*.openai.azure.com", "*.cognitiveservices.azure.com", "*.services.ai.azure.com"), "AZURE_OPENAI_API_KEY": {"hosts": ("*.openai.azure.com", "*.cognitiveservices.azure.com", "*.services.ai.azure.com"),
+11 -1
View File
@@ -16,6 +16,7 @@ import re
from typing import Optional, Sequence from typing import Optional, Sequence
#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``), never K2-era names (``kimi-k2.6``). #: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``), never K2-era names (``kimi-k2.6``).
# From #76427 by @ruizanthony.
_KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)") _KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)")
# Canonical low→high ordering for nearest-level clamping. Includes "none" so an explicit # Canonical low→high ordering for nearest-level clamping. Includes "none" so an explicit
@@ -32,6 +33,7 @@ CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh"
CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh") CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh")
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high. #: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
# : Backward-compat alias (pre-#68365-verification name).
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh") XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high") XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
@@ -59,6 +61,9 @@ SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: widens it to a graded scale (live-verified, monotonic). ``xhigh`` requests the top tier. #: widens it to a graded scale (live-verified, monotonic). ``xhigh`` requests the top tier.
GLM52_EFFORTS: tuple[str, ...] = ("high", "max") GLM52_EFFORTS: tuple[str, ...] = ("high", "max")
GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"} GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"}
# : GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified : live on
# api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every : level accepted with monotonic
# reasoning-token scaling (low=4, medium=11, : high=98, max=125 on the probe prompt).
GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"} GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"}
@@ -80,7 +85,12 @@ def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]: def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
"""Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3).""" """Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3).
K3 is served as the bare slug ``k3``, plan variants like ``k3-256k``, and the ``kimi-k3*`` aliases; its
documented set is low/high/max. Everything earlier speaks low/medium/high. Boundary-matched so K2-era
names (``kimi-k2.6``) never match (detection regex from #76427 by @ruizanthony).
"""
m = (model or "").strip().lower().split("/")[-1] m = (model or "").strip().lower().split("/")[-1]
return KIMI_K3_EFFORTS if _KIMI_K3_SLUG_RE.search(m) else KIMI_K2_EFFORTS return KIMI_K3_EFFORTS if _KIMI_K3_SLUG_RE.search(m) else KIMI_K2_EFFORTS
+11 -2
View File
@@ -130,7 +130,12 @@ class ReasoningParamsMixin:
def _needs_thinking_reasoning_pad(self) -> bool: def _needs_thinking_reasoning_pad(self) -> bool:
"""True when the provider enforces ``reasoning_content`` echo-back on tool-call replays (DeepSeek, Kimi, """True when the provider enforces ``reasoning_content`` echo-back on tool-call replays (DeepSeek, Kimi,
MiMo thinking all 400 without it). Cached per (provider, model, base_url), invalidated by MiMo thinking all 400 without it). Cached per (provider, model, base_url), invalidated by
``switch_model()`` / ``_try_activate_fallback()`` — called ~16× per turn.""" ``switch_model()`` / ``_try_activate_fallback()`` — called ~16× per turn.
DeepSeek v4 thinking and Kimi / Moonshot thinking both reject replays of assistant tool-call
messages that omit ``reasoning_content`` (refs 15250, #17400). Xiaomi MiMo thinking mode has the
same requirement.
"""
key = (self.provider, self.model, getattr(self, "_base_url_lower", self.base_url)) key = (self.provider, self.model, getattr(self, "_base_url_lower", self.base_url))
cached = getattr(self, "_thinking_pad_cache", None) cached = getattr(self, "_thinking_pad_cache", None)
if cached is not None and cached[0] == key: if cached is not None and cached[0] == key:
@@ -162,7 +167,11 @@ class ReasoningParamsMixin:
return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url) return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url)
def _needs_deepseek_tool_reasoning(self) -> bool: def _needs_deepseek_tool_reasoning(self) -> bool:
"""True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400).""" """True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400).
DeepSeek V4 thinking mode requires ``reasoning_content`` on every assistant tool-call turn; omitting
it causes HTTP 400 when the message is replayed in a subsequent API request (#15250).
"""
return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url) return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url)
def _needs_mimo_tool_reasoning(self) -> bool: def _needs_mimo_tool_reasoning(self) -> bool:
+97 -1
View File
@@ -19,6 +19,8 @@ logger = logging.getLogger(__name__)
# Sensitive query-string param names (case-insensitive): opaque tokens / OAuth # Sensitive query-string param names (case-insensitive): opaque tokens / OAuth
# codes / pre-signed signatures with no vendor prefix. # codes / pre-signed signatures with no vendor prefix.
# Ported from nearai/ironclaw#2529 — catches tokens whose values don't match any known vendor prefix regex
# (e.g. opaque tokens, short OAuth codes).
_SENSITIVE_QUERY_PARAMS = frozenset({ _SENSITIVE_QUERY_PARAMS = frozenset({
"access_token", "refresh_token", "id_token", "token", "api_key", "apikey", "access_token", "refresh_token", "id_token", "token", "api_key", "apikey",
"client_secret", "password", "auth", "jwt", "session", "secret", "key", "client_secret", "password", "auth", "jwt", "session", "secret", "key",
@@ -28,6 +30,11 @@ _SENSITIVE_QUERY_PARAMS = frozenset({
# Snapshot at import time so runtime env mutations (e.g. an LLM-generated # Snapshot at import time so runtime env mutations (e.g. an LLM-generated
# `export HERMES_REDACT_SECRETS=false`) cannot disable redaction mid-session. # `export HERMES_REDACT_SECRETS=false`) cannot disable redaction mid-session.
# ON by default; `security.redact_secrets: false` bridges to this env var. # ON by default; `security.redact_secrets: false` bridges to this env var.
# ON by default — secure default per issue #17691. Users who need raw credential values in tool output (e.g.
# working on the redactor itself) can opt out via `security.redact_secrets: false` in config.yaml (bridged
# to this env var in hermes_cli/main.py, gateway/run.py, and cli.py) or `HERMES_REDACT_SECRETS=false` in
# ~/.hermes/.env. An opt-out warning is logged at gateway and CLI startup so operators see the downgrade —
# see `_log_redaction_status()` in gateway/run.py and cli.py.
_REDACT_ENABLED = os.getenv("HERMES_REDACT_SECRETS", "true").lower() in {"1", "true", "yes", "on"} _REDACT_ENABLED = os.getenv("HERMES_REDACT_SECRETS", "true").lower() in {"1", "true", "yes", "on"}
# Known API key prefixes -- match the prefix + contiguous token chars. # Known API key prefixes -- match the prefix + contiguous token chars.
@@ -76,6 +83,7 @@ _PREFIX_PATTERNS = [
r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key
r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key
# GitLab token families (each keeps a full literal prefix for the pre-screen). # GitLab token families (each keeps a full literal prefix for the pre-screen).
# Ported from openclaw/openclaw#112954; follow-up invited in #4541.
r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token
r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret
r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token
@@ -97,12 +105,17 @@ _PREFIX_PATTERNS = [
# tolerate spaces around "=" and allow the keyword embedded anywhere # tolerate spaces around "=" and allow the keyword embedded anywhere
# (``MYTOKEN=…``) — an all-caps key is almost never prose. Bare ``KEY``/``PASS``/ # (``MYTOKEN=…``) — an all-caps key is almost never prose. Bare ``KEY``/``PASS``/
# ``PW`` suffixes are included; _key_has_secret_keyword rejects ``KEYBOARD=``. # ``PW`` suffixes are included; _key_has_secret_keyword rejects ``KEYBOARD=``.
# The regex is IGNORECASE so lowercase env names (``openai_key=…``) are caught here too. The secret name
# must sit at a word boundary (``_``-delimited or whole-word) so generic prose words (``password=``,
# ``token=``, ``KEYBOARD=``, ``PASSAGE=``) do not match — those are handled by the config/form/URL paths,
# and a bare ``password=…`` in a form body must not be swallowed greedily by ``\S+``. See #77484.
_SECRET_ENV_NAMES = r"(?:API_?KEY|KEY|TOKEN|SECRET|PASSWORD|PASSWD|PASS|PW|CREDENTIAL|AUTH)" _SECRET_ENV_NAMES = r"(?:API_?KEY|KEY|TOKEN|SECRET|PASSWORD|PASSWD|PASS|PW|CREDENTIAL|AUTH)"
_ENV_ASSIGN_RE = re.compile(rf"([A-Z0-9_]{{0,50}}{_SECRET_ENV_NAMES}[A-Z0-9_]{{0,50}})\s*=\s*(['\"]?)(\S+)\2") _ENV_ASSIGN_RE = re.compile(rf"([A-Z0-9_]{{0,50}}{_SECRET_ENV_NAMES}[A-Z0-9_]{{0,50}})\s*=\s*(['\"]?)(\S+)\2")
# Lowercase env names: only underscore-boundary forms (``openai_key=``) — NOT # Lowercase env names: only underscore-boundary forms (``openai_key=``) — NOT
# bare ``password=``/``token=``, which appear in prose, URLs, and form bodies. # bare ``password=``/``token=``, which appear in prose, URLs, and form bodies.
# The lookbehind anchors each attempt to the start of an identifier run; without # The lookbehind anchors each attempt to the start of an identifier run; without
# it re.sub retries the greedy prefix at every byte of a long opaque payload. # it re.sub retries the greedy prefix at every byte of a long opaque payload.
# See #77484.
_ENV_ASSIGN_LOWER_RE = re.compile( _ENV_ASSIGN_LOWER_RE = re.compile(
rf"(?<![a-z0-9_])([a-z0-9_]+(?:_|^)(?:key|pass|pw|token|secret|password|passwd|credential|auth)(?=[^a-z0-9_]|$))\s*=\s*(['\"]?)(\S+)\2", rf"(?<![a-z0-9_])([a-z0-9_]+(?:_|^)(?:key|pass|pw|token|secret|password|passwd|credential|auth)(?=[^a-z0-9_]|$))\s*=\s*(['\"]?)(\S+)\2",
re.IGNORECASE, re.IGNORECASE,
@@ -113,6 +126,14 @@ _ENV_ASSIGN_LOWER_RE = re.compile(
# whitespace AND ``&`` (form bodies go pair-by-pair via _redact_form_body); # whitespace AND ``&`` (form bodies go pair-by-pair via _redact_form_body);
# _CFG_DOTTED_RE needs a NAMESPACED key; _CFG_ANCHORED_RE needs line start # _CFG_DOTTED_RE needs a NAMESPACED key; _CFG_ANCHORED_RE needs line start
# (optionally after ``export``). The ``://`` URL guard lives at the call site. # (optionally after ``export``). The ``://`` URL guard lives at the call site.
# The uppercase _ENV_ASSIGN_RE above never matched these, so config-file passwords leaked verbatim (issue
# #16413). These run only in a config-file context, NOT in prose, code, or URLs — three carve-outs preserved
# from the original design (#4367 + the documented web-URL passthrough below): 1. The value is bounded by
# ``[^\s&]`` (stops at whitespace AND ``&``) so form-urlencoded bodies are handled pair-by-pair (by
# _redact_form_body), not greedily swallowed. 2. _CFG_DOTTED_RE only matches when the key is NAMESPACED
# (contains a dot), which is unambiguously a config key — never a prose word. 3. _CFG_ANCHORED_RE matches a
# bare secret-word key only at line start (optionally after ``export``), so conversational ``I have
# password=foo`` mid-sentence is left alone.
_SECRET_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential|auth)" _SECRET_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential|auth)"
_CFG_VALUE = r"(['\"]?)([^\s&]+?)\2(?=[\s&]|$)" _CFG_VALUE = r"(['\"]?)([^\s&]+?)\2(?=[\s&]|$)"
# Linear pre-gate for the _CFG_*_RE subs: no secret keyword => neither can match. # Linear pre-gate for the _CFG_*_RE subs: no secret keyword => neither can match.
@@ -158,6 +179,16 @@ _YAML_ASSIGN_RE = re.compile(
# only at a word boundary: key edge, next to a non-letter, or a camelCase # only at a word boundary: key edge, next to a non-letter, or a camelCase
# transition (``clientSecret``, ``APIToken``); trailing plural ``s`` is part of # transition (``clientSecret``, ``APIToken``); trailing plural ``s`` is part of
# it. Concatenations match via explicit alternatives (``authtoken``, ``apikey``). # it. Concatenations match via explicit alternatives (``authtoken``, ``apikey``).
# The side effect: ordinary prose/document words that merely CONTAIN a keyword also matched — ``Secretary:
# J.Smith`` (secret), ``tokenizer: cl100k_base`` (token), ``author=Smith`` (auth) — mangling legitimate
# content on the surfaces that run these passes (browser snapshots, log lines, kanban summaries, CLI-echoed
# command output). Ported from nearai/ironclaw#6129, where the same substring false positive ("Secretary of
# the Treasury" matching the ``secret`` marker) scrubbed legitimate tool results from the replayed
# transcript and sent the model into a re-fetch loop. Common concatenated compounds keep matching via
# explicit alternatives (``authtoken`` ngrok, ``authkey`` tailscale, ``secretkey`` minio, ``apikey``).
# Embedded occurrences inside a larger word (``secretary``, ``tokenizer``, ``authored``, ``credentialing``)
# no longer match. ALL-CAPS keys keep the legacy embedded matching (``MYTOKEN=…``) — an all-caps key is
# almost never prose, the same rationale as _ENV_ASSIGN_RE.
_KEY_KEYWORD_RE = re.compile( _KEY_KEYWORD_RE = re.compile(
r"(?:api|auth|access|refresh|session|secret)[ _.\\-]?(?:key|token)" r"(?:api|auth|access|refresh|session|secret)[ _.\\-]?(?:key|token)"
r"|token|secret|passwd|password|pass|pw|credential|auth|key", r"|token|secret|passwd|password|pass|pw|credential|auth|key",
@@ -223,6 +254,12 @@ def _should_redact_assignment(key: str, value: str, *, check_keyword: bool) -> b
"""Shared gate for the ENV / JSON / YAML assignment passes: skip programmatic env """Shared gate for the ENV / JSON / YAML assignment passes: skip programmatic env
lookups used as values, optionally require a word-bounded keyword in the key, lookups used as values, optionally require a word-bounded keyword in the key,
then redact when the key is unambiguously credential-bearing or the value looks opaque.""" then redact when the key is unambiguously credential-bearing or the value looks opaque."""
# Programmatic env lookups reference variable *names*, not secret values — masking them corrupts code
# snippets in prose/log contexts (issue #2852): ``KEY=os.getenv('X')``.
# Same programmatic-env-lookup exception as _redact_env above (issue #2852): "apiKey": "os.getenv('X')"
# is a code snippet, not a leaked secret value.
# Same programmatic-env-lookup exception as _redact_env above (issue #2852): api_key: os.getenv('X') is
# a code snippet, not a leaked secret value.
if _ENV_LOOKUP_VALUE_RE.match(value): if _ENV_LOOKUP_VALUE_RE.match(value):
return False return False
if check_keyword and not _key_has_secret_keyword(key): if check_keyword and not _key_has_secret_keyword(key):
@@ -254,6 +291,11 @@ _PRIVATE_KEY_RE = re.compile(r"-----BEGIN[A-Z ]*PRIVATE KEY-----[\s\S]*?-----END
# Database connection strings: protocol://user:PASSWORD@host. The userinfo and # Database connection strings: protocol://user:PASSWORD@host. The userinfo and
# password groups forbid whitespace so a match can never span a line break (a # password groups forbid whitespace so a match can never span a line break (a
# greedy ``[^@]+`` once ran to a decorator's ``@`` on the next code line). # greedy ``[^@]+`` once ran to a decorator's ``@`` on the next code line).
# Database connection strings: protocol://user:PASSWORD@host Catches postgres, mysql, mongodb, redis, amqp
# URLs and redacts the password. A real DSN password never contains whitespace; without this bound the
# greedy [^@]+ would scan past the end of a code line to the next stray "@" (e.g. a Python decorator),
# swallowing intervening lines and corrupting tool OUTPUT for any source containing a postgresql:// f-string
# template. See issue #33801.
_DB_CONNSTR_RE = re.compile( _DB_CONNSTR_RE = re.compile(
r"((?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp)://[^:\s]+:)([^@\s]+)(@)", r"((?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp)://[^:\s]+:)([^@\s]+)(@)",
re.IGNORECASE, re.IGNORECASE,
@@ -265,6 +307,11 @@ _DB_CONNSTR_RE = re.compile(
# bare userinfo. ``user:pass@`` passes through (class forbids ``:``); DB schemes # bare userinfo. ``user:pass@`` passes through (class forbids ``:``); DB schemes
# belong to _DB_CONNSTR_RE. 8+ char floor skips short usernames; the class # belong to _DB_CONNSTR_RE. 8+ char floor skips short usernames; the class
# forbids ``/`` so an ``@`` in a path/query (``?q=user@example.com``) never counts. # forbids ``/`` so an ``@`` in a path/query (``?q=user@example.com``) never counts.
# This is the ``git remote set-url origin https://PASSWORD@github.com/...`` shape from issue #6396 — a
# single opaque credential in the userinfo position with NO ``user:pass`` colon. The colon form
# ``user:pass@`` is deliberately left to pass through (commit "pass web URLs through unchanged", #34029) and
# is NOT matched here — the token class forbids ``:``. DB schemes are handled by _DB_CONNSTR_RE above and
# excluded here. Guards against false positives:
_URL_BARE_TOKEN_RE = re.compile( _URL_BARE_TOKEN_RE = re.compile(
r"((?:https?|wss?|git|ssh|ftp|ftps|sftp)://)" # scheme r"((?:https?|wss?|git|ssh|ftp|ftps|sftp)://)" # scheme
r"([^\s:@/]{8,})" # bare token (no colon/slash/@), 8+ chars r"([^\s:@/]{8,})" # bare token (no colon/slash/@), 8+ chars
@@ -320,6 +367,10 @@ def _mask_control_split_tokens(text: str, mask_fn) -> str:
Match on a control-stripped copy, then mask the corresponding span in the Match on a control-stripped copy, then mask the corresponding span in the
ORIGINAL — only when that span holds solely token-body and control chars, so ORIGINAL — only when that span holds solely token-body and control chars, so
a match can never cross into another line's unrelated text. a match can never cross into another line's unrelated text.
A credential like ``sk-abc\\x1bdef456…`` or ``ghp_abc\\n123def…`` has its token body interrupted, so the
contiguous _PREFIX_RE cannot match it and the secret leaks verbatim (issue #77484). ``EXA_API_KEY=*** is
rejected).
""" """
stripped = _CONTROL_CHARS_RE.sub("", text) stripped = _CONTROL_CHARS_RE.sub("", text)
if stripped == text: if stripped == text:
@@ -428,7 +479,12 @@ def _redact_form_body(text: str) -> str:
def _mask_token_nonreusable(token: str) -> str: def _mask_token_nonreusable(token: str) -> str:
"""Redact a prefix-matched credential to a NON-REUSABLE sentinel: no head/tail """Redact a prefix-matched credential to a NON-REUSABLE sentinel: no head/tail
chars (an agent once wrote a truncated-looking mask back into a config file), chars (an agent once wrote a truncated-looking mask back into a config file),
only the vendor prefix label so the credential KIND stays visible.""" only the vendor prefix label so the credential KIND stays visible.
* cannot be mistaken for a usable-but-truncated key, so an agent that reads it from a config file and
writes it back does NOT corrupt the stored credential into a dead 13-char string (issue #35519); and *
still does not leak the secret material (no head/tail chars).
"""
label = next((sub for sub in _PREFIX_SUBSTRINGS if token.startswith(sub)), "") if token else "" label = next((sub for sub in _PREFIX_SUBSTRINGS if token.startswith(sub)), "") if token else ""
return f"«redacted:{label}…»" if label else "«redacted-secret»" return f"«redacted:{label}…»" if label else "«redacted-secret»"
@@ -451,9 +507,20 @@ def _redact_assignments(text: str) -> str:
_redact_env = _assignment_sub(lambda g: f"{g[0]}={g[1]}{_mask_token(g[2])}{g[1]}", check_keyword=True) _redact_env = _assignment_sub(lambda g: f"{g[0]}={g[1]}{_mask_token(g[2])}{g[1]}", check_keyword=True)
text = _ENV_ASSIGN_RE.sub(_redact_env, text) text = _ENV_ASSIGN_RE.sub(_redact_env, text)
if "://" not in text: # lowercase names would match URL params if "://" not in text: # lowercase names would match URL params
# Skip URLs — the query string may contain ``token=``/``key=`` params that are intentionally
# passed through (see note near the bottom of this function; _redact_strict_url_credentials
# handles the opt-in case). The uppercase regex above is all-caps-only, so it never matches URL
# params; the lowercase one would (issue #77484).
text = _ENV_ASSIGN_LOWER_RE.sub(_redact_env, text) text = _ENV_ASSIGN_LOWER_RE.sub(_redact_env, text)
# The keyword pre-gate is exact and matters: _CFG_DOTTED_RE backtracks # The keyword pre-gate is exact and matters: _CFG_DOTTED_RE backtracks
# quadratically on long unbroken [A-Za-z0-9_.\-] runs. # quadratically on long unbroken [A-Za-z0-9_.\-] runs.
# Lowercase/dotted config keys (issue #16413). Skip URLs entirely — web-URL query params are
# intentionally passed through (see note near the bottom of this function); _DB_CONNSTR_RE still
# guards connection-string passwords. Extra gate: every _CFG_*_RE match requires a secret keyword in
# the key, so a text without any secret keyword cannot match — skipping is exact. This matters
# because _CFG_DOTTED_RE backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs (e.g.
# base64/hex blobs in compaction payloads); the linear keyword scan prevents that pathological path
# on secret-free text.
if "://" not in text and _CFG_SECRET_WORD_RE.search(text): if "://" not in text and _CFG_SECRET_WORD_RE.search(text):
text = _CFG_DOTTED_RE.sub(_redact_env, text) text = _CFG_DOTTED_RE.sub(_redact_env, text)
text = _CFG_ANCHORED_RE.sub(_redact_env, text) text = _CFG_ANCHORED_RE.sub(_redact_env, text)
@@ -507,6 +574,11 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
Every regex sits behind a cheap substring gate that its pattern requires, Every regex sits behind a cheap substring gate that its pattern requires,
so the gates are never false-negative. so the gates are never false-negative.
Set file_read=True for file *content* returned to the agent (read_file / search_files / cat). The old
mask looked like a real-but-truncated key, so an agent reading it from config.yaml and writing it back
silently corrupted the stored credential into a dead 13-char value → 401 (issue #35519). The sentinel is
syntactically invalid as a token, so it can't be mistaken for a usable key or written back as one.
""" """
if text is None: if text is None:
return None return None
@@ -518,6 +590,10 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
# Control/zero-width chars can split a token body so _PREFIX_RE alone misses it. # Control/zero-width chars can split a token body so _PREFIX_RE alone misses it.
if _has_known_prefix_substring(text): if _has_known_prefix_substring(text):
_prefix_sub = _mask_token_nonreusable if file_read else _mask_token _prefix_sub = _mask_token_nonreusable if file_read else _mask_token
# Control/zero-width chars (\\n, \\r, ESC, U+200B, …) split a token body so _PREFIX_RE cannot match
# across them — a secret smuggled as ``sk-abc\\x1bdef…`` leaks verbatim (issue #77484). Mask such
# runs by first matching on a control-stripped copy, then re-masking the corresponding span in the
# original (the stripped copy and the original are aligned 1:1 for non-control chars).
text = _mask_control_split_tokens(text, _prefix_sub) text = _mask_control_split_tokens(text, _prefix_sub)
text = _PREFIX_RE.sub(lambda m: _prefix_sub(m.group(1)), text) text = _PREFIX_RE.sub(lambda m: _prefix_sub(m.group(1)), text)
@@ -534,6 +610,11 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
if "BEGIN" in text and "-----" in text: if "BEGIN" in text and "-----" in text:
text = _PRIVATE_KEY_RE.sub("[REDACTED PRIVATE KEY]", text) text = _PRIVATE_KEY_RE.sub("[REDACTED PRIVATE KEY]", text)
# Database connection string passwords. With code_file=True, a password group that is a pure ``{...}``
# brace expression is an f-string template reference (e.g. f"postgresql://{user}:{pass}@{host}"), not a
# literal credential — preserve it. Literal passwords are still redacted. The regex forbids whitespace
# in the password group, so a single-line template's group(2) is exactly the brace expression. See issue
# #33801.
if "://" in text: if "://" in text:
text = _redact_url_credentials(text, code_file) text = _redact_url_credentials(text, code_file)
@@ -541,6 +622,15 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
text = _JWT_RE.sub(lambda m: _mask_token(m.group(0)), text) text = _JWT_RE.sub(lambda m: _mask_token(m.group(0)), text)
if redact_url_credentials: # opt-in; known credential shapes in URLs are caught above if redact_url_credentials: # opt-in; known credential shapes in URLs are caught above
# NOTE: Web-URL redaction (query params + userinfo + HTTP access-log request targets) is
# intentionally OFF. Many legitimate workflows pass opaque tokens through query strings — magic-link
# checkouts, OAuth callbacks the agent is meant to follow, pre-signed share URLs — and
# blanket-redacting param values by name breaks those skills mid-flow. DB connection-string
# passwords are still caught by _DB_CONNSTR_RE. The ONE userinfo case still redacted is the
# colon-less bare-token form ``scheme://TOKEN@host`` (#6396, handled by _URL_BARE_TOKEN_RE in the
# ``://`` block above): a bare credential in userinfo is never a round-trip workflow token (those
# live in the query string), so masking it can't break a skill. The ``user:pass@`` form is left to
# pass through per #34029.
text = _redact_strict_url_credentials(text) text = _redact_strict_url_credentials(text)
if "&" in text and "=" in text: if "&" in text and "=" in text:
@@ -555,6 +645,10 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
# Commands whose stdout is an env-var dump: terminal redaction runs the # Commands whose stdout is an env-var dump: terminal redaction runs the
# ENV-assignment pass (code_file=False) for these so opaque tokens with no vendor # ENV-assignment pass (code_file=False) for these so opaque tokens with no vendor
# prefix are masked; everything else uses code_file=True (``MAX_TOKENS=100``). # prefix are masked; everything else uses code_file=True (``MAX_TOKENS=100``).
# Commands whose stdout is an environment-variable dump (KEY=value lines), NOT source code.
# ``MY_SERVICE_TOKEN=abc123randomstring``) are still masked. For all other commands, code_file=True is used
# to avoid mangling legitimate source/config dumps (``MAX_TOKENS=100``, ``"apiKey": "x"`` fixtures,
# ``postgresql://{user}`` f-string templates). See issue #43025.
_ENV_DUMP_COMMANDS = frozenset({"env", "printenv", "set", "export", "declare"}) _ENV_DUMP_COMMANDS = frozenset({"env", "printenv", "set", "export", "declare"})
# Commands that read file contents to stdout. A ``.env`` target is a credential # Commands that read file contents to stdout. A ``.env`` target is a credential
@@ -702,6 +796,8 @@ def _has_known_prefix_substring(text: str) -> bool:
# ADDITIVE-ONLY: a plugin can extend what gets masked but cannot weaken a # ADDITIVE-ONLY: a plugin can extend what gets masked but cannot weaken a
# built-in, so it can only over-redact. Keyed by registration source so plugin # built-in, so it can only over-redact. Keyed by registration source so plugin
# unload has a clean seam to drop ONE plugin's patterns. # unload has a clean seam to drop ONE plugin's patterns.
# There is deliberately no public removal API — additive-only stands; unload is a host-owned lifecycle
# concern. See #64229.
_PLUGIN_PREFIX_PATTERNS: dict = {} _PLUGIN_PREFIX_PATTERNS: dict = {}
_registry_lock = threading.Lock() _registry_lock = threading.Lock()
+12 -1
View File
@@ -98,6 +98,10 @@ class _ManagedAttempt:
Relay can invoke callbacks while another still owns the captured Context (hence the Relay can invoke callbacks while another still owns the captured Context (hence the
copy); nested relay calls run unmanaged — see relay_runtime.managed_callback_guard.""" copy); nested relay calls run unmanaged — see relay_runtime.managed_callback_guard."""
def guarded() -> Any: def guarded() -> Any:
# See #77244.
# See #77244.
# Hermes-side callbacks run while the native pipeline drives this stream; nested relay calls
# they make must bypass managed execution (#77244).
with relay_runtime.managed_callback_guard(): with relay_runtime.managed_callback_guard():
return callback(*args) return callback(*args)
@@ -223,7 +227,14 @@ def stream_current(
With ``completed_response_predicate`` set, a factory that ignores ``stream=True`` and returns a With ``completed_response_predicate`` set, a factory that ignores ``stream=True`` and returns a
complete response is unwrapped and returned directly (pre-Relay behavior). Detecting that primes complete response is unwrapped and returned directly (pre-Relay behavior). Detecting that primes
the lazy pipeline: a genuine first chunk is buffered, but provider latency and pre-first-yield the lazy pipeline: a genuine first chunk is buffered, but provider latency and pre-first-yield
errors may surface before this returns.""" errors may surface before this returns.
AnthropicAuxiliaryClient and other shims that ignore ``stream=True``), unwrap and return the completed
response directly. This mirrors the pre-Relay behavior where ``call_llm(stream=True)`` returned the raw
response and the consumer's own ``hasattr(stream, "choices")`` check handled it (#11732, #55933) —
without the unwrap the response stays trapped as ``final_response`` on the inner ManagedLlmStream and
the outer consumer sees an empty stream.
"""
session_id = _current_session_id() session_id = _current_session_id()
# Inside a managed callback (on the Relay session's loop) a nested ManagedLlmStream would be # Inside a managed callback (on the Relay session's loop) a nested ManagedLlmStream would be
# iterated synchronously on that loop, which asyncio forbids; the outer stream tracks this attempt. # iterated synchronously on that loop, which asyncio forbids; the outer stream tracks this attempt.
+7
View File
@@ -1139,6 +1139,13 @@ def resolve_execution_context(session_id: str) -> tuple[RelayRuntime | None, Rel
# Nested managed execution is impossible (see _MANAGED_CALLBACK_DEPTH); the outer scope # Nested managed execution is impossible (see _MANAGED_CALLBACK_DEPTH); the outer scope
# still records the tool-level event. # still records the tool-level event.
if _MANAGED_CALLBACK_DEPTH.get() > 0 or not relay_instrumentation_enabled(): if _MANAGED_CALLBACK_DEPTH.get() > 0 or not relay_instrumentation_enabled():
# A managed Relay callback is already executing on this logical call path (e.g. the native
# ``tools.execute`` pipeline is mid-dispatch of a Hermes tool). Nested managed execution here is
# structurally impossible: the native pipeline binds its Futures to the OUTER call's event loop,
# which is blocked inside the synchronous tool callback until the tool returns. A nested managed LLM
# call (the vision_analyze auxiliary path) therefore awaits a foreign-loop Future that can never
# complete — "attached to a different loop" at best, deadlock at worst, and "Event loop is closed"
# during shutdown when the orphaned Future is completed late (#77244).
return None, None, None return None, None, None
turn = active_turn(session_id) turn = active_turn(session_id)
host = turn.lease.live_runtime() if turn is not None else None host = turn.lease.live_runtime() if turn is not None else None
+1
View File
@@ -30,6 +30,7 @@ def execute(
# Everything the tool transitively calls (incl. auxiliary LLM calls on worker # Everything the tool transitively calls (incl. auxiliary LLM calls on worker
# threads) must bypass managed Relay: the pipeline's Futures bind to THIS loop, # threads) must bypass managed Relay: the pipeline's Futures bind to THIS loop,
# which is blocked until the tool returns. # which is blocked until the tool returns.
# See #77244.
with relay_runtime.managed_callback_guard(): with relay_runtime.managed_callback_guard():
return callback(final_args) return callback(final_args)
+5 -1
View File
@@ -25,7 +25,11 @@ _DOMINANCE_RATIO = 0.5
def is_repetition_dominated(text: str) -> bool: def is_repetition_dominated(text: str) -> bool:
"""True when a single 60+ char substring recurs often enough to cover at least half """True when a single 60+ char substring recurs often enough to cover at least half
of ``text`` — the signature of a repetition loop. Fail-open for non-string/short input.""" of ``text`` — the signature of a repetition loop. Fail-open for non-string/short input.
That shape is the signature of a model repetition loop (issue #86581), and continuing such a fragment is
pointless — the continuation nudge would just stitch more repeated text into the final response.
"""
if not isinstance(text, str): if not isinstance(text, str):
return False return False
n = len(text) n = len(text)
+15 -2
View File
@@ -95,7 +95,12 @@ def strip_interrupted_tool_tails(agent_history: List[Dict[str, Any]]) -> List[Di
def strip_dangling_tool_call_tail(agent_history: List[Dict[str, Any]]) -> List[Dict[str, Any]]: def strip_dangling_tool_call_tail(agent_history: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""Strip a trailing ``assistant(tool_calls)`` with NO answers — a call that killed the gateway itself """Strip a trailing ``assistant(tool_calls)`` with NO answers — a call that killed the gateway itself
(``docker restart``) left zero ``tool`` rows, invisible to ``strip_interrupted_tool_tails``. A partially (``docker restart``) left zero ``tool`` rows, invisible to ``strip_interrupted_tool_tails``. A partially
answered block still resumes. Read-only tails are dropped; side-effecting ones get UNKNOWN-effect results.""" answered block still resumes. Read-only tails are dropped; side-effecting ones get UNKNOWN-effect results.
On resume the model sees an unanswered tool call at the tail and naturally re-issues it — which restarts
the gateway again, producing the infinite reboot loop in #49201. ``strip_interrupted_tool_tails`` does
not catch this because there is no tool result to inspect for an interrupt marker.
"""
if not agent_history: if not agent_history:
return agent_history return agent_history
last = agent_history[-1] last = agent_history[-1]
@@ -124,6 +129,8 @@ def sanitize_replay_history(agent_history: List[Dict[str, Any]]) -> List[Dict[st
# --- Stale dangerous-confirmation text expiry --- # --- Stale dangerous-confirmation text expiry ---
# Short on purpose: a dangerous confirmation must not survive any restart or resume gap. # Short on purpose: a dangerous confirmation must not survive any restart or resume gap.
# ────────────────────────────────────────────────────────────────────── Stale dangerous-confirmation text
# expiry (#59607) ──────────────────────────────────────────────────────────────────────
_DANGEROUS_CONFIRMATION_EXPIRY_SECONDS = 60.0 _DANGEROUS_CONFIRMATION_EXPIRY_SECONDS = 60.0
# Phrases that unlock destructive host actions; case-insensitive substring match so trailing punctuation / # Phrases that unlock destructive host actions; case-insensitive substring match so trailing punctuation /
@@ -155,7 +162,13 @@ def strip_stale_dangerous_confirmations(
) -> List[Dict[str, Any]]: ) -> List[Dict[str, Any]]:
"""Redact IN PLACE dangerous-confirmation text older than ``expiry_seconds`` in user messages: a confirmation """Redact IN PLACE dangerous-confirmation text older than ``expiry_seconds`` in user messages: a confirmation
surviving a restart reads as a fresh re-confirmation minutes later. Untimestamped messages (legacy surviving a restart reads as a fresh re-confirmation minutes later. Untimestamped messages (legacy
transcripts, test scaffolding) are left untouched.""" transcripts, test scaffolding) are left untouched.
See #59607.
On the next inbound message — possibly a casual "are you there?" from the user minutes later — the LLM
sees the stale confirmation and may interpret the new turn as a fresh re-confirmation, re-executing the
destructive action. This is the failure mode reported in #59607.
"""
if not agent_history: if not agent_history:
return agent_history return agent_history
cleaned: List[Dict[str, Any]] = [] cleaned: List[Dict[str, Any]] = []
+1
View File
@@ -76,6 +76,7 @@ _GLOBAL_ENV_EXACT = frozenset({
# API-server LISTENER settings — deployment config (compose/systemd env), # API-server LISTENER settings — deployment config (compose/systemd env),
# which the scoped runner reload must keep seeing or containers silently # which the scoped runner reload must keep seeing or containers silently
# lose the api_server platform. API_SERVER_KEY is a credential: NOT here. # lose the api_server platform. API_SERVER_KEY is a credential: NOT here.
# See #64674, #69379.
"API_SERVER_ENABLED", "API_SERVER_HOST", "API_SERVER_PORT", "API_SERVER_ENABLED", "API_SERVER_HOST", "API_SERVER_PORT",
"API_SERVER_CORS_ORIGINS", "API_SERVER_CORS_ORIGINS",
# Relay-connector ROUTING stamps injected by managed deploys. Every reader # Relay-connector ROUTING stamps injected by managed deploys. Every reader
+2
View File
@@ -242,6 +242,8 @@ def _b64e(raw: bytes) -> str:
def _derive_encrypted_cache_key(access_token: str, salt: bytes) -> bytes: def _derive_encrypted_cache_key(access_token: str, salt: bytes) -> bytes:
"""HKDF the local cache key from the bootstrap BWS token. cryptography is imported """HKDF the local cache key from the bootstrap BWS token. cryptography is imported
lazily: eagerly mapping ``_rust.pyd`` on Windows blocks the updater replacing it.""" lazily: eagerly mapping ``_rust.pyd`` on Windows blocks the updater replacing it."""
# Keep the native cryptography extension lazy. Most CLI commands import this module while building
# argparse, even though only encrypted-cache reads/writes need it. See #73381.
from cryptography.hazmat.primitives import hashes from cryptography.hazmat.primitives import hashes
from cryptography.hazmat.primitives.kdf.hkdf import HKDF from cryptography.hazmat.primitives.kdf.hkdf import HKDF
+9 -1
View File
@@ -174,7 +174,12 @@ def list_sources(*, scope: Optional[str] = None) -> List[SecretSource]:
def list_plugin_sources() -> List[SecretSource]: def list_plugin_sources() -> List[SecretSource]:
"""Sources registered outside the bundled set: global ``"plugin"`` origins """Sources registered outside the bundled set: global ``"plugin"`` origins
plus every scoped registration (bundled sources register with scope=None).""" plus every scoped registration (bundled sources register with scope=None).
Includes both legacy global plugin registrations (``_SOURCE_ORIGINS == "plugin"``) and the current
scope's profile-keyed registrations — every scoped entry is plugin-registered by definition, since
bundled sources register with ``scope=None`` (#64229 profile isolation).
"""
_ensure_builtin_sources() _ensure_builtin_sources()
with _REGISTRY_LOCK: with _REGISTRY_LOCK:
merged = {n: s for n, s in _SOURCES.items() if _SOURCE_ORIGINS.get(n) == "plugin"} merged = {n: s for n, s in _SOURCES.items() if _SOURCE_ORIGINS.get(n) == "plugin"}
@@ -374,6 +379,9 @@ def apply_all(secrets_cfg: dict, home_path: Path,
Profile aliasing: under a named profile an applied ``FOO_<PROFILE>`` Profile aliasing: under a named profile an applied ``FOO_<PROFILE>``
(credential-shaped suffixes only) also hydrates canonical ``FOO``, under the (credential-shaped suffixes only) also hydrates canonical ``FOO``, under the
same guards; disabled with ``secrets.profile_alias: false``. same guards; disabled with ``secrets.profile_alias: false``.
1. 2. 3. 4. See #58073.
See #51447.
""" """
env = environ if environ is not None else os.environ env = environ if environ is not None else os.environ
report = ApplyReport() report = ApplyReport()
+1
View File
@@ -22,6 +22,7 @@ class ActivityProvenance(str, Enum):
UNKNOWN = "unknown" UNKNOWN = "unknown"
# Compression writers: heartbeat, host timeout, cooldown, turn hold. # Compression writers: heartbeat, host timeout, cooldown, turn hold.
# See #72424.
AGENT_COMPRESSION = "agent.compression" AGENT_COMPRESSION = "agent.compression"
AGENT_COMPRESSION_TIMEOUT = "agent.compression_timeout" AGENT_COMPRESSION_TIMEOUT = "agent.compression_timeout"
AGENT_COMPRESSION_COOLDOWN = "agent.compression_cooldown" AGENT_COMPRESSION_COOLDOWN = "agent.compression_cooldown"
+15 -2
View File
@@ -309,7 +309,12 @@ class SessionPersistenceMixin:
def _persist_session(self, messages: List[Dict], conversation_history: List[Dict] = None): def _persist_session(self, messages: List[Dict], conversation_history: List[Dict] = None):
"""Save to JSON log and SQLite on any exit path. Trailing empty-response scaffolding is dropped from """Save to JSON log and SQLite on any exit path. Trailing empty-response scaffolding is dropped from
the live list; the persist override is applied to the DB row only.""" the live list; the persist override is applied to the DB row only.
The persist user-message *override* is NOT applied here — it is resolved inside
``_flush_messages_to_session_db`` and written only to the DB row, never mutating the live message
list used by the API call (#48677 is thus closed for every persist caller, not just this one).
"""
from agent.agent_runtime_helpers import note_turn_persisted from agent.agent_runtime_helpers import note_turn_persisted
with _persist_lock(self): with _persist_lock(self):
self._drop_trailing_empty_response_scaffolding(messages) self._drop_trailing_empty_response_scaffolding(messages)
@@ -356,7 +361,15 @@ class SessionPersistenceMixin:
"""Persist un-flushed messages to SQLite. Dedup is the intrinsic ``_DB_PERSISTED_MARKER`` on each written """Persist un-flushed messages to SQLite. Dedup is the intrinsic ``_DB_PERSISTED_MARKER`` on each written
dict — not positional slices (drift after sequence repair) nor an ``id(msg)`` set (address reuse). The dict — not positional slices (drift after sequence repair) nor an ``id(msg)`` set (address reuse). The
persist override touches the written row only. A compression-closed session adopts its live tip and persist override touches the written row only. A compression-closed session adopts its live tip and
retries exactly once.""" retries exactly once.
Deduplicates via an intrinsic ``_DB_PERSISTED_MARKER`` stamped on each written message dict, so
repeated calls (from multiple exit paths) only write truly new messages — preventing the
duplicate-write bug (#860) without relying on positional slices that can drift after
message-sequence repair, and without a retained ``id(msg)`` set that CPython could alias onto a
freed-then-reused address (#50372). The ``_flushed_db_message_ids`` attribute is now only a one-shot
seed (translated to markers, then cleared each flush), not a persisted set.
"""
# Persistence-isolated agents (background review fork) share the parent's session_id for cache warmth; # Persistence-isolated agents (background review fork) share the parent's session_id for cache warmth;
# a write here would land the curator's turn in the user's real history. # a write here would land the curator's turn in the user's real history.
if getattr(self, "_persist_disabled", False) or not self._session_db: if getattr(self, "_persist_disabled", False) or not self._session_db:
+12 -1
View File
@@ -181,7 +181,18 @@ def iter_configured_hooks(cfg: Optional[Dict[str, Any]]) -> List[ShellHookSpec]:
def re_register_config_hooks() -> None: def re_register_config_hooks() -> None:
"""Re-register after a plugin force-reload cleared the manager's hooks; only this home's keys """Re-register after a plugin force-reload cleared the manager's hooks; only this home's keys
are cleared (profile A's reload never drops B), never re-prompts.""" are cleared (profile A's reload never drops B), never re-prompts.
``PluginManager.discover_and_load(force=True)`` unloads via the ownership ledger and clears the
manager's ``_hooks`` dict, which silently drops shell hooks that were registered from ``config.yaml`` at
startup (they are config-owned, not plugin-owned, so the ledger cannot restore them). Clear the
idempotence set and re-run ``register_from_config()`` so hooks are wired again (#60036 / PR #60267;
tracking #64178 — salvaged from PR #64188).
Only the idempotence keys for the *current* Hermes home are cleared — ``discover_and_load(force=True)``
only unloads the manager scoped to that one home, so clearing every home's keys would make a
force-reload in profile A drop profile B's still-live registration from the ledger and duplicate it on
B's next registration call (#92682 review).
"""
_forget_home_registrations(_registered, _registered_lock) _forget_home_registrations(_registered, _registered_lock)
from hermes_cli.config import load_config from hermes_cli.config import load_config
register_from_config(load_config()) register_from_config(load_config())
+9 -1
View File
@@ -129,7 +129,15 @@ def build_bundle_invocation_message(
loaded_skill_names, missing_skill_names)`` or ``None`` if the bundle wasn't loaded_skill_names, missing_skill_names)`` or ``None`` if the bundle wasn't
found. Uninstalled members are skipped with a note; disabled ones too, since found. Uninstalled members are skipped with a note; disabled ones too, since
``_load_skill_payload`` bypasses the scan-time filter (``platform`` scopes ``_load_skill_payload`` bypasses the scan-time filter (``platform`` scopes
that check — gateway passes it, None resolves from env).""" that check — gateway passes it, None resolves from env).
Disabled skills are also skipped: bundles load members via ``_load_skill_payload`` directly, bypassing
the scan-time disabled filter in ``get_skill_commands()``, so the disabled list must be re-applied here.
``platform`` scopes the check to a specific platform's ``skills.platform_disabled`` config (gateway
dispatch passes it explicitly because the gateway handles multiple platforms in one process); when
*None*, the platform resolves from session env vars and the global disabled list still applies. Mirrors
the stacked-skill gate in gateway dispatch (#58888).
"""
info = get_skill_bundles().get(cmd_key) info = get_skill_bundles().get(cmd_key)
if not info: if not info:
return None return None
+48 -5
View File
@@ -61,7 +61,13 @@ def append_user_instruction(parts: list, instruction: str) -> str:
"""Append the instruction line to ``parts``; return the stable prefix, which """Append the instruction line to ``parts``; return the stable prefix, which
ends exactly at the instruction marker so (registered with ends exactly at the instruction marker so (registered with
``agent.prompt_cache_boundary``) the cache planner can break on the scaffold. ``agent.prompt_cache_boundary``) the cache planner can break on the scaffold.
Single construction site guarantees the prefix is a byte-prefix of the message.""" Single construction site guarantees the prefix is a byte-prefix of the message.
Shared by every builder that ends a static skill scaffold with the caller-supplied volatile instruction
(single-skill invocations, cron job prompts). Keeping construction in one place guarantees the
registered prefix stays a byte-prefix of the built message — the invariant the request-time split
depends on. See #81867.
"""
stable_prefix = "\n".join(parts) + "\n" + _SINGLE_SKILL_INSTRUCTION stable_prefix = "\n".join(parts) + "\n" + _SINGLE_SKILL_INSTRUCTION
parts.append(f"{_SINGLE_SKILL_INSTRUCTION}{instruction}") parts.append(f"{_SINGLE_SKILL_INSTRUCTION}{instruction}")
return stable_prefix return stable_prefix
@@ -116,7 +122,11 @@ def _cut_after(message: str, marker: str, stop_marker: str, find) -> Optional[st
def _resolve_skill_commands_platform() -> Optional[str]: def _resolve_skill_commands_platform() -> Optional[str]:
"""Current platform scope for disabled-skill filtering, or None (CLI, RL, """Current platform scope for disabled-skill filtering, or None (CLI, RL,
scripts). A change invalidates the scan cache so each platform sees its scripts). A change invalidates the scan cache so each platform sees its
own ``skills.platform_disabled`` view.""" own ``skills.platform_disabled`` view.
Used to detect when the active platform has shifted so :func:`get_skill_commands` can drop a stale cache
that was populated for a different platform's ``skills.platform_disabled`` view (#14536).
"""
try: try:
from gateway.session_context import get_session_env from gateway.session_context import get_session_env
resolved_platform = os.getenv("HERMES_PLATFORM") or get_session_env("HERMES_SESSION_PLATFORM") resolved_platform = os.getenv("HERMES_PLATFORM") or get_session_env("HERMES_SESSION_PLATFORM")
@@ -127,7 +137,14 @@ def _resolve_skill_commands_platform() -> Optional[str]:
def _resolve_skill_commands_home() -> str: def _resolve_skill_commands_home() -> str:
"""Effective Hermes home the scan is scoped to (profiles carry their own """Effective Hermes home the scan is scoped to (profiles carry their own
``skills.external_dirs``, so a profile switch must invalidate the cache).""" ``skills.external_dirs``, so a profile switch must invalidate the cache).
A gateway session can switch between profiles that each carry their own ``skills.external_dirs`` (via
``set_hermes_home_override``), but the module-level scan only tracked
``_resolve_skill_commands_platform()``. Switching profiles without a platform change left the previous
profile's skill list cached, so ``get_skill_commands()`` reported a cache miss for skills that only
exist under the new profile (#88023).
"""
from hermes_constants import get_hermes_home from hermes_constants import get_hermes_home
return str(get_hermes_home()) return str(get_hermes_home())
@@ -248,6 +265,10 @@ def _build_skill_message(
parts.append("") parts.append("")
# Everything before the volatile instruction is a stable scaffold; the # Everything before the volatile instruction is a stable scaffold; the
# registered boundary lets the cache planner break there (see append_user_instruction). # registered boundary lets the cache planner break there (see append_user_instruction).
# Everything before the caller-supplied instruction is a stable scaffold; declare the exact boundary
# so the Anthropic cache planner can put a breakpoint on it instead of caching the whole message as
# one atomic block (#81867). The static instruction prose stays on the stable side; the volatile
# instruction (webhook payload, ticket IDs, timestamps) and any runtime note ride in the tail.
stable_prefix = append_user_instruction(parts, user_instruction) stable_prefix = append_user_instruction(parts, user_instruction)
if runtime_note: if runtime_note:
parts += ["", f"[Runtime note: {runtime_note}]"] parts += ["", f"[Runtime note: {runtime_note}]"]
@@ -263,6 +284,9 @@ def _render_skill_block(
"""Bump Curator usage tracking (never fatal) and build the message block for one loaded skill.""" """Bump Curator usage tracking (never fatal) and build the message block for one loaded skill."""
loaded_skill, skill_dir, skill_name = loaded loaded_skill, skill_dir, skill_name = loaded
try: try:
# Track active usage for Curator lifecycle management (#17782)
# Track active usage for Curator lifecycle management (#17782)
# Track active usage for Curator lifecycle management (#17782)
from tools.skill_usage import bump_use from tools.skill_usage import bump_use
bump_use(skill_name, task_id=task_id) bump_use(skill_name, task_id=task_id)
except Exception: except Exception:
@@ -344,6 +368,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
global _skill_commands, _skill_commands_platform, _skill_commands_home global _skill_commands, _skill_commands_platform, _skill_commands_home
platform = _resolve_skill_commands_platform() platform = _resolve_skill_commands_platform()
home = _resolve_skill_commands_home() home = _resolve_skill_commands_home()
# Build into a local map and publish once, at the end. Writing straight into the global made a scan's
# partial results visible to everything else in the process: a second, overlapping scan deduped against
# its own (empty) ``seen_names`` but collided against the first scan's already- published slugs, logging
# one bogus "already claimed" warning per skill — each naming the same skill as its own incumbent
# (#74574).
commands: Dict[str, Dict[str, Any]] = {} commands: Dict[str, Dict[str, Any]] = {}
try: try:
from tools.skills_tool import _skills_dir, _get_disabled_skill_names from tools.skills_tool import _skills_dir, _get_disabled_skill_names
@@ -356,6 +385,7 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
# Precedence: project (through the quarantine chokepoint) > local > external. # Precedence: project (through the quarantine chokepoint) > local > external.
# Resolve the local dir at call time: import-time SKILLS_DIR is frozen to # Resolve the local dir at call time: import-time SKILLS_DIR is frozen to
# the launch home, but a multiplexed profile scope may have changed it. # the launch home, but a multiplexed profile scope may have changed it.
# See #67277.
skills_dir = _skills_dir() skills_dir = _skills_dir()
iters = [iter_project_skill_files(d) for d in get_project_skills_dirs()] iters = [iter_project_skill_files(d) for d in get_project_skills_dirs()]
local = [skills_dir] if skills_dir.exists() else [] local = [skills_dir] if skills_dir.exists() else []
@@ -372,6 +402,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
# could accept the new map under a stale platform tag and serve another # could accept the new map under a stale platform tag and serve another
# platform's disabled-skill view. # platform's disabled-skill view.
with _publish_lock: with _publish_lock:
# Bare assignments are not atomic together: a reader landing between them sees the NEW map still
# carrying the OLD platform tag, and if that stale tag happens to match its own platform it accepts
# the map without rescanning — serving another platform's disabled-skill view, exactly the leak
# #14536 closed. Only the publish/lookup pair is locked; the scan above (file I/O, deferred imports)
# stays outside it.
_skill_commands = commands _skill_commands = commands
_skill_commands_platform = platform _skill_commands_platform = platform
_skill_commands_home = home _skill_commands_home = home
@@ -382,7 +417,10 @@ def get_skill_commands() -> Dict[str, Dict[str, Any]]:
"""Return the current skill commands mapping (scan first if empty). Rescans """Return the current skill commands mapping (scan first if empty). Rescans
when the platform scope (one gateway serving Telegram and Discord) or the when the platform scope (one gateway serving Telegram and Discord) or the
active profile's home (Desktop profile switch) changes, so each sees its active profile's home (Desktop profile switch) changes, so each sees its
own ``platform_disabled`` / ``external_dirs`` view.""" own ``platform_disabled`` / ``external_dirs`` view.
See #14536, #88023.
"""
current_platform = _resolve_skill_commands_platform() current_platform = _resolve_skill_commands_platform()
current_home = _resolve_skill_commands_home() current_home = _resolve_skill_commands_home()
with _publish_lock: with _publish_lock:
@@ -539,7 +577,12 @@ def build_preloaded_skills_prompt(skill_identifiers: list[str], task_id: str | N
"""Load skills for session-wide CLI/TUI preloading; returns (prompt_text, """Load skills for session-wide CLI/TUI preloading; returns (prompt_text,
loaded_skill_names, missing_identifiers). Disabled skills count as missing: loaded_skill_names, missing_identifiers). Disabled skills count as missing:
this path bypasses the scan-time filter, and ``hermes -s <skill>`` must not this path bypasses the scan-time filter, and ``hermes -s <skill>`` must not
force-load an operator-disabled skill.""" force-load an operator-disabled skill.
Disabled skills are treated the same as missing ones: this loads via a raw identifier straight into
``_load_skill_payload``, bypassing ``get_skill_commands()``'s scan-time disabled filter — mirrors the
bundle-invocation gate (#59156).
"""
loaded_names, missing, _disabled, prompt_parts = _load_skill_blocks( loaded_names, missing, _disabled, prompt_parts = _load_skill_blocks(
[(raw or "").strip() for raw in skill_identifiers], [(raw or "").strip() for raw in skill_identifiers],
lambda identifier: _load_skill_payload(identifier, task_id=task_id), lambda identifier: _load_skill_payload(identifier, task_id=task_id),
+23 -2
View File
@@ -289,7 +289,10 @@ def parse_config_string_list(value) -> List[str]:
"""Normalize a config value that may hold a JSON-array string into a list. """Normalize a config value that may hold a JSON-array string into a list.
``hermes config set`` stores lists as quoted JSON/Python-literal strings; ``hermes config set`` stores lists as quoted JSON/Python-literal strings;
treating one as a single name would silently filter nothing. A scalar treating one as a single name would silently filter nothing. A scalar
string still means one name.""" string still means one name.
See #13026, #86661.
"""
if isinstance(value, str): if isinstance(value, str):
if value.strip().startswith("["): if value.strip().startswith("["):
try: try:
@@ -412,7 +415,14 @@ _PROJECT_ROOT_MAX_DEPTH = 64 # walk-up bound for pathological cwds
def find_project_root(start: Optional[Path] = None) -> Optional[Path]: def find_project_root(start: Optional[Path] = None) -> Optional[Path]:
"""Nearest ancestor containing ``.git`` (dir or worktree file), or None. """Nearest ancestor containing ``.git`` (dir or worktree file), or None.
Without *start*, the surface's ``TERMINAL_CWD`` wins over process cwd so Without *start*, the surface's ``TERMINAL_CWD`` wins over process cwd so
cron/API surfaces inherit an interactive trust decision by project identity.""" cron/API surfaces inherit an interactive trust decision by project identity.
When *start* is not given, the surface's working directory wins over the process cwd: ``TERMINAL_CWD``
is the same per-surface workdir the terminal tool and cron jobs use (a cron job sets it from its per-job
``workdir`` without chdir'ing the scheduler process). This is what lets non-interactive surfaces inherit
a prior interactive trust decision by project identity — and a surface with no workdir in a trusted repo
simply resolves no project and loads nothing (#48975).
"""
try: try:
if start is None: if start is None:
from agent.runtime_cwd import scope_terminal_cwd from agent.runtime_cwd import scope_terminal_cwd
@@ -501,6 +511,16 @@ def get_untrusted_project_skills_root() -> Optional[Tuple[Path, int]]:
# cached under HERMES_HOME, never inside the repo); "dangerous" excludes the # cached under HERMES_HOME, never inside the repo); "dangerous" excludes the
# skill from index, list, view and slash commands ("caution" loads, as on the hub). # skill from index, list, view and slash commands ("caution" loads, as on the hub).
# ── Project skill quarantine (scan-time injection defense) ──────────────── Trust (`hermes skills trust`)
# is a REPO-level decision made once; the repo's skill content keeps changing underneath it with every pull.
# The hub install path runs skills_guard on install, but project skills are read straight from a checkout —
# without this gate a `git pull` could inject a malicious skill into an already-trusted repo with no scan
# anywhere (#48974). Every project SKILL.md's parent dir is scanned with the same skills_guard scanner the
# hub uses (content-hash cached, so the cost is one scan per skill per content change). A "dangerous"
# verdict quarantines the skill: it is excluded from the index, skills_list, skill_view, and slash commands.
# "caution" loads (matches hub behavior for prose-level keyword hits) — the quarantine is for
# high-confidence findings only. The scan cache lives under HERMES_HOME, never inside the repo (we don't
# write artifacts into the user's checkout).
_PROJECT_SCAN_SOURCE = "project-local" _PROJECT_SCAN_SOURCE = "project-local"
_PROJECT_QUARANTINE_CACHE: Dict[str, bool] = {} # skill_dir -> quarantined _PROJECT_QUARANTINE_CACHE: Dict[str, bool] = {} # skill_dir -> quarantined
@@ -552,6 +572,7 @@ def normalize_skill_lookup_name(identifier: str) -> str:
# (which follows the live profile-scoped HERMES_HOME), so normalization # (which follows the live profile-scoped HERMES_HOME), so normalization
# must agree with that exact root. Import deferred (cycle). # must agree with that exact root. Import deferred (cycle).
try: try:
# See #67277.
from tools import skills_tool as _skills_tool from tools import skills_tool as _skills_tool
primary_root = _skills_tool._skills_dir() primary_root = _skills_tool._skills_dir()
except Exception: except Exception:
+28 -3
View File
@@ -58,6 +58,15 @@ class StreamDeliveryMixin:
self._deliver_to_stream_callbacks(tail) self._deliver_to_stream_callbacks(tail)
self._record_streamed_assistant_text(tail) self._record_streamed_assistant_text(tail)
# Flush any benign partial-tag tail held by the think scrubber first (#17924): an innocent '<' at
# the end of the stream that turned out not to be a tag prefix should reach the UI. Then flush the
# context scrubber. Order matters — the think scrubber's output feeds into the context scrubber's
# state.
# Suppress reasoning/thinking blocks via the stateful scrubber (#17924). Earlier versions ran
# _strip_think_blocks per-delta here, which destroyed downstream state machines when a tag was split
# across deltas (e.g. MiniMax-M2.7 sends '<think>' and its content as separate deltas — regex case 2
# erased the first delta, so the CLI/gateway state machine never saw the open tag and leaked the
# reasoning content as regular response text).
if think_scrubber is not None: if think_scrubber is not None:
think_tail = think_scrubber.flush() think_tail = think_scrubber.flush()
deliver(ctx_scrubber.feed(think_tail) if think_tail and ctx_scrubber is not None else think_tail) deliver(ctx_scrubber.feed(think_tail) if think_tail and ctx_scrubber is not None else think_tail)
@@ -194,7 +203,10 @@ class StreamDeliveryMixin:
self._deliver_interim(visible, already_streamed=already_streamed, record=undelivered_parts or [visible]) self._deliver_interim(visible, already_streamed=already_streamed, record=undelivered_parts or [visible])
def _ensure_stream_writer_state(self) -> None: def _ensure_stream_writer_state(self) -> None:
"""Lazily create the single-writer guard fields (``AIAgent.__new__``-built instances skip ``agent_init``).""" """Lazily create the single-writer guard fields (``AIAgent.__new__``-built instances skip ``agent_init``).
See #65991.
"""
if getattr(self, "_stream_writer_lock", None) is None: if getattr(self, "_stream_writer_lock", None) is None:
self._stream_writer_lock = threading.Lock() self._stream_writer_lock = threading.Lock()
if getattr(self, "_stream_writer_tls", None) is None: if getattr(self, "_stream_writer_tls", None) is None:
@@ -209,6 +221,8 @@ class StreamDeliveryMixin:
Every attempt (each provider path, each retry) claims right before consuming; claiming bumps Every attempt (each provider path, each retry) claims right before consuming; claiming bumps
the shared token, so an earlier attempt still alive on another thread is superseded and its the shared token, so an earlier attempt still alive on another thread is superseded and its
late chunks fenced out. Stored per-thread: a thread that never claimed can never be fenced. late chunks fenced out. Stored per-thread: a thread that never claimed can never be fenced.
See #65991.
""" """
self._ensure_stream_writer_state() self._ensure_stream_writer_state()
with self._stream_writer_lock: with self._stream_writer_lock:
@@ -217,11 +231,18 @@ class StreamDeliveryMixin:
return token return token
def _stream_writer_is_current(self, token: int) -> bool: def _stream_writer_is_current(self, token: int) -> bool:
"""True when ``token`` is still the active writer, so a stream loop can bail the instant it is superseded.""" """True when ``token`` is still the active writer, so a stream loop can bail the instant it is superseded.
active writer — i.e. no newer stream attempt has claimed the sink since (#65991).
"""
return token == getattr(self, "_stream_writer_token", token) return token == getattr(self, "_stream_writer_token", token)
def _stream_writer_superseded(self) -> bool: def _stream_writer_superseded(self) -> bool:
"""True when this thread claimed the sink but a newer attempt has since claimed it (never for a non-claimer).""" """True when this thread claimed the sink but a newer attempt has since claimed it (never for a non-claimer).
stream attempt has since claimed it — i.e. this thread is a stale writer whose chunks must be
dropped (#65991).
"""
token = getattr(getattr(self, "_stream_writer_tls", None), "token", None) token = getattr(getattr(self, "_stream_writer_tls", None), "token", None)
return token is not None and token != getattr(self, "_stream_writer_token", token) return token is not None and token != getattr(self, "_stream_writer_token", token)
@@ -261,6 +282,7 @@ class StreamDeliveryMixin:
"""Fire all registered stream delta callbacks (display + TTS).""" """Fire all registered stream delta callbacks (display + TTS)."""
# A superseded stream must not interleave its tokens alongside the retry that replaced it. # A superseded stream must not interleave its tokens alongside the retry that replaced it.
if self._stream_writer_superseded(): if self._stream_writer_superseded():
# See #65991.
self._note_dropped_stream_writer("_fire_stream_delta") self._note_dropped_stream_writer("_fire_stream_delta")
return return
# One paragraph break before the first text delta after a tool iteration, without # One paragraph break before the first text delta after a tool iteration, without
@@ -274,6 +296,7 @@ class StreamDeliveryMixin:
# tag was split across deltas; memory-context spans split across chunks must not leak to # tag was split across deltas; memory-context spans split across chunks must not leak to
# the UI. Legacy callers lack the scrubber attributes and get the whole-string fallbacks. # the UI. Legacy callers lack the scrubber attributes and get the whole-string fallbacks.
think_scrubber = getattr(self, "_stream_think_scrubber", None) think_scrubber = getattr(self, "_stream_think_scrubber", None)
# See #5719.
scrubber = getattr(self, "_stream_context_scrubber", None) scrubber = getattr(self, "_stream_context_scrubber", None)
text = think_scrubber.feed(text) if think_scrubber is not None else self._strip_think_blocks(text) text = think_scrubber.feed(text) if think_scrubber is not None else self._strip_think_blocks(text)
text = scrubber.feed(text) if scrubber is not None else sanitize_context(text) text = scrubber.feed(text) if scrubber is not None else sanitize_context(text)
@@ -291,6 +314,8 @@ class StreamDeliveryMixin:
def _fire_reasoning_delta(self, text: str) -> None: def _fire_reasoning_delta(self, text: str) -> None:
"""Fire reasoning callback if registered; superseded writers are fenced like content deltas.""" """Fire reasoning callback if registered; superseded writers are fenced like content deltas."""
if self._stream_writer_superseded(): if self._stream_writer_superseded():
# Single-writer guard (#65991): fence out a superseded stream's reasoning deltas the same way as
# content deltas.
self._note_dropped_stream_writer("_fire_reasoning_delta") self._note_dropped_stream_writer("_fire_reasoning_delta")
return return
self._call_quietly(self.reasoning_callback, text) self._call_quietly(self.reasoning_callback, text)
+32 -1
View File
@@ -20,6 +20,14 @@ from agent.transports.types import NormalizedResponse, ToolCall, Usage
# xAI reserves ``tool_search`` for its server-side tool (HTTP 400 on client # xAI reserves ``tool_search`` for its server-side tool (HTTP 400 on client
# declarations); aliased on the wire, mapped back in normalize_response. # declarations); aliased on the wire, mapped back in normalize_response.
# xAI's chat-completions API reserves the function name ``tool_search`` for its own server-side tool and
# rejects any request declaring a client function with that name (HTTP 400 "The function name tool_search is
# reserved for the tool_search tool", #95003). The Tool Search bridge (tools/tool_search.py) assembles its
# client-side discovery tool under the same literal name for every provider, so Grok providers are unusable
# whenever the bridge is active. Mirror the web_search treatment in transports/codex.py
# (_rename_client_web_search_for_xai): alias the wire declaration and map the alias back in
# normalize_response. The alias value matches _CODEX_TOOL_SEARCH_ALIAS from the Codex-side fix for the same
# reserved-name class (#83122) so the two transports stay consistent.
_XAI_TOOL_SEARCH_ALIAS = "hermes_tool_search" _XAI_TOOL_SEARCH_ALIAS = "hermes_tool_search"
# Persistence-only / cross-transport message keys that strict OpenAI-compatible # Persistence-only / cross-transport message keys that strict OpenAI-compatible
@@ -70,7 +78,19 @@ def _add_prompt_cache_key(
survives compression rotation. A caller-supplied key is authoritative but is survives compression rotation. A caller-supplied key is authoritative but is
bounded to OpenAI's 64-char cap in place. Shares the Responses transport's hash bounded to OpenAI's 64-char cap in place. Shares the Responses transport's hash
so equivalent prefixes hit one bucket across modes. so equivalent prefixes hit one bucket across modes.
``cache_scope_id``, when provided, is the rotation-stable logical scope (compression-lineage root —
agent/prompt_cache_scope.py) and takes precedence over the physical ``session_id`` so the key survives
context-compression session rotation (#79017).
""" """
# Stable prompt-cache routing for the Codex/Responses aux path, mirroring the main transport
# (agent/transports/codex.py::build_kwargs, which sets prompt_cache_key =
# _content_cache_key(instructions, tools)). Without this, MoA acting-aggregator and other auxiliary
# Responses calls stay cache-cold while the main Responses transport is warm (issue #53735). The key is
# content-addressed from the static prefix (instructions + tool schemas) so it stays warm across
# turns/fires. Guard the top-level field the same way the main transport does: xAI Responses takes the
# key in extra_body (not top-level) and GitHub/Copilot Responses opts out of cache-key routing entirely
# — for those hosts, skip it here.
from agent.transports.codex import ( from agent.transports.codex import (
_bound_prompt_cache_key_field, _cache_scope_from_session_id, _content_cache_key _bound_prompt_cache_key_field, _cache_scope_from_session_id, _content_cache_key
) )
@@ -90,7 +110,14 @@ def _add_prompt_cache_key(
def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None: def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None:
"""Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary.""" """Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary.
Hermes' internal effort set extends the wire vocabulary with ``ultra`` (the /reasoning command documents
none..xhigh|max|ultra). OpenAI- compatible wires — OpenRouter chief among them — accept exactly
max|xhigh|high|medium|low|minimal|none and reject the extension with HTTP 400 (#89503). Clamp against
the declared wire vocabulary via the shared policy in ``agent.reasoning_effort``; provider profiles with
narrower sets clamp again downstream.
"""
if not isinstance(reasoning_config, dict): if not isinstance(reasoning_config, dict):
return reasoning_config return reasoning_config
effort = str(reasoning_config.get("effort") or "").strip().lower() effort = str(reasoning_config.get("effort") or "").strip().lower()
@@ -104,6 +131,10 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) ->
return None return None
normalized_model = (model or "").strip().lower().removeprefix("google/") normalized_model = (model or "").strip().lower().removeprefix("google/")
# Gemini-only; Gemma/PaLM on the same provider 400 on the field even as ``{"includeThoughts": False}``. # Gemini-only; Gemma/PaLM on the same provider 400 on the field even as ``{"includeThoughts": False}``.
# ``thinking_config`` is a Gemini-only request parameter. The same ``gemini`` provider also serves Gemma
# (and historically PaLM/Bard); those reject the field with HTTP 400 "Unknown name 'thinking_config':
# Cannot find field" — including the polite ``{"includeThoughts": False}`` form. Omit the field entirely
# on non-Gemini models. (#17426)
if not normalized_model.startswith("gemini"): if not normalized_model.startswith("gemini"):
return None return None
effort = str(reasoning_config.get("effort", "medium") or "medium").strip().lower() effort = str(reasoning_config.get("effort", "medium") or "medium").strip().lower()
+66
View File
@@ -12,6 +12,8 @@ from typing import Any, Callable, Optional
from agent.reasoning_effort import ( from agent.reasoning_effort import (
ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
# Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort):
# per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365).
codex_supported_efforts, codex_supported_efforts,
) )
from agent.transports.base import ProviderTransport from agent.transports.base import ProviderTransport
@@ -21,6 +23,7 @@ logger = logging.getLogger(__name__)
# Cron fires use ``cron_<job_id>_<YYYYMMDD_HHMMSS>``; the per-fire timestamp is # Cron fires use ``cron_<job_id>_<YYYYMMDD_HHMMSS>``; the per-fire timestamp is
# stripped so repeat fires of one job share a cache scope. # stripped so repeat fires of one job share a cache scope.
# See #51395, #52295.
_CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$") _CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$")
@@ -64,6 +67,10 @@ _XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search"
# OpenCode /v1/responses rejects client tools using these names (HTTP 400 # OpenCode /v1/responses rejects client tools using these names (HTTP 400
# "custom function name 'X' is reserved"); xAI reserves ``tool_search`` for # "custom function name 'X' is reserved"); xAI reserves ``tool_search`` for
# Grok's native Tool Search. Aliased as hermes_<name>. # Grok's native Tool Search. Aliased as hermes_<name>.
# OpenCode's /v1/responses endpoints (Zen and Go, including custom providers pointing at opencode.ai)
# reserve certain function names server-side and reject client tools that use them with HTTP 400 ("custom
# function name 'X' is reserved"). Same treatment as the xAI web_search collision: rename on the wire
# (hermes_<name>), map back in normalize_response so Hermes dispatch is unaffected. See #85589.
_OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files") _OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files")
_XAI_RESERVED_TOOL_NAMES = ("tool_search",) _XAI_RESERVED_TOOL_NAMES = ("tool_search",)
_RESERVED_TOOL_ALIAS_PREFIX = "hermes_" _RESERVED_TOOL_ALIAS_PREFIX = "hermes_"
@@ -126,6 +133,11 @@ def _xai_prefers_native_web_search() -> bool:
"""True when xAI Responses should use Grok's native ``web_search`` built-in. """True when xAI Responses should use Grok's native ``web_search`` built-in.
Web-search registry first, then the legacy ``_get_search_backend`` probe; fails closed to native (True). Web-search registry first, then the legacy ``_get_search_backend`` probe; fails closed to native (True).
Delegates to the web-search registry's provider resolution (which reads ``web.search_backend`` /
``web.backend`` from config) and checks whether the resolved provider is xAI. On any resolution failure,
returns True (fail-closed to native — preserves the #48108 incomplete-hang fix rather than risk
reintroducing it).
""" """
try: try:
from agent.web_search_registry import get_active_search_provider from agent.web_search_registry import get_active_search_provider
@@ -160,9 +172,25 @@ def _alias_wire_tools(response_tools: Any, params: dict[str, Any], is_xai_respon
{**t, "name": _XAI_CLIENT_WEB_SEARCH_ALIAS} if is_client_web_search(t) else t for t in response_tools {**t, "name": _XAI_CLIENT_WEB_SEARCH_ALIAS} if is_client_web_search(t) else t for t in response_tools
] ]
wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search" wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search"
# OpenCode Responses backends reserve web_search / search_files as function names (HTTP 400 "custom
# function name 'X' is reserved", #85589). Alias them on the wire; normalize_response maps them back.
if response_tools and _is_opencode_responses_backend(params): if response_tools and _is_opencode_responses_backend(params):
response_tools, _oc_aliases = _alias_reserved_tools(response_tools, _OPENCODE_RESERVED_TOOL_NAMES) response_tools, _oc_aliases = _alias_reserved_tools(response_tools, _OPENCODE_RESERVED_TOOL_NAMES)
wire_aliases.update(_oc_aliases) wire_aliases.update(_oc_aliases)
# xAI server-side web search vs Hermes web providers. grok models on xAI's /v1/responses surface have a
# *native*, server-executed web search. A client-side function literally named ``web_search`` collides
# with that engine: declared as a plain ``function`` rather than ``{"type": "web_search"}``, the search
# dispatches but never reconciles → incomplete turn + 3 retries. Verified live against
# grok-composer-2.5-fast (2026-06); see #48108. Two modes, chosen by the user's web-search backend
# config: 1. **Native** (active/configured backend is ``xai``, or resolution fails): drop the client
# ``web_search`` function and declare xAI's built-in instead. 1:1 swap only when client ``web_search``
# was already present — never an additive grant. 2. **Client** (Firecrawl / Tavily / Exa / … configured
# or resolved): keep Hermes dispatch so ``web.backend`` / ``web.search_backend`` is honored, but rename
# the wire tool to ``hermes_web_search`` so Grok cannot hijack the name. The alias is mapped back to
# ``web_search`` in ``normalize_response``. Request-local alias provenance: every wire alias THIS
# request emits is recorded here and stashed on the transport, so the reverse rewrite in
# ``normalize_response`` applies only to aliases that were actually sent (never to a real tool that
# merely shares an alias-shaped name).
if is_xai_responses and response_tools: if is_xai_responses and response_tools:
response_tools, _xai_aliases = _alias_reserved_tools(response_tools, _XAI_RESERVED_TOOL_NAMES) response_tools, _xai_aliases = _alias_reserved_tools(response_tools, _XAI_RESERVED_TOOL_NAMES)
wire_aliases.update(_xai_aliases) wire_aliases.update(_xai_aliases)
@@ -183,6 +211,10 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
elif reasoning_config.get("effort"): elif reasoning_config.get("effort"):
reasoning_effort = reasoning_config["effort"] reasoning_effort = reasoning_config["effort"]
# Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker
# supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that
# repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a
# model's real ceiling (#87279).
if params.get("is_xai_responses", False): if params.get("is_xai_responses", False):
from agent.model_metadata import is_grok_46_family from agent.model_metadata import is_grok_46_family
@@ -215,6 +247,11 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op
hostname = base_url_hostname(str(base_url or "")).lower() hostname = base_url_hostname(str(base_url or "")).lower()
# Meta Model API: caching is opt-in via prompt_cache_retention (0% hits without). # Meta Model API: caching is opt-in via prompt_cache_retention (0% hits without).
# Meta Model API (api.meta.ai) only achieves prompt-cache hits on the Responses API with
# prompt_cache_retention; chat/completions stays cache-cold (0% vs 93-99% measured). Exact-hostname
# match per #32243.
# Meta Model API: prompt caching only on Responses API (0% on chat/completions vs 93-99% on /responses
# with retention). See #32243.
if hostname == "api.meta.ai": if hostname == "api.meta.ai":
return "24h" return "24h"
parts = hostname.split(".") parts = hostname.split(".")
@@ -229,6 +266,12 @@ def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]],
"""``pck_<sha256[:24]>`` of (scope_id, instructions, name-sorted tools), or None if nothing static. """``pck_<sha256[:24]>`` of (scope_id, instructions, name-sorted tools), or None if nothing static.
Routing hint only; ``scope_id`` keeps unrelated sessions off one bucket. Routing hint only; ``scope_id`` keeps unrelated sessions off one bucket.
``scope_id`` (pass ``_cache_scope_from_session_id(session_id)``) keeps unrelated sessions — independent
conversations, main vs. child/subagent, sibling children — from concentrating onto the same bucket
merely because their static prefix matches (see #78941), while still letting recurring cron fires of one
job share a stable key across their timestamped session_ids (the original #51395/#52295 fix this built
on). Sorting tools by name keeps the hash insertion-order independent.
""" """
if not instructions and not tools: if not instructions and not tools:
return None return None
@@ -421,6 +464,20 @@ class ResponsesApiTransport(ProviderTransport):
cache key / xAI conv header), max_tokens, timeout, request_overrides, provider, base_url, cache key / xAI conv header), max_tokens, timeout, request_overrides, provider, base_url,
is_github_responses, is_codex_backend, is_xai_responses, github_reasoning_extra, is_github_responses, is_codex_backend, is_xai_responses, github_reasoning_extra,
context_management, replay_encrypted_reasoning. context_management, replay_encrypted_reasoning.
params: instructions: str — system prompt (extracted from messages[0] if not given)
reasoning_config: dict | None — {effort, enabled} session_id: str | None — transcript/session id;
drives the Codex ``session_id`` header, and is the cache-scope fallback when no ``cache_scope_id``
is given cache_scope_id: str | None — rotation-stable logical scope id (compression-lineage root;
see agent/prompt_cache_scope.py). Preferred over session_id when deriving the prompt_cache_key
content hash and the xAI x-grok-conv-id header; the Codex x-client-request-id header mirrors the
resulting body key. Keeps the cache warm across context-compression session rotation (#79017)
max_tokens: int | None — max_output_tokens timeout: float | None — per-request timeout forwarded to
the SDK request_overrides: dict | None — extra kwargs merged in provider: str | None — provider name
for backend-specific logic base_url: str | None — endpoint URL base_url_hostname: str | None —
hostname for backend detection is_github_responses: bool — Copilot/GitHub models backend
is_codex_backend: bool — chatgpt.com/backend-api/codex is_xai_responses: bool — xAI/Grok backend
github_reasoning_extra: dict | None — Copilot reasoning params
""" """
from run_agent import DEFAULT_AGENT_IDENTITY from run_agent import DEFAULT_AGENT_IDENTITY
@@ -490,6 +547,8 @@ class ResponsesApiTransport(ProviderTransport):
_bound_prompt_cache_key_field(kwargs) _bound_prompt_cache_key_field(kwargs)
# Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing. # Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing.
# Grok 4.6 accepts Priority Processing, but continue stripping stale or unsupported tier values on
# every other xAI path. See #28490 and #84799.
if is_xai_responses: if is_xai_responses:
from agent.model_metadata import is_grok_46_family from agent.model_metadata import is_grok_46_family
@@ -521,6 +580,13 @@ class ResponsesApiTransport(ProviderTransport):
_merge_extra_headers(kwargs, **{"x-grok-conv-id": _cache_scope}) _merge_extra_headers(kwargs, **{"x-grok-conv-id": _cache_scope})
# xAI reads prompt_cache_key from the body; extra_body survives SDK builds whose # xAI reads prompt_cache_key from the body; extra_body survives SDK builds whose
# Responses.stream() dropped the typed kwarg. An explicit request_overrides value wins. # Responses.stream() dropped the typed kwarg. An explicit request_overrides value wins.
# Scoped like the body cache key below — otherwise cron's per-fire timestamp in session_id
# (cron_<id>_<ts>) pins every fire of the same job to a different xAI backend server (#78941).
# xAI Responses cache-routing — body-level field per
# https://docs.x.ai/developers/advanced-api-usage/prompt-caching/maximizing-cache-hits. A
# caller's request_overrides={"prompt_cache_key": ...} lands on the top-level kwarg set above —
# read it back here so an explicit override actually governs the field xAI reads, instead of
# being silently outrun by the auto-derived cache_key (#78941).
existing_extra_body = kwargs.get("extra_body") existing_extra_body = kwargs.get("extra_body")
kwargs["extra_body"] = dict(existing_extra_body) if isinstance(existing_extra_body, dict) else {} kwargs["extra_body"] = dict(existing_extra_body) if isinstance(existing_extra_body, dict) else {}
kwargs["extra_body"].setdefault("prompt_cache_key", kwargs.get("prompt_cache_key", cache_key)) kwargs["extra_body"].setdefault("prompt_cache_key", kwargs.get("prompt_cache_key", cache_key))
+9
View File
@@ -48,6 +48,14 @@ class CodexAppServerClient:
) -> None: ) -> None:
self._codex_bin = codex_bin self._codex_bin = codex_bin
# codex needs LLM provider creds but must not receive Tier-1 Hermes secrets (gateway/GitHub/infra tokens). # codex needs LLM provider creds but must not receive Tier-1 Hermes secrets (gateway/GitHub/infra tokens).
# codex app-server is a model-driving CLI executor: it runs a model-chosen agentic loop that
# executes shell commands, so it legitimately needs LLM provider credentials
# (inherit_credentials=True) to authenticate against the model endpoint. But the previous
# `os.environ.copy()` also handed it every Tier-1 Hermes secret — gateway bot tokens, GitHub auth,
# Modal/Daytona infra tokens, the dashboard session token, AUXILIARY_* side-LLM keys,
# GATEWAY_RELAY_* auth — none of which a coding subprocess has any use for. Route through the
# centralized helper so Tier-1 + dynamic-internal secrets are always stripped while provider creds
# still flow, matching copilot_acp_client (#29157 sibling spawn-site gap).
spawn_env = hermes_subprocess_env(inherit_credentials=True) spawn_env = hermes_subprocess_env(inherit_credentials=True)
if env: if env:
spawn_env.update(env) spawn_env.update(env)
@@ -71,6 +79,7 @@ class CodexAppServerClient:
# Hide the console the codex child would otherwise flash on Windows (#56747). # Hide the console the codex child would otherwise flash on Windows (#56747).
# Hide-only — stdio pipes stay intact for the app-server wire. # Hide-only — stdio pipes stay intact for the app-server wire.
# See #56747.
from hermes_cli._subprocess_compat import windows_hide_flags from hermes_cli._subprocess_compat import windows_hide_flags
self._proc = subprocess.Popen( self._proc = subprocess.Popen(
@@ -328,6 +328,11 @@ class CodexAppServerSession:
"""Send a user message and block until turn/completed, bridging approvals and projecting items. """Send a user message and block until turn/completed, bridging approvals and projecting items.
post_tool_quiet_timeout: silence this long after a tool completes fast-fails and retires. post_tool_quiet_timeout: silence this long after a tool completes fast-fails and retires.
post_tool_quiet_timeout: if codex emits a tool completion and then goes quiet for this many seconds
without emitting another item or `turn/completed`, fast-fail and mark the session for retirement.
Mirrors openclaw beta.8's post-tool completion watchdog (#81697) so a wedged codex doesn't burn the
full turn deadline.
""" """
result = TurnResult() result = TurnResult()
if self._start_for(result): if self._start_for(result):
+4 -1
View File
@@ -86,7 +86,10 @@ class TTSProvider(CatalogProviderBase):
"""Whether output suits voice-bubble delivery (mirrors """Whether output suits voice-bubble delivery (mirrors
``tts.providers.<name>.voice_compatible``): True → the gateway converts ``tts.providers.<name>.voice_compatible``): True → the gateway converts
to Opus via ffmpeg if needed; False → regular audio attachment. Default to Opus via ffmpeg if needed; False → regular audio attachment. Default
False (opt in).""" False (opt in).
See #17843.
"""
return False return False
+6
View File
@@ -122,6 +122,12 @@ def build_api_request(
api_kwargs = agent._build_api_kwargs(api_messages, tools_for_api=tools_for_api) api_kwargs = agent._build_api_kwargs(api_messages, tools_for_api=tools_for_api)
# Surrogate chokepoint: tool descriptions, extra_body and kwargs strings can carry # Surrogate chokepoint: tool descriptions, extra_body and kwargs strings can carry
# invalid code points (HTTP 400). One walk makes the payload json.dumps()-safe. # invalid code points (HTTP 400). One walk makes the payload json.dumps()-safe.
# Outbound-request surrogate chokepoint (#50959): the messages were scrubbed above, but the rest of the
# request body — tool/function descriptions (session_search's ±-heavy text is the recorded repro),
# extra_body, system strings routed via kwargs — can still carry invalid code points that providers
# reject with a non-retryable HTTP 400 ("invalid unicode code point"). One in-place walk here guarantees
# the entire payload json.dumps()-safe regardless of which leaf produced the string. Fast no-op when the
# payload is clean.
_sanitize_structure_surrogates(api_kwargs) _sanitize_structure_surrogates(api_kwargs)
if agent._force_ascii_payload: if agent._force_ascii_payload:
_sanitize_structure_non_ascii(api_kwargs) _sanitize_structure_non_ascii(api_kwargs)
+45 -3
View File
@@ -172,6 +172,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None:
main_runtime = { main_runtime = {
k: getattr(agent, k, None) for k in ("model", "provider", "base_url", "api_key", "api_mode") k: getattr(agent, k, None) for k in ("model", "provider", "base_url", "api_key", "api_mode")
} }
# See #19027.
maybe_auto_title( maybe_auto_title(
session_db, session_db,
session_id, session_id,
@@ -197,7 +198,17 @@ def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> in
Prefers the LAST user message whose content exactly matches this turn's text, else Prefers the LAST user message whose content exactly matches this turn's text, else
the last user-originated turn; compaction handoffs are never the fallback. the last user-originated turn; compaction handoffs are never the fallback.
Returns -1 when there is no user-originated message.""" Returns -1 when there is no user-originated message.
Compression replaces list entries with fresh copies (and may append a todo-snapshot user message or a
restored user turn AFTER the surviving copy of the current turn's message), so a pre-compression index
is meaningless. Prefer the LAST user message whose content exactly matches this turn's text — the
surviving copy in the common case — so the injection stamp and the #48677 persist override can't land on
a todo-snapshot or historical row. Fall back to the last *user-originated* turn when no exact match
survives (merge-summary-into-tail rewrites the content but the trackers still need a live anchor).
Compaction handoffs must never become the fallback anchor (#80622) — they are reference-only
scaffolding, not the active ask.
"""
from agent.context_compressor import user_originated_turn_view from agent.context_compressor import user_originated_turn_view
fallback = -1 fallback = -1
@@ -225,11 +236,24 @@ def compression_made_progress(
orig_len: int, new_len: int, orig_tokens: int, new_tokens: int orig_len: int, new_len: int, orig_tokens: int, new_tokens: int
) -> bool: ) -> bool:
"""``True`` if a compression pass materially reduced the request: fewer rows, or a """``True`` if a compression pass materially reduced the request: fewer rows, or a
>5% token cut with the same rows (same floor as the overflow-handler retry).""" >5% token cut with the same rows (same floor as the overflow-handler retry).
Compression can succeed by summarising message contents — reducing the estimated request token count —
without reducing the message row count. Treating row count as the sole progress signal false-positives
on size-only wins and surfaces a misleading "Cannot compress further" failure even when post-compression
tokens are well below the model context window. See issue #39548 for an observed case: 220 → 220
messages, ~288k → ~183k tokens on a 1M-context model still triggered auto-reset.
The token reduction must be *material* (>5%) to count as progress — the same floor the overflow-handler
retry path uses (conversation_loop.py, 39550) — so a sub-5% wobble doesn't keep the multi-pass loop
spinning. See #39550.
"""
return new_len < orig_len or (orig_tokens > 0 and new_tokens < orig_tokens * 0.95) return new_len < orig_len or (orig_tokens > 0 and new_tokens < orig_tokens * 0.95)
# Back-compat alias: gateway callers and tests patch ``_compression_made_progress``. # Back-compat alias: gateway callers and tests patch ``_compression_made_progress``.
# Back-compat alias: this predicate was module-private until the gateway's session-hygiene recovery gate
# needed the same semantics (#79624). Keeping the old name bound means existing callers and any test that
# patches ``_compression_made_progress`` continue to work unchanged.
_compression_made_progress = compression_made_progress _compression_made_progress = compression_made_progress
@@ -296,7 +320,12 @@ def _should_idle_compact(
produced (``ContextCompressor.last_compression_rough_tokens``, same rough shape as produced (``ContextCompressor.last_compression_rough_tokens``, same rough shape as
``tokens``) — raises the floor to ``last + floor_tokens`` so the transcript must gain a ``tokens``) — raises the floor to ``last + floor_tokens`` so the transcript must gain a
floor's worth of NEW content first. ``0`` (nothing compacted yet / counter reset) keeps floor's worth of NEW content first. ``0`` (nothing compacted yet / counter reset) keeps
the original semantics exactly.""" the original semantics exactly.
A session that compacted to well above that target therefore stays above it forever, so every later idle
resume re-runs a full summarisation over a transcript that has not grown — minutes of silently blocked
prompt on a slow route, reclaiming nothing (#97239).
"""
if not enabled or idle_after_seconds <= 0 or idle_gap_seconds < idle_after_seconds or cooldown_active: if not enabled or idle_after_seconds <= 0 or idle_gap_seconds < idle_after_seconds or cooldown_active:
return False return False
effective_floor = floor_tokens effective_floor = floor_tokens
@@ -350,6 +379,12 @@ def _publish_runtime_main(agent: Any) -> None:
from agent.prompt_cache_scope import resolve_prompt_cache_scope_safe from agent.prompt_cache_scope import resolve_prompt_cache_scope_safe
# Rotation-stable prompt-cache scope (lineage root), memoized per segment; a new # Rotation-stable prompt-cache scope (lineage root), memoized per segment; a new
# session uses the physical id until build_api_kwargs re-resolves. # session uses the physical id until build_api_kwargs re-resolves.
# Memoized per segment on the agent, so this is a DB walk at most once per segment — except a
# brand-new session whose row lands later in turn setup (_ensure_db_session); that first turn falls
# back to the physical id here and the first build_api_kwargs re-resolves. Stays valid through a
# mid-turn compression rotation because the lineage root is by definition rotation-invariant
# (#79017). Resolved with the never-raising variant OUTSIDE the argument list, so a resolution
# failure can only lose the scope — never the whole runtime binding.
_cache_scope = resolve_prompt_cache_scope_safe(agent) or "" _cache_scope = resolve_prompt_cache_scope_safe(agent) or ""
set_runtime_main( set_runtime_main(
_str_attr(agent, "provider"), _str_attr(agent, "model"), _str_attr(agent, "provider"), _str_attr(agent, "model"),
@@ -569,6 +604,8 @@ def _collect_pre_llm_call_context(
sender_id=getattr(agent, "_user_id", None) or "", sender_id=getattr(agent, "_user_id", None) or "",
) )
try: try:
# Spill oversized per-hook context to disk so a runaway plugin can't inflate every subsequent
# turn's prompt. Ported from openai/codex PR #21069 ("Spill large hook outputs from context").
from tools.hook_output_spill import ( from tools.hook_output_spill import (
get_spill_config as _spill_cfg, spill_if_oversized as _spill_if_oversized get_spill_config as _spill_cfg, spill_if_oversized as _spill_if_oversized
) )
@@ -738,6 +775,11 @@ def build_turn_context(
# Tag log records on this thread with the session ID for ``hermes logs``; bind the # Tag log records on this thread with the session ID for ``hermes logs``; bind the
# skill write-origin ContextVar; restore the primary runtime after a fallback turn. # skill write-origin ContextVar; restore the primary runtime after a fallback turn.
# NOTE: the DB session row is created later, AFTER the system prompt is restored/built (see
# _ensure_db_session() below the system-prompt block). Creating it here — before _cached_system_prompt
# is populated — inserts a row with system_prompt=NULL on a fresh API/gateway agent that carries
# client-managed history, which then trips the "stored system prompt is null; rebuilding from scratch"
# warning and a needless first-turn prefix cache miss. (Issue #45499.)
set_session_context(agent.session_id) set_session_context(agent.session_id)
set_current_write_origin(getattr(agent, "_memory_write_origin", "assistant_tool")) set_current_write_origin(getattr(agent, "_memory_write_origin", "assistant_tool"))
agent._restore_primary_runtime() agent._restore_primary_runtime()
+17
View File
@@ -42,6 +42,14 @@ class CompactionOutcome:
def _clear_overflow_warn(agent: Any) -> None: def _clear_overflow_warn(agent: Any) -> None:
"""Re-arm the context-overflow warning dedup (test doubles may lack the method).""" """Re-arm the context-overflow warning dedup (test doubles may lack the method)."""
# Compression is actually running (block cleared / was never blocked) — reset the blocked-overflow
# warning dedup so a future blocked-over-threshold turn can warn again. Mirrors the turn-context
# preflight reset (silent-overflow fix #62625). getattr guard: test doubles built via object.__new__
# lack the method (gateway test-double pitfall) — treat absence as no-op.
# Compression is actually running (block cleared / was never blocked) — reset the blocked-overflow
# warning dedup so a future blocked-over-threshold turn can warn again (silent-overflow fix #62625).
# getattr guard: test doubles built via object.__new__ lack the method (gateway test-double pitfall) —
# treat absence as no-op.
_clear_warn = getattr(agent, "_clear_context_overflow_warn", None) _clear_warn = getattr(agent, "_clear_context_overflow_warn", None)
if callable(_clear_warn): if callable(_clear_warn):
_clear_warn() _clear_warn()
@@ -94,6 +102,10 @@ def _apply_grown_window(agent: Any, compressor: Any, grown: int) -> None:
def _refund_api_call(agent: Any, api_call_count: int) -> int: def _refund_api_call(agent: Any, api_call_count: int) -> int:
"""A pass that never reached the provider refunds the call count and budget.""" """A pass that never reached the provider refunds the call count and budget."""
# Host progress-aware timeout (#98722, salvaged from #98741): this preflight iteration never reached the
# provider. Refund its provisional call/budget exactly like a successful pre-API compaction, then stop
# before the unchanged oversized request reaches the provider — its overflow error would only invoke
# compression again on the same transcript with the wait budget already spent.
api_call_count -= 1 api_call_count -= 1
agent._api_call_count = api_call_count agent._api_call_count = api_call_count
agent.iteration_budget.refund() agent.iteration_budget.refund()
@@ -200,6 +212,7 @@ def _codex_native_auto_compaction(agent: Any) -> bool:
"""Codex app-server threads are compacted by the codex agent itself; Hermes only """Codex app-server threads are compacted by the codex agent itself; Hermes only
initiates compaction in "hermes" mode.""" initiates compaction in "hermes" mode."""
return ( return (
# See #36801.
getattr(agent, "api_mode", None) == "codex_app_server" getattr(agent, "api_mode", None) == "codex_app_server"
and str( and str(
getattr(agent, "codex_app_server_auto_compaction", "native") or "native" getattr(agent, "codex_app_server_auto_compaction", "native") or "native"
@@ -367,6 +380,10 @@ def _run_preflight_passes(
# Lock-skip: another path holds the lock, so this is a DEFER, not proof of # Lock-skip: another path holds the lock, so this is a DEFER, not proof of
# incompressibility — don't arm the blocker; stop passes this turn. # incompressibility — don't arm the blocker; stop passes this turn.
logger.info( logger.info(
# That is a temporary DEFER, not proof the transcript cannot compress — do NOT arm the
# insufficient-progress blocker (the loop's error handlers must keep their provider-proven
# retry budget) and stop preflight passes for this turn; the lock winner is shrinking the
# same session concurrently. See #69870.
"Preflight compression deferred: compression lock " "Preflight compression deferred: compression lock "
"held by another path (session %s)", "held by another path (session %s)",
agent.session_id or "none", agent.session_id or "none",
+5
View File
@@ -31,6 +31,8 @@ class TurnFacadeMixin:
"""Forwarder — see ``agent.conversation_loop.run_conversation``.""" """Forwarder — see ``agent.conversation_loop.run_conversation``."""
# A review shares this session_id for cache parity: fence review startup or interrupt # A review shares this session_id for cache parity: fence review startup or interrupt
# an admitted request and await its exit before opening live-turn instrumentation. # an admitted request and await its exit before opening live-turn instrumentation.
# Foreground priority is retained if the review does not acknowledge within the bounded deadline
# (#84423).
from agent.background_review import cancel_background_review_for_live_turn from agent.background_review import cancel_background_review_for_live_turn
cancel_background_review_for_live_turn(self) cancel_background_review_for_live_turn(self)
@@ -106,6 +108,9 @@ class TurnFacadeMixin:
# affinity scope falls back to it; accounting handles route aux usage to the session. # affinity scope falls back to it; accounting handles route aux usage to the session.
token = set_conversation_context(self._conversation_root_id()) token = set_conversation_context(self._conversation_root_id())
affinity_token = set_affinity_scope(declared_conversation_scope_safe(self)) affinity_token = set_affinity_scope(declared_conversation_scope_safe(self))
# Publish the session accounting handles the same way so auxiliary calls record their token
# usage into session_model_usage (task dimension) — the fix for aux spend being invisible in
# analytics (issue #23270).
acct_token = set_accounting_context( acct_token = set_accounting_context(
getattr(self, "_session_db", None), getattr(self, "session_id", None) getattr(self, "_session_db", None), getattr(self, "session_id", None)
) )
+3
View File
@@ -220,6 +220,9 @@ def _durable_session_exists(db, session_id: str) -> bool:
# A locked / non-WAL read is not proof the row is absent; treating probe failure as "fresh" # A locked / non-WAL read is not proof the row is absent; treating probe failure as "fresh"
# ran fail-open at the exact contention point. Acquire, or fail closed. # ran fail-open at the exact contention point. Acquire, or fail closed.
logger.warning( logger.warning(
# Acquire (or fail closed if acquire itself cannot) rather than start load/run/flush
# unsynchronized. get_session returns None — it does not raise — when the row is missing. See
# #84234.
"Could not check durable session before turn lease; " "Could not check durable session before turn lease; "
"will acquire rather than run without serialization", "will acquire rather than run without serialization",
exc_info=True, exc_info=True,
+10
View File
@@ -103,6 +103,16 @@ def finish_text_response(
agent._emit_pending_fallback_notice() agent._emit_pending_fallback_notice()
agent._clear_status_buffer() agent._clear_status_buffer()
# Defensive: repair malformed role-alternation before API call. Catches cases where the history got
# wedged into a ``tool → user`` or ``user → user`` tail (e.g. after empty- response scaffolding was
# stripped and a new user message landed after an orphan tool result). Most providers return empty
# content on malformed sequences, which would otherwise retrigger the empty-retry loop indefinitely.
# repair_message_sequence_with_cursor also recomputes the SessionDB flush cursor (_last_flushed_db_idx)
# when repair compacts the list, so the turn-end flush doesn't skip the assistant/tool chain (#44837).
# One-time repeated-heal escalation notice (#96870): if the sanitizer above just crossed the per-session
# heal threshold, deliver the queued notice through the status/warning callback — the normal out-of-band
# delivery channel (gateway status message / CLI print). NEVER appended to messages/api_messages:
# conversation context and the cached prompt prefix stay byte-identical.
from agent.agent_runtime_helpers import ( from agent.agent_runtime_helpers import (
intent_ack_continuation_mode, trailing_continue_intent intent_ack_continuation_mode, trailing_continue_intent
) )
+42 -1
View File
@@ -46,7 +46,12 @@ def _record_kanban_budget_exhausted(
Routed via ``_record_task_failure`` (not ``kanban_block``) so it counts toward the Routed via ``_record_task_failure`` (not ``kanban_block``) so it counts toward the
consecutive-failure circuit breaker. Idempotent via the ``_end_run`` CAS consecutive-failure circuit breaker. Idempotent via the ``_end_run`` CAS
(``WHERE ended_at IS NULL``), so safe from multiple exit paths.""" (``WHERE ended_at IS NULL``), so safe from multiple exit paths.
This is a bounded fallback (#87096): the CAS invariant in ``_end_run`` (``WHERE ended_at IS NULL``)
guarantees idempotence — if another path already closed the run this is a no-op — so it is safe to call
from multiple exit paths.
"""
try: try:
from hermes_cli import kanban_db as _kb from hermes_cli import kanban_db as _kb
_conn = _kb.connect() _conn = _kb.connect()
@@ -149,6 +154,17 @@ def _resolve_budget_fallback(
# A kanban worker must record a terminal outcome whether or not a fallback path # A kanban worker must record a terminal outcome whether or not a fallback path
# was eligible, so the dispatcher learns the worker could not complete. # was eligible, so the dispatcher learns the worker could not complete.
_kanban_task = os.environ.get("HERMES_KANBAN_TASK") if budget_exhausted else None _kanban_task = os.environ.get("HERMES_KANBAN_TASK") if budget_exhausted else None
# If running as a kanban worker, signal the dispatcher that the worker could not complete (rather than
# treating it as a protocol violation). This applies whether the user-facing fallback came from the
# summary call or an explicitly pending continuation; both exhausted the task budget and must advance
# the failure circuit. We route through ``_record_task_failure(outcome="timed_out")`` rather than
# ``kanban_block`` so this counts toward the dispatcher's consecutive-failure circuit breaker (#29747
# gap 2).
# Bounded fallback (#87096): budget was exhausted but none of the normal fallback paths were eligible
# (interrupted / failed / anomalous exit_reason). If running as a kanban worker we must still record a
# terminal outcome so the task does not remain in an ambiguous lifecycle state. The worker's run is
# closed via ``_record_task_failure`` (compare-and-swap receipt path) which is a no-op if another path
# closed it — the CAS invariant in ``_end_run`` (``WHERE ended_at IS NULL``) guarantees idempotence.
if _kanban_task: if _kanban_task:
_record_kanban_budget_exhausted(_kanban_task, api_call_count, agent.max_iterations, logger) _record_kanban_budget_exhausted(_kanban_task, api_call_count, agent.max_iterations, logger)
return final_response, _turn_exit_reason, preserved_verification_fallback return final_response, _turn_exit_reason, preserved_verification_fallback
@@ -202,6 +218,16 @@ def _close_transcript_tail(agent, messages, final_response, interrupted, failed)
# row; enforce "delivered final_response ⇒ assistant row" here. Compare content, # row; enforce "delivered final_response ⇒ assistant row" here. Compare content,
# not role, so a matching verification candidate isn't dup'd. # not role, so a matching verification candidate isn't dup'd.
if final_response and not interrupted: if final_response and not interrupted:
# Some recovery/fallback paths return a real final_response without adding a closing assistant
# message to the transcript (e.g. the partial-stream and prior-turn-content recovery ``break`` sites
# in ``conversation_loop``). If persisted as-is, the durable session can end at a tool/user message
# even though the caller — and the gateway platform — already saw a completed assistant response.
# The next turn then replays a user-only backlog and the model re-answers every "unanswered"
# message. Close the durable turn at the source, at the single chokepoint every recovery ``break``
# flows through, so the invariant "delivered final_response ⇒ assistant row in transcript" holds
# regardless of which path produced it. (#43849 / #44100) Compare content (not just role) so a
# verification candidate that matches the final response is not duplicated at budget exhaustion.
# (#65919 §7)
_tail = messages[-1] if messages else None _tail = messages[-1] if messages else None
if not isinstance(_tail, dict) or _tail.get("role") != "assistant": if not isinstance(_tail, dict) or _tail.get("role") != "assistant":
append_message(messages, {"role": "assistant", "content": final_response}) append_message(messages, {"role": "assistant", "content": final_response})
@@ -219,6 +245,8 @@ def _close_transcript_tail(agent, messages, final_response, interrupted, failed)
# Request is complete, so replace API-local voice/model/skill guidance with the # Request is complete, so replace API-local voice/model/skill guidance with the
# clean user input before the durable snapshot (earlier flushes still needed them). # clean user input before the durable snapshot (earlier flushes still needed them).
# Earlier turn-start flushes use the DB-only override because their messages are still needed for the
# API request; this finalizer runs after that request is complete (#48677 / #63766).
_apply_override = getattr(agent, "_apply_persist_user_message_override", None) _apply_override = getattr(agent, "_apply_persist_user_message_override", None)
if callable(_apply_override): if callable(_apply_override):
_apply_override(messages) _apply_override(messages)
@@ -302,6 +330,13 @@ def _append_file_mutation_footer(agent, final_response, logger):
"""Append the verifier advisory when ``write_file`` / ``patch`` calls failed and were """Append the verifier advisory when ``write_file`` / ``patch`` calls failed and were
never superseded by a successful write to the same path (surfaces over-claiming).""" never superseded by a successful write to the same path (surfaces over-claiming)."""
try: try:
# File-mutation verifier footer. This catches the specific case — reported by Ben Eng
# (#15524-adjacent) — where a model issues a batch of parallel patches, half of them fail with
# "Could not find old_string", and the model summarises the turn claiming every file was edited. The
# user then has to manually run ``git status`` to catch the lie. With this footer the truth is
# surfaced on every turn, so over-claiming is structurally impossible past the model. Gate: only
# applied when a real text response exists for this turn and the user didn't interrupt.
# Empty/interrupted turns already have other surface text that shouldn't be augmented.
_failed = getattr(agent, "_turn_failed_file_mutations", None) or {} _failed = getattr(agent, "_turn_failed_file_mutations", None) or {}
if _failed and agent._file_mutation_verifier_enabled(): if _failed and agent._file_mutation_verifier_enabled():
footer = agent._format_file_mutation_failure_footer(_failed) footer = agent._format_file_mutation_failure_footer(_failed)
@@ -473,6 +508,12 @@ def finalize_turn(
# Surrogate chokepoint: RAW SDK text with a lone UTF-16 surrogate crashes downstream # Surrogate chokepoint: RAW SDK text with a lone UTF-16 surrogate crashes downstream
# consumers (stdout, Telegram ``utf16_len``, JSON); scrub once where it leaves the loop. # consumers (stdout, Telegram ``utf16_len``, JSON); scrub once where it leaves the loop.
# Class-level surrogate chokepoint (#80366, #55143, #55309, #19819): ``final_response`` is often the RAW
# SDK content (``assistant_message.content``), not the sanitized copy stored in history by
# ``build_assistant_message``. Any lone UTF-16 surrogate (U+D800–U+DFFF) in it crashes downstream
# consumers — oneshot stdout writes, Telegram's ``utf16_len`` length check, Signal formatting, JSON
# envelope encodes — on every provider (Ollama, NVIDIA NIM, …). Scrub once here, where model text leaves
# the conversation loop, so every delivery surface receives valid Unicode.
if isinstance(final_response, str): if isinstance(final_response, str):
final_response = _sanitize_surrogates(final_response) final_response = _sanitize_surrogates(final_response)
+10
View File
@@ -273,6 +273,11 @@ def begin_iteration(
# Grace call: budget exhausted but the model gets one more call. Consume the # Grace call: budget exhausted but the model gets one more call. Consume the
# flag so the loop exits after this iteration regardless of outcome. # flag so the loop exits after this iteration regardless of outcome.
if agent._budget_grace_call: if agent._budget_grace_call:
# Iteration budget: the LLM is only notified when it actually exhausts the iteration budget
# (api_call_count >= max_iterations). At that point we inject ONE message, allow one final API call,
# and if the model doesn't produce a text response, force a user-message asking it to summarise. No
# intermediate pressure warnings — they caused models to "give up" prematurely on complex tasks
# (#7915).
agent._budget_grace_call = False agent._budget_grace_call = False
elif not agent.iteration_budget.consume(): elif not agent.iteration_budget.consume():
_turn_exit_reason = "budget_exhausted" _turn_exit_reason = "budget_exhausted"
@@ -340,6 +345,11 @@ def apply_retry_restarts(
retry_count += 1 retry_count += 1
_retry.restart_with_compressed_messages = False _retry.restart_with_compressed_messages = False
if _should_skip_model_call_for_reference_handoff( if _should_skip_model_call_for_reference_handoff(
# Compression rebuilt the list (tail messages are fresh compaction copies), so the
# pre-compression index of this turn's user message is stale. Re-anchor both index trackers: the
# api_content stamp below, the loop's injection site, and the flush's persist-override row
# (#48677) must all target the surviving dict, not a stale position. Exact-content match first
# so a todo-snapshot user message appended after the tail can't steal the anchor.
messages, user_message messages, user_message
): ):
logger.info( logger.info(
+12 -1
View File
@@ -127,6 +127,11 @@ class TurnLivenessWatchdog:
return None return None
# Observational only: the commit below can still veto the abort if progress # Observational only: the commit below can still veto the abort if progress
# resumed; the definitive settlement is _surface_committed_abort. # resumed; the definitive settlement is _surface_committed_abort.
# Pre-commit surface is OBSERVATIONAL only: it reports the stall and that a recovery attempt is
# beginning. It must not claim the abort or the lease withdrawal has committed — the next operation
# can still veto the outcome. The definitive aborted/lease-stopped settlement is published by
# _surface_committed_abort only after _commit_abort succeeds and the turn is deactivated (#95663
# review).
self._surface_stall(snapshot) self._surface_stall(snapshot)
message = f"Turn made no progress for {int(snapshot.idle_seconds)}s; aborting to release the session." message = f"Turn made no progress for {int(snapshot.idle_seconds)}s; aborting to release the session."
if not self._commit_abort(snapshot, message): if not self._commit_abort(snapshot, message):
@@ -178,7 +183,13 @@ class TurnLivenessWatchdog:
) )
def _surface_committed_abort(self, snapshot: ActivitySnapshot) -> None: def _surface_committed_abort(self, snapshot: ActivitySnapshot) -> None:
"""Publish the definitive settlement once the abort has authority.""" """Publish the definitive settlement once the abort has authority.
Runs only once ``_commit_abort`` succeeded (the interrupt was published) and the turn lease was
deactivated: the turn IS force-aborted and lease renewal IS stopped, so stating that is now true.
Separated from the pre-commit surface so a declined abort never reports a committed outcome (#95663
review).
"""
logger.error( logger.error(
"Turn liveness watchdog aborted turn for session %s: " "Turn liveness watchdog aborted turn for session %s: "
"no progress for %.1fs; turn interrupted and lease renewal " "no progress for %.1fs; turn interrupted and lease renewal "
+11
View File
@@ -56,6 +56,17 @@ def handle_outer_loop_error(
_outer_error_count += 1 _outer_error_count += 1
# Interpreter shutdown makes every executor op raise: break. # Interpreter shutdown makes every executor op raise: break.
# Phase-aware error classification. The huge outer try/except spans both the actual API request and all
# local post-processing of the returned assistant message. Deterministic local bugs (e.g. passing a
# multimodal content list into a regex helper after a vision turn or context compaction) should not be
# retried: they will fail identically on every iteration and only burn the iteration budget. We classify
# an error as local by inspecting the traceback: if the exception propagated through any of the known
# local post-processing helpers and never entered the interruptible API-call helpers, it is almost
# certainly a local processing bug. (#66267) Interpreter shutdown: if the process is tearing down, every
# executor-backed operation (API call, tool dispatch, memory sync) raises ``RuntimeError: cannot
# schedule new futures after interpreter shutdown``. Retrying is pointless — the executor is gone for
# good — and each retry just spams another traceback. Break immediately so the turn exits cleanly.
# (#93217)
if sys.is_finalizing() or _is_interpreter_shutdown_error(e): if sys.is_finalizing() or _is_interpreter_shutdown_error(e):
error_msg = f"Interpreter is shutting down — cannot continue (API call #{api_call_count}): {e}" error_msg = f"Interpreter is shutting down — cannot continue (API call #{api_call_count}): {e}"
try: try:
+21
View File
@@ -109,6 +109,9 @@ class _Recovery(OverflowVerdict):
"failed": True, "failed": True,
} }
if compression_exhausted: if compression_exhausted:
# Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The
# bloated transcript remains intact while future input can move to a clean session instead of
# replaying the summarize-timeout loop.
result["compression_exhausted"] = True result["compression_exhausted"] = True
result.update(extra) result.update(extra)
return self.done("return", result) return self.done("return", result)
@@ -190,6 +193,14 @@ class _Recovery(OverflowVerdict):
if deferred is not None: if deferred is not None:
return deferred, False, original_tokens return deferred, False, original_tokens
messages = self.messages messages = self.messages
# Re-measure after compression. Same-message-count compression (tool-result pruning, in-place
# summarization) can materially reduce request size without reducing the message array (#39550), and
# — the image-dominated case — compaction's historical-media aging (#97160) can free megabytes of
# base64 that the token estimate never counted. Bytes are the yardstick for a 413; tokens are kept
# only for status display.
# Re-estimate tokens after compression. Same-message-count compression (tool-result pruning,
# in-place summarization) can materially reduce request size without reducing the message array.
# (#39550)
new_tokens = estimate_messages_tokens_rough(messages) new_tokens = estimate_messages_tokens_rough(messages)
shrank_tokens = new_tokens > 0 and new_tokens < original_tokens * 0.95 shrank_tokens = new_tokens > 0 and new_tokens < original_tokens * 0.95
if len(messages) < original_len: if len(messages) < original_len:
@@ -223,6 +234,13 @@ def _recover_payload_too_large(st: _Recovery, _retry: TurnRetryState) -> Overflo
messages = st.messages messages = st.messages
original_len = len(messages) original_len = len(messages)
# A 413 is a BYTE-size error, so this branch scores progress in BYTES of the serialized messages payload
# — exact and free — never the token estimate. The estimator prices every image at a flat per-image
# token cost (see estimate_messages_tokens_rough) so screenshots don't trigger premature compaction;
# that deliberate byte-blindness means compaction can free megabytes of base64 (real case: two vision
# results = 96.6% of the request body but ~3.7% of the estimate) while the token delta stays under any
# threshold. Token-scored progress here burned all attempts on "no progress" and wedged the session
# permanently. (#88960 / #47339)
original_bytes = serialized_messages_bytes(messages) original_bytes = serialized_messages_bytes(messages)
deferred = st.compress(st.request_tokens()) deferred = st.compress(st.request_tokens())
if deferred is not None: if deferred is not None:
@@ -439,6 +457,9 @@ def recover_from_overflow(
# failover or generic retries. The classifier also covers 400/disconnect + # failover or generic retries. The classifier also covers 400/disconnect +
# large-session heuristics. # large-session heuristics.
st.is_context_length_error = ( st.is_context_length_error = (
# Check for context-length errors BEFORE generic 4xx handler. The classifier detects context
# overflow from: explicit error messages, generic 400 + large session heuristic (#1630), and server
# disconnect + large session pattern (#2153).
classified.reason == FailoverReason.context_overflow classified.reason == FailoverReason.context_overflow
or wrapped_output_cap_budget is not None or wrapped_output_cap_budget is not None
) )
+38
View File
@@ -262,9 +262,17 @@ def compress_after_tool_results(
) )
_compressor = agent.context_compressor _compressor = agent.context_compressor
# Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens
# is the actual context size the provider reported plus the assistant turn — a tight lower bound for the
# next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves
# ample headroom; if tool results push past it, the next API call will report the real total and trigger
# compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage
# data), fall back to rough estimate to avoid missing compression. Without this, a session can grow
# unbounded after disconnects because should_compress(0) never fires. (#2153)
if _compressor.last_prompt_tokens > 0: if _compressor.last_prompt_tokens > 0:
# Only prompt_tokens: thinking models inflate completion_tokens with # Only prompt_tokens: thinking models inflate completion_tokens with
# reasoning that uses no context → premature compression. # reasoning that uses no context → premature compression.
# Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026)
_real_tokens = _compressor.last_prompt_tokens _real_tokens = _compressor.last_prompt_tokens
elif _compressor.last_prompt_tokens == -1: elif _compressor.last_prompt_tokens == -1:
# Compression just ran, no API prompt count yet: don't treat a rough # Compression just ran, no API prompt count yet: don't treat a rough
@@ -274,6 +282,12 @@ def compress_after_tool_results(
# Include tool schemas (20-30K tokens the messages-only estimate misses) and # Include tool schemas (20-30K tokens the messages-only estimate misses) and
# stay route-aware: on a compacted native-Codex session the generic # stay route-aware: on a compacted native-Codex session the generic
# durable-history figure would false-trigger. # durable-history figure would false-trigger.
# Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate
# misses, which can skip compression past the configured threshold (#14695). Route-aware
# (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure
# overstates the wire and would false-trigger compression here exactly like the pre-API guard — this
# fallback runs precisely when no provider usage is available (post-disconnect / gateway restart),
# the unanchored case from #97602's repro.
_real_tokens = _midturn_request_pressure_tokens( _real_tokens = _midturn_request_pressure_tokens(
agent, messages, active_system_prompt or "", agent, messages, active_system_prompt or "",
estimate_request_tokens_rough(messages, tools=agent.tools or None), estimate_request_tokens_rough(messages, tools=agent.tools or None),
@@ -298,6 +312,30 @@ def compress_after_tool_results(
if messages is _post_tool_input and compression_skipped_due_to_lock(agent): if messages is _post_tool_input and compression_skipped_due_to_lock(agent):
# Lock-skip no-op is a temporary defer, not evidence about compressibility: # Lock-skip no-op is a temporary defer, not evidence about compressibility:
# refund so a lock-loser loop doesn't burn the budget toward exhausted. # refund so a lock-loser loop doesn't burn the budget toward exhausted.
# #69870 lock-skip / #97488 transient-block: this pass no-oped for a TEMPORARY reason (another
# path holds the compression lock, or a timed cooldown/backoff guard is active). That is a
# temporary DEFER, not evidence about compressibility — refund the attempt (it must not burn the
# shared overflow-recovery budget toward compression_exhausted → gateway auto-reset,
# #9893/#35809) and leave the insufficient-progress blocker unarmed. Proceed with the current
# request: if it truly does not fit, the provider's 413/overflow handler returns the soft
# compression_deferred result with that stronger signal.
# #69870 lock-skip: the provider proved the request does not fit, but this compression pass
# no-oped only because another path holds the session's compression lock. Temporary defer, not
# exhaustion — refund the attempt and end the turn softly so the gateway does NOT auto-reset the
# session (#9893/#35809).
# #97488 transient-block: compression no-oped because a timed guard (host-timeout cooldown /
# structural backoff) is active — a temporary defer, not evidence of incompressibility. Never
# classify it as compression_exhausted (gateway auto-reset).
# bypass_cooldown=True, # #100661 provider-proven overflow
# #97488: timed transient guard — defer, never exhaustion (gateway auto-reset).
# #69870 lock-skip: the provider proved the request does not fit, but this compression pass
# no-oped only because another path holds the session's compression lock. Temporary defer, not
# exhaustion — refund the attempt and end the turn softly so the gateway does NOT auto-reset the
# session (#9893/#35809).
# #97488 transient-block: a timed guard (host-timeout cooldown / structural backoff) no-oped
# this pass — defer softly, never compression_exhausted (which would auto-reset the session).
# #69870 lock-skip: this pass no-oped because another path holds the session's compression lock
# — a temporary defer, not evidence about compressibility.
compression_attempts -= 1 compression_attempts -= 1
else: else:
conversation_history = conversation_history_after_compression( conversation_history = conversation_history_after_compression(
+20
View File
@@ -145,6 +145,9 @@ def _recover_unicode_encode_error(
# Non-ASCII in the API key makes httpx fail encoding the Authorization header — the # Non-ASCII in the API key makes httpx fail encoding the Authorization header — the
# usual persistent cause after message/tool sanitization. Entra ID bearer providers # usual persistent cause after message/tool sanitization. Entra ID bearer providers
# are callables minting ASCII JWTs; skip them (``_strip_non_ascii`` would crash). # are callables minting ASCII JWTs; skip them (``_strip_non_ascii`` would crash).
# Sanitize the API key — non-ASCII characters in credentials (e.g. ʋ instead of v from a bad copy-paste)
# cause httpx to fail when encoding the Authorization header as ASCII. This is the most common cause of
# persistent UnicodeEncodeError that survives message/tool sanitization (#6843).
_credential_sanitized = False _credential_sanitized = False
_raw_key = getattr(agent, "api_key", None) or "" _raw_key = getattr(agent, "api_key", None) or ""
if _raw_key and isinstance(_raw_key, str): if _raw_key and isinstance(_raw_key, str):
@@ -543,6 +546,8 @@ def recover_after_classification(
# Anthropic OAuth subscription rejected the 1M-context beta: disable it for this # Anthropic OAuth subscription rejected the 1M-context beta: disable it for this
# session, rebuild the client, retry once. Reactive so capable subscriptions keep 1M. # session, rebuild the client, retry once. Reactive so capable subscriptions keep 1M.
if ( if (
# See PR #17680 for the original report (we chose reactive recovery over the proposed unconditional
# omit so capable subscriptions don't silently lose the capability).
classified.reason == FailoverReason.oauth_long_context_beta_forbidden classified.reason == FailoverReason.oauth_long_context_beta_forbidden
and agent.api_mode == "anthropic_messages" and agent.api_mode == "anthropic_messages"
and agent._is_anthropic_oauth and agent._is_anthropic_oauth
@@ -698,6 +703,8 @@ def nonretryable_client_error_result(
logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error) logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error)
# Skip persistence on likely context-overflow (400 + large session): persisting the # Skip persistence on likely context-overflow (400 + large session): persisting the
# failed message grows the session and repeats the failure. # failed message grows the session and repeats the failure.
# Persisting the failed user message would make the session even larger, causing the same failure on the
# next attempt. (#1630)
if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80): if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80):
_vlines(agent, "⚠️ Skipping session persistence for large failed session to prevent growth loop.") _vlines(agent, "⚠️ Skipping session persistence for large failed session to prevent growth loop.")
else: else:
@@ -981,6 +988,9 @@ def compute_error_backoff(
_ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After")
if _ra_raw: if _ra_raw:
try: try:
# Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap
# caused us to retry before the actual reset window and re-trip the limit. 600s covers all
# realistic provider reset windows while still rejecting pathological values. (#26293)
_retry_after = min(float(_ra_raw), 600) _retry_after = min(float(_ra_raw), 600)
except (TypeError, ValueError): except (TypeError, ValueError):
pass pass
@@ -1307,6 +1317,9 @@ def route_classified_error(
# Overhead-aware request size so recovery arms on the true request # Overhead-aware request size so recovery arms on the true request
# (msgs + tools + system), not the tool-blind message count. # (msgs + tools + system), not the tool-blind message count.
messages, active_system_prompt = agent._compress_context( messages, active_system_prompt = agent._compress_context(
# Route the overhead-aware _real_tokens (computed above) into compression, not the bare
# last_prompt_tokens — which is 0 in the no-usage fallback, hiding the true request size
# from the engine's overflow guard (upstream PR #77169 review).
messages, system_message, messages, system_message,
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
task_id=effective_task_id, task_id=effective_task_id,
@@ -1331,6 +1344,12 @@ def route_classified_error(
is_rate_limited = classified.reason in _RATE_LIMIT_REASONS is_rate_limited = classified.reason in _RATE_LIMIT_REASONS
# Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only the max_tokens # Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only the max_tokens
# clamp fixes it. Parsed once; gates the eager-fallback exemption and overflow entry. # clamp fixes it. Parsed once; gates the eager-fallback exemption and overflow entry.
# Relay-wrapped output-cap errors: some gateways wrap an upstream "[400]: max_tokens (...) exceeds
# model's maximum output tokens (...)" as HTTP 429, which classifies as rate_limit. The failure is a
# deterministic request-shape problem — falling back to another provider (or burning generic retries)
# can't fix it, but the output-cap clamp below can, in one retry (#72281). Parse once here; the result
# gates both the eager-fallback exemption and the widened is_context_length_error entry, and is reused
# as available_out inside the handler.
_wrapped_output_cap_budget = ( _wrapped_output_cap_budget = (
parse_available_output_tokens_from_error(error_msg) parse_available_output_tokens_from_error(error_msg)
if classified.reason == FailoverReason.rate_limit else None if classified.reason == FailoverReason.rate_limit else None
@@ -1348,6 +1367,7 @@ def route_classified_error(
if _should_fallback and agent._fallback_index < len(agent._fallback_chain): if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
# No eager fallback while credential pool rotation may recover. Exception: an # No eager fallback while credential pool rotation may recover. Exception: an
# upstream-aggregator 429 — the pool can't help, always fall back. # upstream-aggregator 429 — the pool can't help, always fall back.
# Fixes #11314.
_is_upstream = classified.reason == FailoverReason.upstream_rate_limit _is_upstream = classified.reason == FailoverReason.upstream_rate_limit
pool_may_recover = ( pool_may_recover = (
False if _is_upstream else _ra()._pool_may_recover_from_rate_limit(agent._credential_pool) False if _is_upstream else _ra()._pool_may_recover_from_rate_limit(agent._credential_pool)
+4
View File
@@ -234,6 +234,10 @@ def assemble_api_request(
approx_tokens = estimate_messages_tokens_rough(api_messages, charge_stale_thinking=False) approx_tokens = estimate_messages_tokens_rough(api_messages, charge_stale_thinking=False)
# Route-aware: native Responses compaction prunes the wire payload, so the raw # Route-aware: native Responses compaction prunes the wire payload, so the raw
# history figure overstates it and fires needless local compression. # history figure overstates it and fires needless local compression.
# Route-aware pressure: when the upcoming request is eligible for native Responses compaction the
# transport will checkpoint-prune the payload before sending — the generic durable-history figure
# overstates the wire by orders of magnitude on a compacted session and fires a 600s local compression
# the main request never needed (#96995, mirroring the turn-prologue preflight #96644/#96155).
request_pressure_tokens = _midturn_request_pressure_tokens( request_pressure_tokens = _midturn_request_pressure_tokens(
agent, api_messages, effective_system or "", approx_tokens agent, api_messages, effective_system or "", approx_tokens
) )
+5
View File
@@ -205,6 +205,11 @@ def run_tool_round(
agent._session_messages = messages agent._session_messages = messages
# Touch activity so slow post-tool work plus a slow follow-up API call can't exceed # Touch activity so slow post-tool work plus a slow follow-up API call can't exceed
# the gateway inactivity timeout (HERMES_AGENT_TIMEOUT). # the gateway inactivity timeout (HERMES_AGENT_TIMEOUT).
# Touch activity before continuing so the gateway's inactivity monitor never sees a stale timestamp
# between tool completion and the start of the next API call. Without this, a tool-call result (which
# takes ~0s to process) followed by slow post-tool processing (compression, persist) and a slow
# follow-up API call can exceed the gateway inactivity timeout (HERMES_AGENT_TIMEOUT, default 1800s) and
# the gateway kills the session before the next activity touch fires (#69559, #69131).
agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}") agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}")
return _verdict("continue") return _verdict("continue")
+8
View File
@@ -436,6 +436,14 @@ def continue_codex_incomplete(
agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({n}/3)") agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({n}/3)")
# Spinner/heartbeat notice: these retries can take minutes and otherwise look # Spinner/heartbeat notice: these retries can take minutes and otherwise look
# like infinite thinking. # like infinite thinking.
# #70773: same FD-recycle corruption vector as #67142. The shared OpenAI client's connection pool
# must NOT be closed from this watchdog/poll thread — worker threads from previous stale-killed
# attempts may still be unwinding their SSL BIOs. The request-local client is already closed above
# via _close_request_client_once. The shared client will be replaced lazily by
# _ensure_primary_openai_client on the next request.
# Surface the continuation on the live spinner/status line (CLI/TUI/Desktop) and gateway heartbeat:
# each of these retries can spend minutes waiting on the provider, and without a distinct notice the
# user only sees a generic thinking spinner ("infinite thinking", #64434).
agent._emit_wait_notice( agent._emit_wait_notice(
f"↻ model returned reasoning with no final answer — asking it to continue ({n}/3)" f"↻ model returned reasoning with no final answer — asking it to continue ({n}/3)"
) )
+9 -1
View File
@@ -17,6 +17,8 @@ _ONE_MILLION = Decimal("1000000")
_NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1" _NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1"
# Below $0.01, render at 4 dp so cheap-model costs never display as $0.00. # Below $0.01, render at 4 dp so cheap-model costs never display as $0.00.
# Sub-cent cost threshold: below $0.01, render at 4 decimal places so the display is non-zero (e.g. $0.0046
# instead of $0.00). See #79220.
_SUBCENT_THRESHOLD = Decimal("0.01") _SUBCENT_THRESHOLD = Decimal("0.01")
# Attached to every CostResult with status="included" so consumers can # Attached to every CostResult with status="included" so consumers can
@@ -27,13 +29,19 @@ _INCLUDED_NOTE = "subscription-included; no provider invoice for usage"
def format_cost_label(amount: Decimal) -> str: def format_cost_label(amount: Decimal) -> str:
"""Cost display label: zero → "$0.00"; sub-cent → "~$0.0046" (4 dp, or """Cost display label: zero → "$0.00"; sub-cent → "~$0.0046" (4 dp, or
"~$<0.0001" when it rounds to 0.0000 so the label never reads as zero); "~$<0.0001" when it rounds to 0.0000 so the label never reads as zero);
else "~$1.23". Shared by per-response labels and insights cost buckets.""" else "~$1.23". Shared by per-response labels and insights cost buckets.
This fixes #79220 where sub-cent per-turn costs on cheap models (DeepSeek, etc.) rendered as "$0.00"
despite amount_usd carrying full Decimal precision.
"""
if amount == _ZERO: if amount == _ZERO:
return "$0.00" return "$0.00"
if amount < _SUBCENT_THRESHOLD: if amount < _SUBCENT_THRESHOLD:
label = f"~${amount:.4f}" label = f"~${amount:.4f}"
# Compare the rendered label: a naive `< 0.00005` threshold misses # Compare the rendered label: a naive `< 0.00005` threshold misses
# the exact boundary under ROUND_HALF_EVEN. # the exact boundary under ROUND_HALF_EVEN.
# A positive amount that rounds to 0.0000 at 4 dp would render "~$0.0000" — a zero-looking label,
# the exact #79220 dishonesty.
return label if label != "~$0.0000" else "~$<0.0001" return label if label != "~$0.0000" else "~$<0.0001"
return f"~${amount:.2f}" return f"~${amount:.2f}"
+4
View File
@@ -135,6 +135,10 @@ def _transaction() -> Iterator[sqlite3.Connection]:
``sqlite3.Connection`` as a context manager only commits/rolls back; without ``sqlite3.Connection`` as a context manager only commits/rolls back; without
the close, each call leaks a connection (and WAL/SHM fds) until GC runs. the close, each call leaks a connection (and WAL/SHM fds) until GC runs.
Using ``with _connect()`` alone therefore leaks a connection — and its WAL/SHM file descriptors — on
every call, deferring the close to the garbage collector, which over a long-running process can exhaust
``RLIMIT_NOFILE`` (the cron-ledger sibling of this bug was #69567 / PR #69594).
""" """
conn = _connect() conn = _connect()
try: try:
+8 -1
View File
@@ -294,7 +294,14 @@ class VisionMessagePrepMixin:
def _anthropic_preserve_dots(self) -> bool: def _anthropic_preserve_dots(self) -> bool:
"""True for anthropic-compatible endpoints that keep dots in model names (DashScope, MiniMax, Xiaomi """True for anthropic-compatible endpoints that keep dots in model names (DashScope, MiniMax, Xiaomi
MiMo, OpenCode Go/Zen, ZAI/Zhipu; Bedrock's dotted inference-profile IDs 400 on the hyphenated form).""" MiMo, OpenCode Go/Zen, ZAI/Zhipu; Bedrock's dotted inference-profile IDs 400 on the hyphenated form).
Alibaba/DashScope keeps dots (e.g. qwen3.5-plus). OpenCode Go/Zen keeps dots for non-Claude models
(e.g. minimax-m2.5-free). ``global.anthropic.claude-opus-4-7``,
``us.anthropic.claude-sonnet-4-5-20250929-v1:0``) and rejects the hyphenated form with ``HTTP 400
The provided model identifier is invalid``. Regression for #11976; mirrors the opencode-go fix for
#5211
"""
if (getattr(self, "provider", "") or "").lower() in { if (getattr(self, "provider", "") or "").lower() in {
"alibaba", "minimax", "minimax-cn", "opencode-go", "opencode-zen", "zai", "bedrock", "xiaomi", "vertex", "alibaba", "minimax", "minimax-cn", "opencode-go", "opencode-zen", "zai", "bedrock", "xiaomi", "vertex",
}: }:
+5 -1
View File
@@ -24,7 +24,11 @@ from agent.provider_base import ProviderBase
def get_provider_env(name: str) -> str: def get_provider_env(name: str) -> str:
"""Config-aware env lookup (``os.environ`` first, then ``~/.hermes/.env``) so """Config-aware env lookup (``os.environ`` first, then ``~/.hermes/.env``) so
credentials set through the config layer are visible in gateway sessions / credentials set through the config layer are visible in gateway sessions /
delegate children / subprocess runs. Stripped value, or ``""`` when unset.""" delegate children / subprocess runs. Stripped value, or ``""`` when unset.
Falls back to a bare ``os.getenv`` when the config module is unavailable (stripped installs, early
import contexts). See #40190.
"""
try: try:
from hermes_cli.config import get_env_value from hermes_cli.config import get_env_value
+6
View File
@@ -160,6 +160,12 @@ def _disabled_web_plugin_for(configured: Optional[str] = None, *, capability: Op
backend because a disabled provider fails the availability gate and silently backend because a disabled provider fails the availability gate and silently
drops to the default. Bundled web plugins live under ``web/<vendor>`` with drops to the default. Bundled web plugins live under ``web/<vendor>`` with
the provider name differing only by hyphen/underscore, so both are normalized. the provider name differing only by hyphen/underscore, so both are normalized.
When a user sets ``web.extract_backend: firecrawl`` (or the search equivalent) but also lists
``web-firecrawl`` in ``plugins.disabled``, the provider never registers and the dispatcher would
otherwise emit a misleading "No web extract provider configured. Set web.extract_backend to ..." error —
even though the backend IS configured correctly. This helper detects that case so the dispatcher can
point the user at the actual cause (issue #40190 follow-up: pi314's disabled-plugin symptom).
""" """
def _norm(s: str) -> str: def _norm(s: str) -> str:
return s.strip().lower().replace("-", "_") return s.strip().lower().replace("-", "_")

Some files were not shown because too many files have changed in this diff Show More