review-fix(comments): restore lost #NNNN rationale comments across non-test source (mechanical sweep, condensed, code unchanged)
For each issue anchor present in BASE 63279301bc non-test .py and absent on HEAD, the BASE comment/docstring block was re-attached at the HEAD location of the code it explained (matched by the distinctive code line / enclosing def). Sentences already covered by an existing HEAD comment were deduped; the issue number always survives. Insert-only: no code lines changed.
This commit is contained in:
@@ -236,6 +236,8 @@ class SlashCommandsMixin:
|
||||
|
||||
original_count = len(state.history)
|
||||
# Include system prompt + tool schemas so the figure reflects real request pressure.
|
||||
# See #6217.
|
||||
# See #6217.
|
||||
_sys_prompt = getattr(agent, "_cached_system_prompt", "") or ""
|
||||
_tools = getattr(agent, "tools", None) or None
|
||||
approx_tokens = _estimate_tokens(state.history, agent, _sys_prompt, _tools)
|
||||
|
||||
@@ -103,6 +103,9 @@ def _decode_text_bytes(data: bytes, mime_type: str | None) -> str | None:
|
||||
return data.decode(encoding)
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
# Binary (ELF/Mach-O/PE), not a shell script: feeding its decoded bytes back into the guard tokenizes
|
||||
# machine code into bogus NUL-bearing paths and crashes the scanner (#77703). Mirror
|
||||
# lifecycle_guard._read_referenced_script and treat it as nothing to scan.
|
||||
return data.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
|
||||
@@ -185,6 +185,9 @@ def main(argv: list[str] | None = None) -> None:
|
||||
# MCP discovery from config.yaml runs in a background daemon thread so the ACP server is
|
||||
# responsive immediately (blocking here cost 2-5 s); per-session MCP servers registered via
|
||||
# asyncio.to_thread are unaffected. Metadata-only hosts can opt out of the global startup.
|
||||
# Previously this blocked asyncio.run() for 2-5 s. (ACP also registers per-session MCP servers
|
||||
# dynamically via asyncio.to_thread inside the event loop; that path is unaffected.) Moved from
|
||||
# model_tools.py module scope to avoid freezing the gateway's loop on lazy import (#16856).
|
||||
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
|
||||
try:
|
||||
from hermes_cli.mcp_startup import start_background_mcp_discovery
|
||||
|
||||
@@ -37,6 +37,7 @@ def _build_permission_options(
|
||||
# A gate that re-asks every time (allow_session=False, e.g. protected
|
||||
# agent-instruction writes) collapses to the same two options as a Smart
|
||||
# DENY override — offering a scope Hermes discards would re-prompt every write.
|
||||
# See #81887.
|
||||
once_only = smart_denied or not allow_session
|
||||
options = [PermissionOption(option_id="allow_once", kind="allow_once", name="Allow once")]
|
||||
if not once_only:
|
||||
|
||||
@@ -554,6 +554,14 @@ class HermesACPAgent(SlashCommandsMixin, acp.Agent):
|
||||
Best-effort: a corrupt message must not turn the load into an error."""
|
||||
if replay_verb:
|
||||
try:
|
||||
# Per ACP spec, `session/load` must stream the prior conversation back to the client via
|
||||
# `session/update` notifications BEFORE responding, so the client receives the full
|
||||
# transcript within the load request's lifetime. Awaiting the replay here matches Codex /
|
||||
# Claude Code / OpenCode / Pi and the Zed client (which registers the session-update routing
|
||||
# entry before awaiting the loadSession RPC specifically so in-call history replay updates
|
||||
# can find the thread). Deferring this via `loop.call_soon` (as we did briefly in May 2026)
|
||||
# broke every spec-compliant ACP client that measures notifications synchronously against
|
||||
# the load response — see #12285 follow-up.
|
||||
await self._replay_session_history(state)
|
||||
except Exception:
|
||||
logger.warning(
|
||||
|
||||
@@ -44,6 +44,11 @@ def _normalize_cwd_for_compare(cwd: str | None) -> str:
|
||||
# ``/private/tmp``) that otherwise drop a workspace's own sessions; it is lexical
|
||||
# for missing paths (e.g. WSL-translated drives).
|
||||
try:
|
||||
# Resolve symlink aliases so equivalent spellings of the same directory compare equal — macOS
|
||||
# reports editor workspaces as ``/var/...`` while sessions get stored under ``/private/var/...``
|
||||
# (and ``/tmp`` vs ``/private/tmp``), which made ACP history filters silently drop a workspace's own
|
||||
# sessions. WSL-translated Windows drives — keep the previous normpath behavior. Ported from
|
||||
# PrimeIntellect-ai/prime-agent#628.
|
||||
return os.path.realpath(expanded)
|
||||
except OSError:
|
||||
return os.path.normpath(expanded)
|
||||
@@ -340,6 +345,14 @@ class SessionManager:
|
||||
# incrementally (append_message) and keeps pre-compaction turns as archived
|
||||
# active=0 rows; replace_messages() would DELETE those (and, after a compression
|
||||
# id rotation, clobber the ended parent transcript). Skip it in that case.
|
||||
# Calling replace_messages() here would then be a redundant double-write that DELETEs exactly
|
||||
# those archived rows (and, after a compression-driven id rotation where agent.session_id no
|
||||
# longer equals state.session_id, clobbers the ended parent transcript) — silent data loss for
|
||||
# any ACP conversation long enough to compress. Only fall back to the destructive atomic replace
|
||||
# when the agent is NOT persisting itself to this DB (e.g. a test agent factory, or a fresh
|
||||
# create/fork whose copied history the agent has not flushed yet). That path still rolls back on
|
||||
# a mid-rewrite failure so the previously persisted conversation survives (salvaged from
|
||||
# #13675).
|
||||
agent = state.agent
|
||||
if getattr(agent, "_session_db", None) is db and getattr(agent, "_session_db_created", False):
|
||||
return
|
||||
|
||||
@@ -306,6 +306,11 @@ def _format_read_file_result(tool_name: str, data: Args, a: Args) -> Optional[st
|
||||
@_structured()
|
||||
def _format_search_files_result(tool_name: str, data: Args, args: Args) -> Optional[str]:
|
||||
files, matches = data.get("files"), data.get("matches")
|
||||
# Surface file/image attachments as compact text markers. The thread-context fetch is text-only, so
|
||||
# without this the agent has no idea prior messages carried images/files at all (#69185, #32315): "@bot
|
||||
# what do you think of the chart above?" reads as a question about nothing. Markers keep context bounded
|
||||
# — the agent can ask for a re-share (or the caller may separately deliver the thread root's image, see
|
||||
# _collect_thread_root_images).
|
||||
if isinstance(files, list):
|
||||
shown = min(len(files), 20)
|
||||
lines = ["File search results", f"Found {_plural(data.get('total_count', len(files)), 'file')}; showing {shown}.", ""]
|
||||
|
||||
@@ -304,6 +304,11 @@ def _resolve_codex_usage_credentials(
|
||||
# and hand back a DIFFERENT pool account's usage; such errors must propagate to the fail-open outer guard.
|
||||
# account_id is best-effort: a partial singleton store must not sink a usable credential.
|
||||
try:
|
||||
# Tier 2: the native runtime resolver. It ALREADY falls back to the credential pool when the
|
||||
# singleton is empty (see ``resolve_codex_runtime_credentials`` — issue #32992), so in a pool-only
|
||||
# setup this returns a usable ``source="credential_pool"`` token. A refresh/network error must
|
||||
# propagate — the outer ``fetch_account_usage`` guard fails open (shows nothing this turn) rather
|
||||
# than reporting the wrong account.
|
||||
creds = resolve_codex_runtime_credentials(refresh_if_expiring=True)
|
||||
account_id: Optional[str] = None
|
||||
try:
|
||||
|
||||
@@ -33,6 +33,8 @@ class ActivityTrackingMixin:
|
||||
|
||||
``_touch_activity`` stamps under it and the liveness watchdog samples/commits under it, so a stall
|
||||
observation can never abort a turn that resumed in between.
|
||||
|
||||
Created lazily so ``AIAgent.__new__``-based test doubles keep working. See #95663.
|
||||
"""
|
||||
return _activity_lock(self)
|
||||
|
||||
@@ -48,6 +50,9 @@ class ActivityTrackingMixin:
|
||||
projection. ``provenance`` names special writers (compression); ``force_persist`` bypasses the
|
||||
SessionDB rate limit. Module-level lock helper, not ``self._liveness_activity_lock()``: doubles bind
|
||||
only ``_touch_activity`` (tests/run_agent/test_session_activity_persist.py).
|
||||
|
||||
Bridge is rate-limited (60s) and best-effort — it never raises into the agent loop. See #31752.
|
||||
See #72016, #72039.
|
||||
"""
|
||||
from agent.session_activity import (
|
||||
bound_activity_description, normalize_activity_provenance,
|
||||
@@ -117,6 +122,8 @@ class ActivityTrackingMixin:
|
||||
|
||||
Keeps ``_last_activity_ts`` so idle/watchdog clocks stay continuous across turns; clears description +
|
||||
provenance so idle agents / SessionDB listings stop advertising the last mid-turn stamp.
|
||||
|
||||
See #15654, #72039.
|
||||
"""
|
||||
self._last_activity_desc = ""
|
||||
self._last_activity_provenance = ActivityProvenance.UNKNOWN
|
||||
|
||||
@@ -443,6 +443,11 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url):
|
||||
# Covers api.meta.ai → codex_responses (prompt caching: 0% on chat vs 93-99%).
|
||||
# URL-driven, not provider-name-driven: `providers.meta` may point anywhere.
|
||||
try:
|
||||
# Note: provider="meta" without an api.meta.ai base_url (or with a non-api.meta.ai base_url)
|
||||
# intentionally falls through to chat_completions here. The wire protocol for Meta is URL-driven
|
||||
# BY DESIGN, not provider-name-driven, because user config `providers.meta` may point at any
|
||||
# OpenAI-compatible endpoint, and forcing `codex_responses` on the provider name alone would
|
||||
# break custom endpoints named "meta" that do not host the Responses API. See #63425.
|
||||
from hermes_cli.providers import host_mandated_api_mode as _host_mandated_api_mode
|
||||
_mandated = _host_mandated_api_mode(base_url or "")
|
||||
except Exception:
|
||||
@@ -453,6 +458,8 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url):
|
||||
def _finalize_routing(agent, api_mode, credential_pool):
|
||||
# Credential-pool validation runs AFTER provider auto-detection so a pool scoped to
|
||||
# "anthropic" isn't rejected for provider=None + anthropic.com URL.
|
||||
# Regression from #63048 which placed this check before the URL-based auto-detection block above (fixed
|
||||
# #63425).
|
||||
if credential_pool is not None:
|
||||
try:
|
||||
from agent.credential_pool import credential_pool_matches_provider
|
||||
@@ -481,6 +488,13 @@ def _finalize_routing(agent, api_mode, credential_pool):
|
||||
# exceptions live in _provider_model_requires_responses_api.
|
||||
_base_lower = str(agent.base_url or "").lower()
|
||||
if (
|
||||
# GPT-5.x models usually require the Responses API path, but some providers have exceptions (for
|
||||
# example Copilot's gpt-5-mini still uses chat completions). ACP runtimes are excluded: an ACP
|
||||
# client handles its own routing and does not implement the Responses API surface. Keyed on the
|
||||
# `acp://` scheme, not one vendor, so every ACP client is covered. When api_mode was explicitly
|
||||
# provided, respect it — the user knows what their endpoint supports (#10473). Exception: Azure
|
||||
# OpenAI serves gpt-5.x on /chat/completions and does NOT support the Responses API — skip the
|
||||
# upgrade for Azure (openai.azure.com), even though it looks OpenAI-compatible.
|
||||
api_mode is None
|
||||
and agent.api_mode == "chat_completions"
|
||||
and agent.provider != "copilot-acp"
|
||||
@@ -647,6 +661,11 @@ def _init_prompt_cache_config(agent):
|
||||
# unknown values keep "5m". A falsy/off value disables caching entirely (OAuth plans
|
||||
# billing cache writes, proxies adding their own cache_control); the disable survives
|
||||
# /model switches and fallback re-derivation.
|
||||
# Anthropic supports "5m" (default) and "1h" cache TTL tiers. Read from config.yaml under
|
||||
# prompt_caching.cache_ttl; unknown values keep "5m". 1h tier costs 2x on write vs 1.25x for 5m, but
|
||||
# amortizes across long sessions with >5-minute pauses between turns (#14971). This is useful for OAuth
|
||||
# subscription users where cache writes bill against "extra usage" or for third-party proxies that
|
||||
# inject their own cache_control markers (#13477).
|
||||
agent._cache_ttl = "5m"
|
||||
with suppress(Exception):
|
||||
from hermes_cli.config import load_config_readonly as _load_pc_cfg
|
||||
@@ -721,6 +740,7 @@ def _init_anthropic_client(agent, api_key, base_url, _provider_timeout):
|
||||
return
|
||||
# ANTHROPIC_TOKEN fallback only for native Anthropic — other anthropic_messages providers
|
||||
# must use their own key or Anthropic credentials leak to third-party endpoints.
|
||||
# Falling back would send Anthropic credentials to third-party endpoints (Fixes #1739, #minimax-401).
|
||||
_is_native_anthropic = agent.provider == "anthropic"
|
||||
effective_key = api_key or (resolve_anthropic_token() if _is_native_anthropic else None) or ""
|
||||
|
||||
@@ -742,6 +762,10 @@ def _init_anthropic_client(agent, api_key, base_url, _provider_timeout):
|
||||
agent._anthropic_api_key = effective_key
|
||||
# OAuth only for native Anthropic: third-party anthropic_messages providers must never
|
||||
# trip OAuth paths — those inject Claude-Code identity headers → 401/403.
|
||||
# Only mark the session as OAuth-authenticated when the token genuinely belongs to native Anthropic.
|
||||
# Third-party providers (MiniMax, Kimi, GLM, LiteLLM proxies) that accept the Anthropic protocol must
|
||||
# never trip OAuth code paths — doing so injects Claude-Code identity headers and system prompts that
|
||||
# cause 401/403 on their endpoints. See #1739.
|
||||
from agent.anthropic_adapter import _is_oauth_token as _is_oat
|
||||
agent._is_anthropic_oauth = _is_oat(effective_key) if (_is_native_anthropic and isinstance(effective_key, str)) else False
|
||||
agent._anthropic_client = build_anthropic_client(effective_key, base_url, timeout=_provider_timeout)
|
||||
@@ -758,6 +782,12 @@ def _init_moa_client(agent, api_key):
|
||||
# build_moa_facade relays "moa.*" events through tool_progress_callback so every surface
|
||||
# shows each reference's answer before the aggregator acts. Display-only; shared with
|
||||
# fallback-restore so a restored facade keeps emitting.
|
||||
# build_moa_facade wires the reference relay that routes reference-model outputs to the agent's
|
||||
# tool_progress_callback so every surface that already consumes it (CLI spinner/scrollback, TUI,
|
||||
# desktop, gateway) can show each reference's answer as a labelled block before the aggregator acts. The
|
||||
# facade emits "moa.reference", "moa.progress", "moa.phase", and "moa.aggregating" events, forwarded
|
||||
# through the same callback the tool lifecycle uses. Best-effort and cache-safe — display-only events,
|
||||
# they never touch the message history. See #53802.
|
||||
agent.client = build_moa_facade(agent, agent.model)
|
||||
agent._client_kwargs = {}
|
||||
agent.api_key = api_key or "moa-virtual-provider"
|
||||
@@ -837,6 +867,9 @@ def _routed_client_kwargs(agent, fallback_model, _provider_timeout) -> Dict[str,
|
||||
# No credentials: try the fallback chain BEFORE failing (an exhausted single-entry pool
|
||||
# must not die with a misleading "No LLM provider configured"); only explicitly named
|
||||
# providers keep the missing-key diagnostic.
|
||||
# An exhausted single-entry pool (typically ``openrouter`` under free-tier daily quotas) must still
|
||||
# reach the chain instead of dying at init with a misleading "No LLM provider configured" error. See
|
||||
# #17929.
|
||||
_explicit = (agent.provider or "").strip().lower()
|
||||
for _fb in _fallback_entries(fallback_model):
|
||||
try:
|
||||
@@ -1237,6 +1270,11 @@ def _init_memory(agent, _agent_cfg, skip_memory, platform):
|
||||
agent._iters_since_skill = 0
|
||||
# skip_memory skips the external *provider*; enabled_toolsets=["memory"] still gets the
|
||||
# built-in store so the memory tool never sees store=None.
|
||||
# Flush/background agents can still pass enabled_toolsets=["memory"] so the built-in file store exists
|
||||
# and the memory tool does not fail with store=None (#65429). A toolset on disabled_toolsets is not a
|
||||
# request: a caller that denylists memory while its default toolset still names it must not get
|
||||
# MEMORY.md loaded by an enabled-only check. (Cron agents now run with skip_memory=False and take the
|
||||
# normal path here.)
|
||||
_memory_toolset_requested = (
|
||||
"memory" in (agent.enabled_toolsets or [])
|
||||
and "memory" not in (agent.disabled_toolsets or [])
|
||||
@@ -1768,6 +1806,9 @@ def _select_context_engine(_agent_cfg):
|
||||
# parent's. Uncopyable state (locks, DB conns) → built-in with an ACCURATE message.
|
||||
import copy
|
||||
try:
|
||||
# Copy can fail for engines holding uncopyable state (locks, DB connections, clients); in
|
||||
# that case fall back to the built-in compressor with an ACCURATE message rather than
|
||||
# silently mislabelling it "not found". See #42449.
|
||||
_selected_engine = copy.deepcopy(_candidate)
|
||||
except Exception as _copy_err:
|
||||
_copy_failed = True
|
||||
@@ -1812,6 +1853,9 @@ def _build_context_engine(agent, _agent_cfg, cs, _custom_providers, _effective_c
|
||||
# External engines own compaction policy — the host threshold (and its Codex
|
||||
# autoraise) never reaches the plugin, so drop the notice.
|
||||
agent._compression_threshold_autoraised = None
|
||||
# External engines own compaction policy: the host compression threshold (including the Codex
|
||||
# gpt-5.5 autoraise above) only configures the built-in ContextCompressor and never reaches the
|
||||
# plugin, so the autoraise notice would announce a change that does not apply. (#44439)
|
||||
from agent.model_metadata import get_model_context_length
|
||||
_plugin_ctx_len = get_model_context_length(
|
||||
agent.model, base_url=agent.base_url, api_key=getattr(agent, "api_key", ""),
|
||||
@@ -1924,6 +1968,14 @@ def _inject_context_engine_tools(agent):
|
||||
# Context engine tool schemas (lcm_*), deduped against existing names (plugins may
|
||||
# register the same schemas; duplicates 400 provider-side) and gated on enabled_toolsets
|
||||
# so `platform_toolsets: telegram: []` can't leak them.
|
||||
# Skip names that are already present — the _ra().get_tool_definitions() quiet_mode cache returned a
|
||||
# shared list pre-#17335, so a stray mutation here would poison subsequent agent inits in the same
|
||||
# Gateway process and trip provider-side 'duplicate tool name' errors. Even with the cache fix, dedup is
|
||||
# the right defense against plugin paths that may register the same schemas via ctx.register_tool().
|
||||
# Mirrors the memory tools dedup above. Respect the platform's enabled_toolsets configuration (#5544):
|
||||
# context engine tools follow the same gating pattern as memory provider tools — without the gate,
|
||||
# `platform_toolsets: telegram: []` would still leak lcm_* tools into the tool surface and incur the
|
||||
# same local-model latency penalty.
|
||||
agent._context_engine_tool_names: set = set()
|
||||
if (
|
||||
agent.context_compressor
|
||||
@@ -1939,6 +1991,7 @@ def _inject_context_engine_tools(agent):
|
||||
if _schema is None:
|
||||
# A nameless tool makes strict providers 400 and disables the whole toolset.
|
||||
_ra().logger.warning(
|
||||
# Skip it. See #47707.
|
||||
"Context engine returned a tool schema with no resolvable "
|
||||
"name; skipping to avoid poisoning the request (%r)",
|
||||
_raw_schema,
|
||||
@@ -2002,6 +2055,11 @@ def _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length):
|
||||
)
|
||||
# Recalibrate the compressor to the served window: every request runs at num_ctx, so a
|
||||
# trigger derived from the probed model window could sit above it and never fire.
|
||||
# A config that sets only model.ollama_num_ctx (without model.context_length) previously left the
|
||||
# compressor targeting the probed window while the server truncated/rejected at num_ctx — the compaction
|
||||
# trigger could sit several times ABOVE the real served window and never fire. Clamp the compressor's
|
||||
# window to the effective num_ctx so threshold math operates on the context the server actually serves.
|
||||
# (Overlaps #60103's silent-clamp dead zone; this is the init-order half.)
|
||||
_cc_window = getattr(agent.context_compressor, "context_length", 0) or 0
|
||||
if agent._ollama_num_ctx and agent._ollama_num_ctx > 0 and _cc_window and agent._ollama_num_ctx < _cc_window:
|
||||
_ra().logger.info(
|
||||
@@ -2020,6 +2078,9 @@ def _emit_compression_summary(agent, cs):
|
||||
_autoraise = agent._compression_threshold_autoraised or {}
|
||||
_autoraise_notice = None
|
||||
if (
|
||||
# A change in the raised threshold (or the autoraised model) updates the marker state and
|
||||
# re-notifies once. The config display gate (compression.codex_gpt55_autoraise_notice) still
|
||||
# suppresses the banner entirely without disabling the threshold autoraise. See #54432.
|
||||
bool(_autoraise)
|
||||
and cs.enabled
|
||||
and cs.autoraise_notice_enabled
|
||||
|
||||
@@ -223,6 +223,14 @@ def sanitize_tool_call_arguments(
|
||||
``cursor["prefix"]`` holds strong refs (not ``id()``: address reuse aliases) to the
|
||||
messages validated last call; the ``is``-identical prefix is skipped. Safe because only
|
||||
the surrogate sanitizers mutate live dicts; every other path replaces dicts, breaking identity.
|
||||
|
||||
Safety argument for skipping: a message in the matched prefix was fully scanned before — every tool_call
|
||||
argument was either already valid JSON or was rewritten to ``"{}"`` (valid). The only code paths that
|
||||
mutate ``function["arguments"]`` on live history dicts between calls are the surrogate / non-ASCII
|
||||
sanitizers, which substitute characters *inside* JSON string values and cannot invalidate JSON syntax.
|
||||
Compression, repair, undo, and steer paths replace or reorder message dicts, which breaks the identity
|
||||
match and forces a re-scan. Holding strong references (the objects themselves, not ``id()``s) makes
|
||||
address reuse aliasing (#50372-style) impossible.
|
||||
"""
|
||||
log = logger or logging.getLogger(__name__)
|
||||
if not isinstance(messages, list):
|
||||
@@ -252,6 +260,8 @@ def sanitize_tool_call_arguments(
|
||||
continue
|
||||
# Canonical ``call_id || id`` precedence so scan and stub share the id the pipeline
|
||||
# uses; bare ``id`` misses Codex call_id results and orphans a stub.
|
||||
# Keying on bare ``id`` here would fail to find a result built with ``call_id`` (Codex Responses
|
||||
# format) and insert a duplicate stub that itself becomes an orphan (#58168).
|
||||
tool_call_id = _ra().AIAgent._get_tool_call_id_static(tool_call) or None
|
||||
function_name = function.get("name", "?")
|
||||
# Log the FULL (bounded) argument string: we are about to overwrite the only copy, which
|
||||
@@ -365,6 +375,15 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None:
|
||||
else:
|
||||
# Drop a stale ``tool_calls: []`` at the source: strict providers (DeepSeek v4, Kimi) 400 on
|
||||
# it and it persists into replayed history.
|
||||
# Neither turn carries tool calls, but the surviving turn may still carry a stale ``tool_calls: []``
|
||||
# from the earlier message. An empty array is semantically "no tool calls", yet strict
|
||||
# OpenAI-compatible providers (DeepSeek v4, Moonshot/Kimi) reject it with HTTP 400 ("Invalid
|
||||
# 'messages[N].tool_calls': empty array..."). Drop the key HERE, at the source:
|
||||
# ``sanitize_api_messages`` only fixes the per-call wire copy, so a ``[]`` left on the repaired turn
|
||||
# survives in the live/persisted trajectory returned to callers (gateway/WebUI transcripts, session
|
||||
# resume, subagents, cron) and is replayed on the next turn — which is how #58755 kept reproducing
|
||||
# after the chokepoint fix (#77921). Popping is non-destructive: an empty array carries no
|
||||
# information.
|
||||
prev.pop("tool_calls", None)
|
||||
# Concatenate plain-text content only; leave multimodal (list) content alone.
|
||||
prev_content = prev.get("content")
|
||||
@@ -374,6 +393,8 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None:
|
||||
joined = "\n".join(p for p in (prev_content.strip(), new_content.strip()) if p)
|
||||
prev["content"] = joined
|
||||
# A falsy new_content leaves ``joined`` == prev_content; that is not a rewrite.
|
||||
# "") strips to nothing and ``joined`` collapses back to ``prev_content`` unchanged -- that must NOT
|
||||
# count as a rewrite (wz-heng, #78063 review).
|
||||
content_rewritten = joined != prev_content
|
||||
elif not prev_content and new_content is not None:
|
||||
prev["content"] = new_content
|
||||
@@ -384,6 +405,18 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None:
|
||||
prev["reasoning_content"] = msg["reasoning_content"]
|
||||
# A stale ``api_content`` sidecar overrides ``content`` at API-build time and would replay
|
||||
# pre-merge bytes; drop it only when content actually changed.
|
||||
# ``prev`` may carry an ``api_content`` sidecar (the exact bytes previously sent to the API, e.g. a
|
||||
# sanitize-divergence stamp — see ``_flush_messages_to_session_db``) from BEFORE this merge. The sidecar
|
||||
# takes priority over ``content`` at API-build time (``conversation_loop``'s ``api_messages`` build
|
||||
# substitutes it back in for role ``assistant``), so leaving it in place while ``prev["content"]``
|
||||
# changes would silently replay the pre-merge bytes and discard everything this merge just concatenated
|
||||
# on — the same stale-field-survives-the-merge shape as the ``tool_calls`` gap above, just for a
|
||||
# different field. Only drop it when the merge actually changed the resulting value (e.g. the later
|
||||
# turn's content is ``None``, or either side is multimodal/list — both branches skip the reassignment
|
||||
# and ``prev["content"]`` is untouched; a falsy ``new_content`` that strips to nothing also leaves
|
||||
# ``joined`` equal to the original ``prev_content``): in those cases the sidecar is still the exact
|
||||
# bytes previously sent for the UNCHANGED content, and dropping it would break the prompt-cache replay
|
||||
# invariant for no reason (wz-heng, #78063 review).
|
||||
if content_rewritten:
|
||||
drop_stale_api_content(prev)
|
||||
|
||||
@@ -416,6 +449,11 @@ def _drop_stray_tool_results(messages: List[Dict]) -> Tuple[List[Dict], int]:
|
||||
alias is not replayed to strict providers."""
|
||||
repairs = 0
|
||||
known_tool_ids: Dict[str, int] = {} # alias -> group id; reset by assistant/user turns
|
||||
# Pass 1: drop stray tool messages that don't follow a known assistant tool call. A Responses call can
|
||||
# have several equivalent spellings (call_id, id, response_item_id, or a composite ``call|item`` id), so
|
||||
# consume the whole alias group when one spelling is matched. Alias expansion lives in
|
||||
# ``agent.message_sanitization.tool_call_id_variants`` / ``tool_result_id_variants`` (single policy
|
||||
# owner) — which also handles SDK tool_call objects, preserving the #91768 dict-or-object tolerance.
|
||||
matched_tool_groups: set = set()
|
||||
next_tool_group = 0
|
||||
filtered: List[Dict] = []
|
||||
@@ -644,6 +682,18 @@ def _is_entitlement_403(agent, status_code, error_context) -> bool:
|
||||
if status_code != 403:
|
||||
return False
|
||||
haystack = " ".join(
|
||||
# Subscription/entitlement 403s look like auth failures on the wire but refresh cannot fix them —
|
||||
# the OAuth token is already valid, the account simply lacks the entitlement. Without this guard,
|
||||
# the refresh path keeps minting fresh tokens against the same unsubscribed account and the main
|
||||
# agent loop spins re-issuing the same 403 until the user Ctrl+C's. Defense-in-depth for #26847:
|
||||
# xAI's backend has been seen to 403 standard SuperGrok subscribers with bodies that don't match the
|
||||
# existing entitlement keyword set in ``_is_entitlement_failure``. Any 403 against ``xai-oauth`` is
|
||||
# treated as entitlement here so the refresh loop can't spin in those cases either. Exception
|
||||
# (#29344): xAI's ``[WKE=unauthenticated:...]`` suffix and the ``OAuth2 access token could not be
|
||||
# validated`` phrasing are xAI's authoritative "this is a stale token, not entitlement" signal. When
|
||||
# either fires we must NOT apply the catch-all override — refresh is the recoverable path for these
|
||||
# bodies, and blanket-classifying them as entitlement was the bug that left long-running TUI
|
||||
# sessions stuck on stale tokens until the user exited and reopened.
|
||||
str(error_context.get(k) or "").lower()
|
||||
for k in ("message", "reason", "code", "error")
|
||||
if isinstance(error_context, dict)
|
||||
@@ -746,6 +796,11 @@ def recover_with_credential_pool(
|
||||
# The pool belongs to the PRIMARY provider: acting on fallback errors would corrupt its state
|
||||
# and reset base_url to the primary endpoint. Empty pool provider means unscoped; empty agent
|
||||
# provider is a mismatch (swap would leave provider="" model="").
|
||||
# Defensive guard: if a fallback provider is active and its provider name doesn't match the pool's
|
||||
# provider, the pool belongs to the PRIMARY provider. Mutating it based on fallback errors would corrupt
|
||||
# the primary's credential state (see #33088) and, via _swap_credential, overwrite the agent's base_url
|
||||
# back to the primary's endpoint — every subsequent request then goes to the wrong host and 404s (see
|
||||
# #33163). The pool should only act when the agent is still on the same provider that seeded the pool.
|
||||
current_provider = (getattr(agent, "provider", "") or "").strip().lower()
|
||||
pool_provider = (getattr(pool, "provider", "") or "").strip().lower()
|
||||
if pool_provider and not credential_pool_matches_provider(
|
||||
@@ -860,6 +915,10 @@ def _rebuild_primary_client(agent, rt: Dict[str, Any], *, reason: str) -> None:
|
||||
# reference_callback relay survives recovery.
|
||||
from agent.moa_loop import build_moa_facade
|
||||
agent.client = build_moa_facade(agent, agent.model)
|
||||
# MoA is a virtual chat-completions provider. It never has real OpenAI client kwargs; restoring it
|
||||
# after a fallback must recreate the facade, not call OpenAI() with an empty api_key. Use the shared
|
||||
# factory so the restored facade keeps the reference_callback relay wired at init — a bare
|
||||
# MoAClient() would silently stop emitting moa.reference/moa.aggregating display events (#53802).
|
||||
agent._anthropic_client = None
|
||||
elif agent.api_mode == "anthropic_messages":
|
||||
_build_anthropic_client_from_runtime(agent, rt)
|
||||
@@ -885,6 +944,11 @@ def try_recover_primary_transport(
|
||||
try:
|
||||
# Never hard-close the shared client here: stale streaming workers may still be unwinding on
|
||||
# the old pool; _retire_shared_openai_client defers FD release to GC.
|
||||
# Retire the existing client to release stale connections. #70773: never hard-close the shared
|
||||
# client here — this runs on the conversation-loop thread while workers from stale-killed streaming
|
||||
# attempts may still be unwinding their SSL BIOs on the old pool. ``_retire_shared_openai_client``
|
||||
# shuts the sockets down (FD-safe from any thread) and defers the FD release to GC, which cannot
|
||||
# complete until every borrowing thread has unwound.
|
||||
if getattr(agent, "client", None) is not None:
|
||||
with contextlib.suppress(Exception):
|
||||
agent._retire_shared_openai_client(agent.client, reason="primary_recovery")
|
||||
@@ -893,6 +957,9 @@ def try_recover_primary_transport(
|
||||
if agent.api_mode == "anthropic_messages":
|
||||
_build_anthropic_client_from_runtime(agent, rt)
|
||||
elif (agent.provider or "").strip().lower() == "moa":
|
||||
# MoA is a virtual provider with empty client_kwargs — rebuilding via _create_openai_client
|
||||
# would raise "api_key client option must be set". Recreate the facade through the shared
|
||||
# factory so the reference_callback relay survives recovery (#53802).
|
||||
from agent.moa_loop import build_moa_facade
|
||||
agent.client = build_moa_facade(agent, agent.model)
|
||||
else:
|
||||
@@ -1021,6 +1088,8 @@ def _rebind_primary_credential_pool(agent, primary_provider, matches_primary, lo
|
||||
return
|
||||
if matches_primary(entry):
|
||||
# _swap_credential rebuilds the client and reapplies base-url-scoped headers.
|
||||
# ``_swap_credential`` rebuilds the OpenAI/Anthropic client, reapplies base-url-scoped headers, and
|
||||
# carries the accumulated base_url / OAuth-detection fixes (#33163).
|
||||
agent._swap_credential(entry)
|
||||
logger.info(
|
||||
"Restore re-selected pool entry %s (%s)",
|
||||
@@ -1043,6 +1112,11 @@ def restore_primary_runtime(agent) -> bool:
|
||||
# _fallback_index past the chain end and silently block future fallbacks.
|
||||
agent._fallback_index = 0
|
||||
return False
|
||||
# Reset the chain index even when no fallback was activated this turn. Without this, a turn where
|
||||
# _try_activate_fallback() was called but returned False (chain exhausted or provider not configured)
|
||||
# leaves _fallback_index >= len(_fallback_chain) while _fallback_activated stays False. The next turn
|
||||
# skips this block entirely, stranding the index and silently blocking all future fallback attempts for
|
||||
# the session. Fixes #20465.
|
||||
if getattr(agent, "_rate_limited_until", 0) > time.monotonic():
|
||||
return False # primary still in rate-limit cooldown, stay on fallback
|
||||
rt = agent._primary_runtime
|
||||
@@ -1150,6 +1224,7 @@ def extract_reasoning(agent, assistant_message) -> Optional[str]:
|
||||
if not parts and isinstance(content, list):
|
||||
# DeepSeek V4 Pro returns typed content blocks ({"type": "thinking", ...}); dropping them
|
||||
# makes the next turn fail with HTTP 400 "thinking must be passed back".
|
||||
# Refs #21944.
|
||||
for block in content:
|
||||
if isinstance(block, dict) and block.get("type") == "thinking":
|
||||
_add((block.get("thinking") or block.get("text") or "").strip())
|
||||
@@ -1253,7 +1328,12 @@ def _raw_cache_ttl_from_config(default: Any) -> Any:
|
||||
|
||||
|
||||
def prompt_caching_disabled_from_config() -> bool:
|
||||
"""True when ``prompt_caching.cache_ttl`` is configured as off (same detection as ``agent_init``)."""
|
||||
"""True when ``prompt_caching.cache_ttl`` is configured as off (same detection as ``agent_init``).
|
||||
|
||||
Same disable detection as ``agent_init`` (via ``cache_ttl_means_disabled``) so stub-based policy paths
|
||||
(MoA slot decoration, auxiliary fallback replan) honor the same config contract without holding a live
|
||||
``AIAgent`` (#76085 / #33555).
|
||||
"""
|
||||
return cache_ttl_means_disabled(_raw_cache_ttl_from_config("5m"))
|
||||
|
||||
|
||||
@@ -1281,11 +1361,18 @@ def plan_cache_sections_for_destination(
|
||||
"""Plan request-local cache sections for one resolved destination (MoA / auxiliary senders):
|
||||
stripped copies (non-caching route) or a ``build_prompt_cache_plan`` layout; never mutates
|
||||
inputs. ``cache_disabled``/``cache_ttl`` default to live config so the operator's disable and
|
||||
tier are honored; ``static_system_prefix`` gives the system prompt the main loop's early breakpoint."""
|
||||
tier are honored; ``static_system_prefix`` gives the system prompt the main loop's early breakpoint.
|
||||
|
||||
``cache_disabled`` threads the operator's ``prompt_caching.cache_ttl`` disable into the blank policy
|
||||
stub. When omitted, the live config is consulted so MoA/auxiliary paths cannot re-enable markers after
|
||||
the user turned caching off (#76085).
|
||||
"""
|
||||
from agent.prompt_caching import (
|
||||
build_prompt_cache_plan, effective_cache_ttl, envelope_tool_part_cache_markers_supported,
|
||||
strip_anthropic_cache_control, strip_anthropic_tool_cache_control,
|
||||
)
|
||||
# The policy function reads agent.* only as fallbacks for kwargs we don't pass; blank_cache_policy_stub
|
||||
# is the only sanctioned stub so _cache_disabled cannot be left off again (#76085).
|
||||
stub = blank_cache_policy_stub(cache_disabled)
|
||||
dest = dict(provider=provider, base_url=base_url, api_mode=api_mode, model=model)
|
||||
should_cache, native_layout = anthropic_prompt_cache_policy(stub, **dest)
|
||||
@@ -1386,6 +1473,11 @@ def anthropic_prompt_cache_policy(
|
||||
envelope (OpenRouter / OpenAI-wire proxies; Qwen/Alibaba too). The operator disable is read
|
||||
from ``_cache_disabled`` (not ``_cache_ttl``, unset during init) so it survives switches
|
||||
and restores. Branch ORDER is load-bearing (see inline notes).
|
||||
|
||||
Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct Alibaba (DashScope) also honour
|
||||
Anthropic-style ``cache_control`` markers on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi
|
||||
#3393 documented this for opencode-go Qwen. Without markers these providers serve zero cache hits,
|
||||
re-billing the full prompt on every turn.
|
||||
"""
|
||||
if getattr(agent, "_cache_disabled", False):
|
||||
return (False, False)
|
||||
@@ -1403,6 +1495,10 @@ def anthropic_prompt_cache_policy(
|
||||
is_claude = "claude" in model_lower
|
||||
# Kimi/Moonshot via OpenRouter uses the same envelope cache_control as Claude; without this it
|
||||
# serves ~1% cache hits. Family matcher covers bare k1./k2. slugs.
|
||||
# Without this branch moonshotai/kimi-k2.6 falls through to (False, False), serving ~1% cache hits on
|
||||
# 64K-token prompts and re-billing the full prompt on every turn. Observed within-turn progression with
|
||||
# cache enabled: 1% → 67% → 84% → 97% (#25970). Reuses the canonical family matcher (covers bare
|
||||
# k1./k2./k25 release slugs the substring check missed).
|
||||
from agent.anthropic_adapter import _model_name_is_kimi_family
|
||||
is_kimi = _model_name_is_kimi_family(eff_model) or "moonshot" in model_lower
|
||||
is_openrouter = base_url_host_matches(eff_base_url, "openrouter.ai")
|
||||
@@ -1471,6 +1567,11 @@ def anthropic_prompt_cache_policy(
|
||||
# Qwen/Alibaba on OpenCode and DashScope accept envelope cache_control on the OpenAI wire
|
||||
# (pi-mono's "alibaba" cacheControlFormat). DeepSeek on OpenCode is excluded: its relay 400s on
|
||||
# block-array content. Family set/predicate shared with the effective_cache_ttl clamp.
|
||||
# Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire transport that accepts
|
||||
# Anthropic-style cache_control markers and rewards them with real cache hits. Without this branch
|
||||
# qwen3.6-plus on opencode-go reports 0% cached tokens and burns through the subscription on every turn.
|
||||
# OpenCode Zen's relay rejects the Anthropic-style content block format that cache markers produce
|
||||
# (content becomes a block array instead of a plain string), causing HTTP 400 (#77217).
|
||||
from agent.prompt_caching import ALIBABA_FAMILY_PROVIDERS, is_qwen_model
|
||||
if provider_lower in ALIBABA_FAMILY_PROVIDERS and is_qwen_model(model_lower):
|
||||
return True, False
|
||||
@@ -1569,9 +1670,20 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo
|
||||
from agent.ssl_verify import resolve_httpx_verify
|
||||
# Treat client_kwargs as read-only: callers pass agent._client_kwargs, and in-place mutation
|
||||
# leaks into later requests (a torn-down httpx transport got reused).
|
||||
# Callers pass agent._client_kwargs (or shallow copies of it) in; any in-place mutation leaks back into
|
||||
# the stored dict and is reused on subsequent requests. #10933 hit this by injecting an httpx.Client
|
||||
# transport that was torn down after the first request, so the next request wrapped a closed transport
|
||||
# and raised "Cannot send a request, as the client has been closed" on every retry. The revert resolved
|
||||
# that specific path; this copy locks the contract so future transport/keepalive work can't reintroduce
|
||||
# the same class of bug.
|
||||
client_kwargs = dict(client_kwargs)
|
||||
# The MoA virtual provider has no OpenAI wire endpoint; the facade *is* the client. Rebuild the
|
||||
# facade, never a native client (TypeError; relay re-wire).
|
||||
# Rebuilding a native OpenAI client while agent.provider == "moa" (client replacement, stream-retry pool
|
||||
# cleanup, credential rotation, fallback+restore) drops the facade: the next primary call either raises
|
||||
# a `_moa_prepared_request` TypeError (#78382) or, when _client_kwargs carry an unrelated relay
|
||||
# base_url, leaks the request to a foreign gateway. Rebuild the facade instead (build_moa_facade also
|
||||
# re-wires the reference relay, see #53802).
|
||||
if (getattr(agent, "provider", "") or "").strip().lower() == "moa":
|
||||
from agent.moa_loop import build_moa_facade
|
||||
return build_moa_facade(agent, getattr(agent, "model", None) or "default")
|
||||
@@ -1604,12 +1716,26 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo
|
||||
# behind a per-client view whose ``close()`` is a no-op for the pool, so a closed wrapper
|
||||
# never takes a sibling's (or the successor's) connections with it
|
||||
# (tests/agent/test_shared_http_transport.py).
|
||||
# Without this, a peer that drops mid-stream leaves the socket in a state where epoll_wait never fires,
|
||||
# ``httpx`` read timeout may not trigger, and the agent hangs until manually killed. Probes after 30s
|
||||
# idle, retry every 10s, give up after 3 → dead peer detected within ~60s. Safety against #10933: the
|
||||
# ``client_kwargs = dict(client_kwargs)`` above means this injection only lands in the local per-call
|
||||
# copy, never back into ``agent._client_kwargs``. Each ``_create_openai_client`` invocation therefore
|
||||
# gets its OWN fresh ``httpx.Client`` whose lifetime is tied to the OpenAI client it is passed to. When
|
||||
# the OpenAI client is closed (rebuild, teardown, credential rotation), the paired ``httpx.Client``
|
||||
# closes with it, and the next call constructs a fresh one — no stale closed transport can be reused.
|
||||
if "http_client" not in client_kwargs:
|
||||
keepalive_http = agent._build_keepalive_http_client(client_kwargs.get("base_url", ""), verify=httpx_verify)
|
||||
if keepalive_http is not None:
|
||||
client_kwargs["http_client"] = keepalive_http
|
||||
# Retries belong to the outer conversation loop (honors Retry-After); SDK retries would
|
||||
# double-retry inside it. auxiliary_client keeps SDK retries as it isn't wrapped.
|
||||
# Delegate all rate-limit / 5xx retry to hermes's outer conversation loop, which honors Retry-After and
|
||||
# applies adaptive/jittered backoff. The OpenAI SDK default (max_retries=2) uses its own 1-2s backoff
|
||||
# that ignores Retry-After and double-retries inside our loop — the same deadlock the Anthropic clients
|
||||
# hit (#26293). This is the single chokepoint every primary OpenAI/aggregator client passes through
|
||||
# (init, switch_model, recovery, restore, request-scoped); auxiliary_client builds its own clients and
|
||||
# keeps SDK retries because it is NOT wrapped by the conversation loop.
|
||||
client_kwargs.setdefault("max_retries", 0)
|
||||
_ensure_copilot_headers(client_kwargs)
|
||||
# OpenCode Free is served anonymously: any unrecognized bearer is a 401, so an empty
|
||||
@@ -1768,6 +1894,9 @@ def _build_switched_client(agent, new_provider, api_key, base_url, api_mode, new
|
||||
)
|
||||
# Read live config, not agent._custom_providers, so mid-session ssl_ca_cert / ssl_verify
|
||||
# edits are honored.
|
||||
# Read custom_providers from live config (not the init-time snapshot on ``agent._custom_providers``)
|
||||
# so ssl_ca_cert / ssl_verify edits are honored when switching mid-session, matching the
|
||||
# context-length reload below (#15779).
|
||||
apply_custom_provider_tls_to_client_kwargs(
|
||||
agent._client_kwargs, str(effective_base or ""),
|
||||
get_compatible_custom_providers(load_config_readonly()),
|
||||
@@ -1906,6 +2035,7 @@ def _build_primary_runtime_snapshot(agent, api_mode) -> Dict[str, Any]:
|
||||
"reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False),
|
||||
# Overrides must travel with the switched-to identity or a later recovery/restore resurrects
|
||||
# PRE-switch overrides from the stale init snapshot.
|
||||
# See #75091.
|
||||
"request_overrides": dict(getattr(agent, "request_overrides", {}) or {}),
|
||||
"runtime_capabilities": dict(getattr(agent, "runtime_capabilities", {}) or {}),
|
||||
"compressor_model": getattr(cc, "model", agent.model),
|
||||
@@ -1973,6 +2103,12 @@ def switch_model(
|
||||
snapshot and re-raises (callers catch)."""
|
||||
old_model = agent.model
|
||||
old_provider = agent.provider
|
||||
# ── Reload credential pool for the new provider (issue #52727) ── Without this,
|
||||
# ``recover_with_credential_pool`` sees a ``pool.provider != agent.provider`` mismatch and
|
||||
# short-circuits, leaving the new provider with no rotation/recovery on 401/429 and burning the original
|
||||
# pool's entries. Only reload when the provider actually changed (or the pool was missing) —
|
||||
# re-selecting the same provider must not churn the pool reference. A reload failure is logged +
|
||||
# swallowed: the switch itself must still complete.
|
||||
old_norm = (old_provider or "").strip().lower()
|
||||
new_norm = (new_provider or "").strip().lower()
|
||||
api_mode, base_url, destination_capabilities = _resolve_switch_destination(
|
||||
@@ -2131,6 +2267,12 @@ def repair_tool_call(agent, tool_name: str) -> str | None:
|
||||
# VolcEngine api/plan leaks XML attribute fragments into tool_use.name (`terminal"
|
||||
# parameter="command" ...`); trim at the first quote/angle bracket. Do NOT split on whitespace:
|
||||
# "write file" must reach ``_norm`` -> ``write_file``.
|
||||
# `terminal" parameter="command" string="true` `execute_code" parameter="code" string="true`
|
||||
# `session_search" parameter="session_id" string="true` We trim at the first unambiguous XML/quote
|
||||
# character so the rest of the repair pipeline (lowercase / snake_case / fuzzy match) can resolve the
|
||||
# cleaned name to a real tool. Crucially we DO NOT split on whitespace: legitimate inputs like "write
|
||||
# file" must keep flowing through ``_norm`` -> ``write_file`` (covered by test_space_to_underscore in
|
||||
# tests/run_agent/test_repair_tool_call_name.py). See #33007.
|
||||
for _xml_sep in ('"', "'", "<", ">"):
|
||||
_idx = tool_name.find(_xml_sep)
|
||||
if _idx > 0:
|
||||
@@ -2171,12 +2313,19 @@ _INTERRUPTED_PLACEHOLDER = "[response interrupted]"
|
||||
|
||||
# Escalate repeated heals once per session window, then stay quiet. Default threshold; tunable via
|
||||
# ``agent.sanitizer_heal_escalation_threshold`` (<= 0 disables).
|
||||
# Repeated heals of the same poisoned transcript used to WARNING on every send (#96870).
|
||||
# ``_EMPTY_HEAL_ESCALATE_AFTER`` is the built-in default; deployments tune it via
|
||||
# ``agent.sanitizer_heal_escalation_threshold`` in config.yaml (<= 0 disables escalation entirely — WARNINGs
|
||||
# still fire per window).
|
||||
_EMPTY_HEAL_ESCALATE_AFTER = 3
|
||||
_EMPTY_HEAL_WINDOW_S = 600.0
|
||||
_empty_heal_log_state: Dict[str, Dict[str, Any]] = {}
|
||||
_empty_heal_log_lock = threading.Lock()
|
||||
# Sessions already told ONCE (out-of-band, never in conversation context); kept apart from the
|
||||
# windowed log state so a new window never re-arms the notice.
|
||||
# Session keys that already received the one-time user notice. Separate from the windowed log state so a new
|
||||
# 10-minute window never re-notifies: the user is told ONCE per session, ever (#96870 — out-of-band,
|
||||
# delivery channel only, never injected into conversation context).
|
||||
_empty_heal_user_notified: set = set()
|
||||
# One-shot pending notices keyed by session, drained via ``consume_pending_sanitizer_heal_notice``
|
||||
# and delivered via the status/warning callback.
|
||||
@@ -2374,12 +2523,31 @@ def _drop_invalid_roles(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
|
||||
def _drop_empty_tool_calls_arrays(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""Strict providers 400 on ``tool_calls: []``; normalize on shallow copies so history stays byte-stable."""
|
||||
# --- Drop empty / malformed tool_calls arrays on assistant messages --- An assistant message carrying
|
||||
# ``tool_calls: []`` (an empty array) — or a non-list value under the key — is semantically identical to
|
||||
# an assistant message with no tool calls, but strict OpenAI-compatible providers reject the empty array
|
||||
# outright: DeepSeek v4 returns HTTP 400 "Invalid 'messages[N].tool_calls': empty array. Expected an
|
||||
# array with minimum length 1, but got an empty array instead." (#58755, follow-up to #56980). Empty
|
||||
# arrays reach here from session resume, host-fed histories, or the consecutive-assistant merge in
|
||||
# ``repair_message_sequence`` (which preserves a pre-existing ``[]`` on the surviving turn). This is the
|
||||
# final pre-API chokepoint, so normalize defensively — and, per the #56980 review, do it HERE on the
|
||||
# per-call copy rather than in ``repair_message_sequence``, which would destructively rewrite the
|
||||
# persisted trajectory. Shallow-copy the message before dropping the key so stored history (and prompt
|
||||
# caching) stays byte-stable.
|
||||
normalized: List[Dict[str, Any]] = []
|
||||
dropped = 0
|
||||
for msg in messages:
|
||||
if (
|
||||
isinstance(msg, dict)
|
||||
and msg.get("role") == "assistant"
|
||||
# Defense-in-depth: a strict OpenAI-compatible provider (e.g. onerouter / Qwen, DeepSeek v4)
|
||||
# rejects an assistant message carrying ``tool_calls: []`` (empty array) with HTTP 400 "Empty
|
||||
# tool_calls is not supported in message." The pre-API sanitizer in agent_runtime_helpers drops
|
||||
# these, but only on the conversation_loop path — other routes can reach the wire without it.
|
||||
# For every request that serializes through this transport (conversation loop and any caller
|
||||
# using it), this is the last boundary, so normalize here. Requests built by fully separate
|
||||
# payload paths (e.g. some auxiliary clients) never pass through this layer and are out of scope
|
||||
# for it. (#58755 follow-up)
|
||||
and "tool_calls" in msg
|
||||
and not (isinstance(msg["tool_calls"], list) and msg["tool_calls"])
|
||||
):
|
||||
@@ -2444,6 +2612,22 @@ def _pair_tool_calls_positionally(messages: List[Dict[str, Any]]) -> List[Dict[s
|
||||
"""Positional tool_call <-> tool_result pairing: strict providers (DeepSeek v4, Kimi) require
|
||||
results IMMEDIATELY after their call. Drops positional orphans, stubs unanswered declared
|
||||
ids; matching is alias-aware."""
|
||||
# --- Positional tool_call <-> tool_result pairing --- Strict OpenAI-compatible providers (DeepSeek v4,
|
||||
# Kimi) enforce the POSITIONAL invariant: an assistant message carrying tool_calls must be IMMEDIATELY
|
||||
# followed by tool messages covering every tool_call_id. The previous implementation compared global id
|
||||
# sets, which misses the failure mode where a result exists somewhere in the transcript but not in the
|
||||
# run right after its call — an interrupted turn or a compression window can displace a result past a
|
||||
# user turn. The id then survives in the global result set, so the call looks answered, no stub is
|
||||
# injected, and the provider rejects the payload with HTTP 400 "An assistant message with 'tool_calls'
|
||||
# must be followed by tool messages responding to each 'tool_call_id' (insufficient tool messages
|
||||
# following tool_calls message)". Rewritten as a single rolling walk on the per-call copy (#94704): (a)
|
||||
# tool results that do not immediately follow an assistant message declaring their id are dropped
|
||||
# (positional orphans — includes results appearing BEFORE their call, which strict providers also
|
||||
# reject); (b) declared ids not covered by the immediately-following tool run get a stub result injected
|
||||
# at the end of that run, even when a mispositioned result exists elsewhere. Matching is variant-aware
|
||||
# (``tool_call_id_variants`` / ``tool_result_id_variants``): a result keyed on ANY alias spelling
|
||||
# (``id`` / ``call_id`` / ``response_item_id`` / composite bridge) answers the call, preserving the
|
||||
# unified alias policy from #55626/#63000/#93251.
|
||||
paired: List[Dict[str, Any]] = []
|
||||
declared_calls: Dict[str, tuple] = {}
|
||||
dropped = 0
|
||||
@@ -2504,6 +2688,24 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
|
||||
(not ids ever seen) because llama.cpp reuses one constant id, and whole variant groups so
|
||||
alias-keyed results are not deleted."""
|
||||
outstanding: Dict[str, int] = {} # every alias of an unanswered call -> its group id
|
||||
# 3. Deduplicate tool_call_ids. Strict providers (DeepSeek) reject a payload where the same tool_call_id
|
||||
# appears more than once with HTTP 400 "Duplicate value for 'tool_call_id'" (#58327). Duplicates can
|
||||
# arise from retries, crash/resume glitches, or a compression window that re-emits a tool result. This
|
||||
# is the final pre-API chokepoint, so dedup defensively here even though repair_message_sequence also
|
||||
# consumes matched ids. (a) collapse duplicate tool_calls WITHIN an assistant message (b) drop tool
|
||||
# results that answer no OUTSTANDING tool call (b) tracks outstanding calls rather than every id ever
|
||||
# seen, because ``tool_call_id`` is NOT globally unique in practice: llama.cpp emits a single constant
|
||||
# id for every tool call it ever returns (verified: three separate completions from one server all
|
||||
# carry the same id). A seen-once-drop-forever rule reads the SECOND legitimate tool result of such a
|
||||
# session as a duplicate and deletes it, so from the second tool call onward the model never sees any
|
||||
# result — it announces its next action and the turn dies with the work unfinished. Outstanding-call
|
||||
# semantics keep both protections intact: a re-emitted result still answers no pending call and is
|
||||
# still dropped, while a genuine new call that reuses the id re-arms that id first. Variant-group
|
||||
# tracking: answering or deduping one spelling consumes its siblings too. A Codex/Responses tool_call
|
||||
# registers ``id`` (fc_...), ``call_id`` (call_...), ``response_item_id``, and composite spellings
|
||||
# (#55626/#58168/#63000); tracking only the coalesced id here made a result keyed on any OTHER variant
|
||||
# look like it answered no outstanding call, so this pass deleted the very result step 2's
|
||||
# variant-aware matching had just preserved (issue #93251 — whole parallel batches vanished).
|
||||
outstanding_groups: Dict[int, frozenset] = {}
|
||||
next_group_id = 0
|
||||
deduped: List[Dict[str, Any]] = []
|
||||
@@ -2536,6 +2738,9 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
|
||||
continue
|
||||
if candidate_groups:
|
||||
# Consume EVERY variant of the matched call; ids are re-armed by the next call reusing them.
|
||||
# Consume the whole alias group so a SECOND result replaying any sibling spelling falls into
|
||||
# the drop branch below — strict providers reject duplicate tool_call_ids with HTTP 400
|
||||
# (#58327, #66974). Credit: #55436.
|
||||
group_id = min(candidate_groups)
|
||||
for variant in outstanding_groups.pop(group_id, frozenset()):
|
||||
if outstanding.get(variant) == group_id:
|
||||
@@ -2552,6 +2757,22 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
|
||||
def _realign_tool_result_names(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""Align each tool result's wire ``name`` with its call's function name (per-call copy only):
|
||||
Google 400s on a mismatch, routine when tool_search bridges via ``tool_call``."""
|
||||
# 4. Google matches functionResponse.name against functionCall.name and rejects a mismatch with HTTP 400
|
||||
# "Request contains an invalid argument" (INVALID_ARGUMENT); behind an OpenAI-compatible gateway that
|
||||
# surfaces only as a generic "Provider returned error". When tool_search defers MCP/plugin tools the
|
||||
# model calls the bridge tool ``tool_call``, while ``make_tool_result_message()`` labels the result
|
||||
# with the unwrapped internal tool name (``mcp__github__create_issue``) that dispatch, hooks, logging,
|
||||
# and guardrails need. #72089 fixed exactly this for the native Gemini adapter, which now prefers
|
||||
# ``tool_name_by_call_id`` over the result name; requests that reach Gemini through the
|
||||
# OpenAI-compatible path (OpenRouter, Vertex/LiteLLM proxies, any OpenAI-shaped gateway) skip that
|
||||
# translation entirely and still send the internal name on the wire. Normalizing here rather than in
|
||||
# the OpenAI-compat serializer keeps it provider-agnostic: Gemini reaches Hermes under many model
|
||||
# strings and base URLs, so sniffing for "is this really Google?" is unreliable, and every other
|
||||
# provider either ignores the field or agrees with the call name. Runs on the per-call copy, so the
|
||||
# stored trajectory keeps the real tool name for the session DB and the UI — only the wire payload
|
||||
# changes. A no-op for the native Gemini path, which already resolves the same name. A result whose
|
||||
# assistant call frame is missing entirely never reaches here — pass 1 above drops it as an orphan —
|
||||
# so the only results this pass sees are ones whose call name is knowable.
|
||||
call_names: Dict[str, str] = {}
|
||||
for msg in messages:
|
||||
if msg.get("role") == "assistant":
|
||||
@@ -2688,7 +2909,14 @@ def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int:
|
||||
"""Re-pad or strip assistant turns' reasoning_content for the CURRENT provider after a
|
||||
fallback switch: ``api_messages`` is shaped for the primary; require-side providers
|
||||
(DeepSeek/Kimi/MiMo) 400 without the pad, strict ones (Mistral, Cerebras, Groq) 400/422
|
||||
with it. Idempotent; returns the number of assistant turns changed."""
|
||||
with it. Idempotent; returns the number of assistant turns changed.
|
||||
|
||||
* Switching TO a strict provider that rejects the field (Mistral, Cerebras, Groq, SambaNova, …):
|
||||
assistant turns built under a reasoning primary carry a ``reasoning_content`` pad (often a single space
|
||||
``" "``), and the strict provider rejects it with HTTP 400/422 ("Extra inputs are not permitted"). This
|
||||
is the exact cross-provider fallback bug from #45655 — a DeepSeek primary pads history with ``" "``, the
|
||||
request falls back to Mistral, and Mistral 422s on the stale pad.
|
||||
"""
|
||||
from agent.message_sanitization import reapply_reasoning_echo
|
||||
return reapply_reasoning_echo(api_messages, agent._needs_thinking_reasoning_pad())
|
||||
|
||||
@@ -2701,7 +2929,11 @@ def _iter_httpx_pools_with_owner(http_client: Any):
|
||||
``owner`` is ``None`` for a pool this client owns outright, or the ``_SharedTransport`` view
|
||||
id when the pool is process-shared with other clients
|
||||
(``process_bootstrap.build_keepalive_http_client``). Callers must then touch only the
|
||||
in-flight requests stamped with that owner."""
|
||||
in-flight requests stamped with that owner.
|
||||
|
||||
Walking the default transport alone makes ``force_close_tcp_sockets`` return 0 while a stream is still
|
||||
mid-recv — the interrupt logs success and the provider keeps burning the slot (#72975).
|
||||
"""
|
||||
seen_pools: set[int] = set()
|
||||
try:
|
||||
transports = [getattr(http_client, "_transport", None)]
|
||||
|
||||
@@ -568,6 +568,17 @@ def build_anthropic_kwargs(
|
||||
kwargs["tool_choice"] = _TOOL_CHOICE_MAP.get(tool_choice) or {
|
||||
"type": "tool", "name": to_wire(tool_choice) if to_wire else tool_choice
|
||||
}
|
||||
# Map reasoning_config to Anthropic's thinking parameter. Claude 4.6+ models use adaptive thinking +
|
||||
# output_config.effort. Older models use manual thinking with budget_tokens. MiniMax Anthropic-compat
|
||||
# endpoints support thinking (manual mode only, not adaptive). Haiku does NOT support extended thinking
|
||||
# — skip entirely. Kimi / Moonshot models also use adaptive thinking: their Anthropic-compatible
|
||||
# endpoints (api.moonshot.cn/anthropic, api.kimi.com/coding) accept ``thinking.type="adaptive"`` +
|
||||
# ``output_config.effort``, and the replay-validation 400s that originally motivated dropping the
|
||||
# parameter (#13848) no longer occur. (Kimi on chat_completions enables thinking via extra_body in the
|
||||
# ChatCompletionsTransport — see #13503.) On 4.7+ the `thinking.display` field defaults to "omitted",
|
||||
# which silently hides reasoning text that Hermes surfaces in its CLI. We request "summarized" so the
|
||||
# reasoning blocks stay populated — matching 4.6 behavior and preserving the activity-feed UX during
|
||||
# long tool runs.
|
||||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
kwargs.update(_thinking_kwargs(reasoning_config, model, effective_max_tokens))
|
||||
# Safety net so upstream 4.6 -> 4.7 migrations don't need coordinated edits everywhere callers
|
||||
@@ -638,6 +649,9 @@ def _stream_final_message(stream_fn, api_kwargs, log_prefix, on_stream_event, on
|
||||
try:
|
||||
on_stream_event(event)
|
||||
except TimeoutError:
|
||||
# The callback is the caller's deadline seam (#99692: the host waiting on this summary has
|
||||
# already given up). Abandon the stream — the ``with`` closes it — instead of streaming an
|
||||
# answer nobody will read.
|
||||
raise
|
||||
except Exception:
|
||||
logger.debug("%son_stream_event callback failed", log_prefix, exc_info=True)
|
||||
|
||||
@@ -76,7 +76,12 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool:
|
||||
"""DeepSeek's ``/anthropic`` route. In thinking mode DeepSeek requires prior-turn ``thinking``
|
||||
blocks to round-trip while the generic third-party path strips them; its blocks are unsigned,
|
||||
so it gets the same strip-signed / keep-unsigned policy as Kimi. Pinned to the ``/anthropic``
|
||||
path so the OpenAI-compatible base URL is not misclassified."""
|
||||
path so the OpenAI-compatible base URL is not misclassified.
|
||||
|
||||
Per DeepSeek's published compatibility matrix the blocks are unsigned (no Anthropic-proprietary
|
||||
signature, no ``redacted_thinking`` support), so this endpoint is handled with the same strip-signed /
|
||||
keep-unsigned policy used for Kimi's ``/coding`` endpoint. See hermes-agent#16748.
|
||||
"""
|
||||
return base_url_host_matches(base_url or "", "api.deepseek.com") and "/anthropic" in _normalized_lower(base_url)
|
||||
|
||||
|
||||
|
||||
@@ -109,6 +109,7 @@ def normalize_model_name(model: str, preserve_dots: bool = False) -> str:
|
||||
if model.lower().startswith("anthropic/"):
|
||||
model = model[len("anthropic/"):]
|
||||
if not preserve_dots and not _is_bedrock_model_id(model) and model.lower().startswith(("claude-", "anthropic/")):
|
||||
# Only convert dots to hyphens for Anthropic/Claude models. See issue #17171.
|
||||
model = model.replace(".", "-")
|
||||
return model
|
||||
|
||||
@@ -150,6 +151,8 @@ def convert_tools_to_anthropic(tools: List[Dict]) -> List[Dict]:
|
||||
for t in tools or []:
|
||||
fn = t.get("function", {})
|
||||
name = fn.get("name", "")
|
||||
# Defensive dedup: Anthropic rejects requests with duplicate tool names. Upstream injection paths
|
||||
# already dedup, but this guard converts a hard API failure into a warning. See: #18478
|
||||
if name and name in seen_names:
|
||||
logger.warning("convert_tools_to_anthropic: duplicate tool name '%s' — dropping second occurrence", name)
|
||||
continue
|
||||
@@ -370,6 +373,12 @@ def _replay_ordered_blocks(m: Dict[str, Any], ordered_blocks: List[Any]) -> Opti
|
||||
def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Assistant message -> Anthropic content blocks (thinking, text, tool_use, Kimi/DeepSeek
|
||||
reasoning_content injection)."""
|
||||
# apply_anthropic_cache_control marks an assistant turn with non-empty text by writing cache_control
|
||||
# INTO ``content`` (see _apply_cache_marker's list branch), not at the top level. This branch rebuilds
|
||||
# the message from ordered_blocks and never reads ``content``, so that marker would be dropped -- and
|
||||
# because _can_carry_marker already counted this message as a carrier, the breakpoint is burned rather
|
||||
# than relocated. #56195 covered the complementary shape (blank content -> top-level marker); this is
|
||||
# the interleaved thinking + preamble-text + tool_use shape.
|
||||
content = m.get("content", "")
|
||||
ordered_blocks = m.get("anthropic_content_blocks")
|
||||
if isinstance(ordered_blocks, list) and ordered_blocks:
|
||||
@@ -395,6 +404,8 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
# (injected as a fallback upstream). Prepend, since thinking must precede text/tool_use. Skip
|
||||
# when reasoning_details already supplied (signed) thinking blocks: a duplicate unsigned one
|
||||
# would be downgraded to a spurious text block on the last assistant message.
|
||||
# See hermes-agent#13848. Accept empty string "" — _copy_reasoning_content_for_api() injects "" as a
|
||||
# tier-3 fallback for Kimi tool-call messages that had no reasoning.
|
||||
reasoning_content = m.get("reasoning_content")
|
||||
if isinstance(reasoning_content, str) and not _has_block_type(blocks, _THINKING_TYPES):
|
||||
blocks.insert(0, {"type": "thinking", "thinking": reasoning_content})
|
||||
@@ -598,7 +609,13 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
|
||||
"""Anthropic requires messages[0].role == user; prepend a placeholder turn otherwise. A second
|
||||
auto-compaction can leave a role=assistant summary first, which the API rejects (often masked
|
||||
as a misleading tool_use/tool_result 400). The filler must be non-whitespace text or it trades
|
||||
that 400 for the blank-block one."""
|
||||
that 400 for the blank-block one.
|
||||
|
||||
The inserted text block must be non-whitespace: Anthropic separately rejects any text content block
|
||||
whose text is empty or whitespace-only ("text content blocks must contain non-whitespace text"), so a
|
||||
single space here traded the "leading assistant turn" 400 for that one (#69512 class). Uses the same
|
||||
placeholder as every other synthesized filler block in this module for consistency.
|
||||
"""
|
||||
if result and result[0].get("role") != "user":
|
||||
result.insert(0, {"role": "user", "content": [_text_block(_EMPTY_TEXT_PLACEHOLDER)]})
|
||||
|
||||
|
||||
@@ -57,6 +57,15 @@ class ApiErrorSummaryMixin:
|
||||
the pool. xAI returns the same permission-denied text for BOTH cases; a ``[WKE=unauthenticated:...]``
|
||||
suffix (or "access token could not be validated") means stale token → return False so the refresh path
|
||||
runs.
|
||||
|
||||
Disambiguator for xAI (#29344): the same ``code`` text ("The caller does not have permission to
|
||||
execute the specified operation") is returned for BOTH an unsubscribed account AND a stale OAuth
|
||||
access token. xAI ships an explicit signal in the ``error`` field that tells the two apart: a
|
||||
``[WKE=unauthenticated:...]`` suffix (and/or the ``OAuth2 access token could not be validated``
|
||||
phrasing) means the credentials failed validation — that's recoverable by refreshing the token, NOT
|
||||
by surfacing an entitlement message. When either signal is present we return False eagerly so the
|
||||
credential-pool refresh path runs, letting long-running TUI sessions recover from stale tokens
|
||||
without an exit/reopen cycle.
|
||||
"""
|
||||
if status_code not in {401, 403, None}:
|
||||
return False
|
||||
@@ -162,6 +171,9 @@ class ApiErrorSummaryMixin:
|
||||
|
||||
# SDK may leave body empty while httpx has the payload. Redact: the body is attacker-influenced
|
||||
# and may echo Authorization / x-api-key / request JSON.
|
||||
# Redact before returning: the raw provider/proxy error body is attacker-influenced and may echo
|
||||
# Authorization / x-api-key / request JSON, which would otherwise leak into final_response + logs
|
||||
# (this path widens exposure vs the old empty-body "HTTP 400" string). See #36109.
|
||||
response = getattr(error, "response", None)
|
||||
if response is not None:
|
||||
try:
|
||||
|
||||
+214
-3
@@ -177,6 +177,12 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any:
|
||||
_apply_required_codex_headers(kwargs, access_token=api_key, base_url=base_url)
|
||||
# Hermes owns aux retry/fallback policy; the SDK default (max_retries=2) would triple
|
||||
# wall time on a hung endpoint before Hermes sees one failure.
|
||||
# Hermes owns auxiliary retry + provider/model fallback policy (the same-provider transient retry in
|
||||
# call_llm plus the except-chain fallback). The OpenAI SDK's own default (max_retries=2 → up to 3
|
||||
# attempts) silently multiplies the effective wall time of every aux call by 3× on a slow/hung endpoint,
|
||||
# so a 120s timeout can stall ~360s before Hermes sees a single failure (issue #54465). Disable
|
||||
# SDK-internal retries by default and let Hermes control the budget; explicit callers can still override
|
||||
# via kwargs.
|
||||
kwargs.setdefault("max_retries", 0)
|
||||
return OpenAI(api_key=api_key, base_url=base_url, **kwargs)
|
||||
|
||||
@@ -184,6 +190,13 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any:
|
||||
# Interrupt protection for atomic aux tasks: a compression summary killed by an ordinary
|
||||
# gateway interrupt degrades to a static marker, so a thread-local flag marks such calls
|
||||
# protected. Explicit host cancel (Ctrl+C, /stop) still overrides it, timeouts still fire.
|
||||
# ── Interrupt protection for atomic auxiliary tasks ────────────────────── Some auxiliary tasks must NOT be
|
||||
# aborted mid-flight by a gateway interrupt (e.g. an incoming user message while the agent is busy). Context
|
||||
# compression is the prime case: if the summary LLM call is interrupted part-way, compression falls back to
|
||||
# a static "summary unavailable" marker and the real handoff is lost (#23975). A thread-local flag lets such
|
||||
# a task mark its in-flight LLM call as interrupt-protected; the Codex Responses stream's cancellation check
|
||||
# honors it. TIMEOUTS still fire (a hung call must die), and all OTHER aux tasks (vision, web_extract,
|
||||
# title_generation, …) remain freely interruptible.
|
||||
_aux_interrupt_protection = threading.local()
|
||||
|
||||
|
||||
@@ -279,6 +292,12 @@ _aux_provider_response = threading.local()
|
||||
# Absolute monotonic deadline of the waiting HOST. The stream's own ceiling
|
||||
# (_aux_stream_total_ceiling, >= the host's and started later) would otherwise leave an
|
||||
# orphaned stream still billing after every host-ceiling timeout.
|
||||
# Absolute wall-clock deadline (time.monotonic) of the HOST waiting for this auxiliary call, when it has one
|
||||
# (#99692). Liveness alone is not enough: a host also stops waiting at its own total ceiling, and the
|
||||
# streamed consumer below bounds itself only by _aux_stream_total_ceiling() — a budget derived from the aux
|
||||
# request timeout, which is >= the host ceiling for every configured value AND starts counting later. So the
|
||||
# stream that outlives its abandoned host is not an edge case; it is the guaranteed outcome of every
|
||||
# total-ceiling timeout.
|
||||
_aux_stream_deadline = threading.local()
|
||||
|
||||
|
||||
@@ -411,6 +430,12 @@ def aux_stream_deadline(deadline: Optional[float]):
|
||||
``None`` is a passthrough; re-entrant-safe. Host->worker return leg of the progress hook:
|
||||
without it the isolated provider daemon streams to its own ceiling after the host stopped
|
||||
waiting, billing a summary the commit fence refuses.
|
||||
|
||||
``8207862212`` releases the compression OWNER when the fence is cancelled, but the isolated provider
|
||||
daemon (:func:`_run_protected_sync_provider_call`) that holds the socket keeps streaming to its own
|
||||
``_aux_stream_total_ceiling`` budget — >= the host's ceiling by construction — billing an abandoned
|
||||
summary the commit fence is already guaranteed to refuse, and stacking one fresh orphan per turn on a
|
||||
session that compression never managed to shrink. See #99692.
|
||||
"""
|
||||
previous = getattr(_aux_stream_deadline, "value", None)
|
||||
_aux_stream_deadline.value = deadline if isinstance(deadline, (int, float)) else previous
|
||||
@@ -446,6 +471,9 @@ def _run_protected_sync_provider_call(callback: Callable[[dict[str, Any]], Any],
|
||||
dispatch_hook = getattr(_aux_dispatch, "hook", None)
|
||||
provider_response_hook = getattr(_aux_provider_response, "hook", None)
|
||||
host_deadline = _current_aux_stream_deadline()
|
||||
# #99692: the stream is consumed on the daemon below, and thread-locals do not cross that boundary — an
|
||||
# owner-thread-only deadline would leave the fix inert on exactly the path large-session compression
|
||||
# takes (protected call + hard-cancel source installed).
|
||||
provider_context = contextvars.copy_context()
|
||||
done = threading.Event()
|
||||
outcome: dict[str, Any] = {}
|
||||
@@ -1104,6 +1132,18 @@ class _CodexStreamGuard:
|
||||
self.total_timeout = total_timeout
|
||||
self._start = time.monotonic()
|
||||
self.no_progress_timeout = _AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS
|
||||
# Progress-aware stream deadlines (supersedes the old single absolute kill at ``total_timeout``).
|
||||
# Three regimes: 1. First token: the stream must produce its first substantive payload within
|
||||
# ``no_progress_timeout`` (60s default) or we fail fast and let the caller's normal retry/fallback
|
||||
# chain run — a dead (or keepalive-only zombie) Codex stream no longer holds the full 300s
|
||||
# compression budget before falling back (masoria report, Aug 2026: 3 stacked 300s waits -> 20+ min
|
||||
# stuck on "Summarizing"). 2. Streaming: every substantive event re-arms the deadline by
|
||||
# ``no_progress_timeout`` — a live stream is never killed by an absolute total, so a long reasoning
|
||||
# summary that is actually producing tokens completes instead of timing out at 300s and falling back
|
||||
# (#54915's original complaint, fixed properly). Keepalive/lifecycle frames do NOT re-arm, mirroring
|
||||
# the commit-fence progress gating (#96707). 3. Hard ceiling: an absolute backstop from
|
||||
# ``_aux_stream_total_ceiling`` (max(600s, 4x configured timeout) — the same bound the streamed
|
||||
# chat.completions path uses) so a pathological one-token-per-59s drip still terminates.
|
||||
if total_timeout is not None:
|
||||
self.no_progress_timeout = min(self.no_progress_timeout, float(total_timeout))
|
||||
self.hard_deadline = self._start + _aux_stream_total_ceiling(total_timeout)
|
||||
@@ -1193,6 +1233,9 @@ class _CodexStreamGuard:
|
||||
# FD-safe — ``close()`` releases the raw TLS fd while the owner's OpenSSL BIO still
|
||||
# caches it, the kernel recycles it (e.g. into a SQLite handle), and the owner's TLS
|
||||
# flush corrupts that file. The owner does the real close in its ``finally``.
|
||||
# This callback has two callers — ``_check_cancelled`` on the owning thread, and the daemon watchdog
|
||||
# ``threading.Timer``, which is a stranger thread. The owning thread performs the real close in the
|
||||
# ``finally`` below, which is where the FD release belongs. See #70773.
|
||||
self.timeout_release_pending.set()
|
||||
if threading.get_ident() == self._owner_tid:
|
||||
_close_quietly(self._client, "client close during timeout failed")
|
||||
@@ -1212,6 +1255,9 @@ class _CodexStreamGuard:
|
||||
# The aux client cache wraps this same client; drop the entry so the next aux call
|
||||
# doesn't reuse the dead transport and fail fast.
|
||||
try:
|
||||
# After we close the httpx transport above, the cache must drop that entry — otherwise the next
|
||||
# auxiliary call (compression retry, memory flush, etc.) reuses the dead client and fails fast
|
||||
# with a connection error. See issue #23432.
|
||||
_evict_cached_client_instance(self._client)
|
||||
except Exception:
|
||||
logger.debug("Codex auxiliary: cache eviction on timeout failed", exc_info=True)
|
||||
@@ -1227,6 +1273,8 @@ class _CodexStreamGuard:
|
||||
# interrupt (degraded fallback marker); explicit host cancel has its own exception.
|
||||
if _aux_interrupt_cancel_requested():
|
||||
raise AuxiliaryExplicitCancellation()
|
||||
# Explicit host cancellation has its own frozen exception; timeouts above still fire and other
|
||||
# aux tasks remain interruptible. See #23975.
|
||||
if is_interrupted() and not _aux_interrupt_protected():
|
||||
raise InterruptedError("Codex auxiliary Responses stream interrupted")
|
||||
except InterruptedError:
|
||||
@@ -1260,6 +1308,8 @@ class _CodexStreamGuard:
|
||||
# TTFP telemetry records every frame, but forward progress (compression commit fence,
|
||||
# no-progress window) counts only substantive payloads — keepalives must not re-arm,
|
||||
# so a zombie stream dies at the same window as a dead connection.
|
||||
# #93650: keep bulk wire-format payload out of the SDK's GIL-holding request transform on auxiliary
|
||||
# calls too.
|
||||
if _codex_event_has_content(_event):
|
||||
self.record_progress()
|
||||
self.saw_content.set()
|
||||
@@ -1289,6 +1339,15 @@ class _CodexCompletionsAdapter:
|
||||
def _build_responses_kwargs(self, kwargs: Dict[str, Any]) -> Tuple[Dict[str, Any], str, Any]:
|
||||
"""chat.completions kwargs → Responses API kwargs, ``(resp_kwargs, model, timeout)``; mirrors codex.py::build_kwargs."""
|
||||
from utils import base_url_host_matches
|
||||
# Separate system/instructions from replayable conversation messages, then route the rest through
|
||||
# the SINGLE shared chat->Responses converter used by the main agent transport
|
||||
# (agent/transports/codex.py). Maintaining a private conversion loop here let chat-style messages
|
||||
# with role="tool" leak straight into Responses input[] — which the Responses API rejects with
|
||||
# "Invalid value: 'tool'. Supported values are: 'assistant', 'system', 'developer', and 'user'."
|
||||
# (issue #5709, hit hard by flush_memories() / compression replaying real session history that
|
||||
# includes assistant tool_calls + role="tool" results). The shared converter encodes assistant tool
|
||||
# calls as `function_call` items and tool results as `function_call_output` items with a valid
|
||||
# call_id, so every Responses path normalizes tool history identically and cannot drift.
|
||||
from agent.codex_responses_adapter import _chat_messages_to_responses_input
|
||||
model = kwargs.get("model", self._model)
|
||||
host = str(getattr(self._client, "base_url", "") or "")
|
||||
@@ -1309,6 +1368,9 @@ class _CodexCompletionsAdapter:
|
||||
# Copilot binds replayed codex_message_items ids to a backend connection that doesn't
|
||||
# survive credential rotation (401 on replay) — same guard as build_kwargs. Aux calls
|
||||
# never send ``context_management`` (main-turn feature): no compaction checkpoint.
|
||||
# Auxiliary calls (context compression, flush_memories, MoA aggregation) go through this adapter
|
||||
# instead of agent/transports/codex.py's build_kwargs, so they need the same guard applied
|
||||
# independently. See #32716.
|
||||
input_items = _chat_messages_to_responses_input(
|
||||
replay_messages, is_github_responses=is_copilot, native_compaction_eligible=False
|
||||
)
|
||||
@@ -1374,6 +1436,9 @@ class _CodexCompletionsAdapter:
|
||||
# conversation (rotation-stable logical scope, else the physical session id). Skip the
|
||||
# key where the main transport does: xAI takes it in extra_body, GitHub opts out.
|
||||
try:
|
||||
# Reuse the Responses transport's single authoritative hash algorithm and session-scope
|
||||
# normalization so equivalent static prefixes route to the same cache bucket across modes,
|
||||
# without concentrating unrelated sessions into one shared bucket (see #78941).
|
||||
from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key
|
||||
from agent.transports.codex import _default_prompt_cache_retention_for_request
|
||||
if not (is_xai or is_github) and "prompt_cache_key" not in resp_kwargs:
|
||||
@@ -1471,6 +1536,11 @@ class _AsyncAuxiliaryClientBase:
|
||||
self.api_key = sync_wrapper.api_key
|
||||
self.base_url = sync_wrapper.base_url
|
||||
if hasattr(sync_wrapper, "_real_client"):
|
||||
# Mirror the sync wrapper's _real_client so cache eviction by leaf OpenAI client (e.g.
|
||||
# _close_client_on_timeout in #23482) drops this async entry too. Without this, sync and async
|
||||
# cache entries diverge on poisoning: the sync entry is evicted but the async entry keeps
|
||||
# reusing the closed transport, failing every subsequent async aux call with 'Connection error'
|
||||
# until the gateway restarts.
|
||||
self._real_client = sync_wrapper._real_client
|
||||
|
||||
|
||||
@@ -1587,6 +1657,9 @@ class _AnthropicCompletionsAdapter:
|
||||
# response_format: top-level gets the same translation as the extra_body form; when both
|
||||
# are present the extra_body form wins. Passthrough excludes ``reasoning``/``response_format``
|
||||
# (already TRANSLATED to native fields — raw would 400 on strict gateways) and ``_`` Hermes plumbing.
|
||||
# The adapter builds the Messages body from a fixed allow-list of kwargs, so before this an
|
||||
# unrecognized top-level kwarg was dropped on the floor: the request succeeded but the schema
|
||||
# contract silently became prompt compliance (#85626 review, point 2).
|
||||
top_level_response_format = kwargs.get("response_format")
|
||||
if top_level_response_format is not None:
|
||||
_translate_anthropic_response_format(anthropic_kwargs, top_level_response_format)
|
||||
@@ -2471,6 +2544,10 @@ def set_runtime_main(
|
||||
Context-local so concurrent gateway sessions don't clobber each other; legacy mirrors are
|
||||
updated for old readers. ``cache_scope`` is the rotation-stable logical cache scope,
|
||||
preferred over ``session_id`` for prompt_cache_key derivation.
|
||||
|
||||
``cache_scope`` is the rotation-stable logical cache scope (compression- lineage root —
|
||||
agent/prompt_cache_scope.py) resolved once per turn by turn_context; auxiliary Responses calls prefer it
|
||||
over ``session_id`` for prompt_cache_key derivation (#79017).
|
||||
"""
|
||||
runtime = {
|
||||
"provider": (provider or "").strip().lower(),
|
||||
@@ -2538,6 +2615,8 @@ def _resolve_custom_runtime() -> Tuple[Optional[str], Optional[str], Optional[st
|
||||
if base_url_host_matches(custom_base, "openrouter.ai"):
|
||||
return None, None, None # requested='custom' falls back to OpenRouter when unconfigured.
|
||||
# Local servers (Ollama, vLLM, ...) ignore auth but the SDK needs a non-empty key.
|
||||
# Use a placeholder key — the OpenAI SDK requires a non-empty string but local servers ignore the
|
||||
# Authorization header. Same fix as cli.py _ensure_runtime_credentials() (PR #2556).
|
||||
if not isinstance(custom_key, str) or not custom_key.strip():
|
||||
custom_key = "no-key-required"
|
||||
if not isinstance(custom_mode, str) or not custom_mode.strip():
|
||||
@@ -2931,6 +3010,7 @@ def _is_rate_limit_error(exc: Exception) -> bool:
|
||||
OpenAI's RateLimitError may omit .status_code — matched by class name. A generic 429 without
|
||||
billing keywords counts as a rate limit.
|
||||
"""
|
||||
# (PR #8023 pattern)
|
||||
if type(exc).__name__ == "RateLimitError":
|
||||
return True
|
||||
if getattr(exc, "status_code", None) != 429:
|
||||
@@ -3116,6 +3196,8 @@ def _is_invalid_aux_response_error(exc: Exception) -> bool:
|
||||
# Tasks on a user-visible critical path (compression blocks resuming an oversized session; vision
|
||||
# stalls the serialised turn queue). A same-provider retry after a full-budget timeout costs another
|
||||
# whole ``timeout`` window, so they skip straight to fallback; fast blips still retry.
|
||||
# Fast blips (a streaming-close or a 5xx) still retry, since those are cheap. See issue #54465 for the
|
||||
# compression case.
|
||||
_TIMEOUT_NO_RETRY_TASKS = frozenset({"compression", "vision"})
|
||||
|
||||
|
||||
@@ -3294,6 +3376,10 @@ def _prepare_same_provider_retry(
|
||||
)
|
||||
# Preserve per-request attribution headers (e.g. Copilot ``x-initiator``) so the retry keeps capability gating.
|
||||
if extra_headers:
|
||||
# Copilot's ``x-initiator: user``) across the rebuilt-client retry — dropping them here would let a
|
||||
# recovery retry silently lose capability gating (#60293).
|
||||
# Preserve per-request attribution headers across the rebuilt-client retry — see the sync variant
|
||||
# above (#60293).
|
||||
retry_kwargs["extra_headers"] = dict(extra_headers)
|
||||
if _is_anthropic_compat_endpoint(resolved_provider, retry_base):
|
||||
retry_kwargs["messages"] = _convert_openai_images_to_anthropic(retry_kwargs["messages"])
|
||||
@@ -3432,7 +3518,13 @@ def _coerce_positive_timeout(raw: Any) -> Optional[float]:
|
||||
|
||||
def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[float]:
|
||||
"""Per-entry ``timeout`` for a configured fallback candidate, or None (keep the task-level
|
||||
timeout). Inheriting the primary's deadline used to kill healthy-but-slower fallbacks."""
|
||||
timeout). Inheriting the primary's deadline used to kill healthy-but-slower fallbacks.
|
||||
|
||||
A fallback candidate previously inherited the exact timeout the primary provider was called with. When
|
||||
that deadline was tuned for the primary (or the primary simply consumed its whole budget before failing
|
||||
over), the fallback aborted on the same clock even when independently healthy — a 163k-token compression
|
||||
that needs ~90s on the fallback died at the primary's 30s deadline every turn (#62452).
|
||||
"""
|
||||
entry = _fallback_chain_entry(task, fb_label)
|
||||
return _coerce_positive_timeout(entry.get("timeout") if entry else None)
|
||||
|
||||
@@ -3598,7 +3690,12 @@ def _call_fallback_candidate_sync(
|
||||
) -> Optional[Any]:
|
||||
"""Call one fallback candidate with stale-credential recovery: on an auth error refresh its
|
||||
credentials and retry once with a rebuilt client; if that also auth-fails, quarantine the
|
||||
provider and return None so the caller moves on. Non-auth errors raise."""
|
||||
provider and return None so the caller moves on. Non-auth errors raise.
|
||||
|
||||
``effective_timeout`` is the task-level deadline; a configured-chain candidate with its own ``timeout``
|
||||
entry gets that instead, so a fallback tuned differently from the primary is allowed its own budget
|
||||
(#62452).
|
||||
"""
|
||||
destination, fb_kwargs, rebuild = _plan_fallback_candidate(
|
||||
fb_client, fb_model, fb_label, task=task, effective_timeout=effective_timeout,
|
||||
apply_fast_lane=True, messages=messages, tools=tools, temperature=temperature,
|
||||
@@ -3751,6 +3848,17 @@ def _try_main_agent_model_fallback(
|
||||
# too-small aux models; runtime chains must too, or compression stops at a reachable-but-too-small
|
||||
# candidate. ``None`` (unknown) passes through.
|
||||
|
||||
# ── Context-window screening for runtime fallback chains (issue #52392) ── When the runtime auxiliary
|
||||
# fallback chain selects a candidate that is reachable but has a context window smaller than the compression
|
||||
# task requires, the call errors out instead of continuing to the next, viable candidate. The startup
|
||||
# feasibility check in ``agent.conversation_compression.check_compression_model_feasibility`` already
|
||||
# filters too-small auxiliary models at startup, but the runtime fallback chain
|
||||
# (``_try_configured_fallback_chain`` and ``_try_main_fallback_chain``) does not apply the same filter, so
|
||||
# compression can stop at the first alive door even if the room behind it is too small. The helpers below
|
||||
# screen each candidate by its effective context window before it is returned. ``None`` results from
|
||||
# ``get_model_context_length`` are passed through (we cannot prove a model is too small, so we do not block
|
||||
# it). This preserves the existing fallback surface for unrecognised/custom models while closing the gap on
|
||||
# the well-known ones.
|
||||
def _task_minimum_context_length(task: Optional[str]) -> Optional[int]:
|
||||
"""Minimum context length for an auxiliary task; None = no floor (only ``compression`` has one)."""
|
||||
return MINIMUM_CONTEXT_LENGTH if task == "compression" else None
|
||||
@@ -3991,6 +4099,7 @@ def _try_main_provider_route(
|
||||
explicit_base_url = None
|
||||
elif runtime_base_url:
|
||||
# Config-less named custom provider (live runtime only): anonymous custom arm + runtime key.
|
||||
# See #34777.
|
||||
resolved_provider = "custom"
|
||||
explicit_api_key = runtime_api_key or None
|
||||
elif runtime_api_key:
|
||||
@@ -4121,6 +4230,7 @@ def _to_async_client(sync_client, model: str, is_vision: bool = False):
|
||||
_apply_required_codex_headers(async_kwargs, access_token=sync_client.api_key, base_url=sync_base_url)
|
||||
async_kwargs = {**_openai_http_client_kwargs(sync_base_url, async_mode=True), **async_kwargs}
|
||||
# Hermes owns the auxiliary retry/timeout budget; disable SDK-internal retries.
|
||||
# See #54465.
|
||||
async_kwargs.setdefault("max_retries", 0)
|
||||
return AsyncOpenAI(**async_kwargs), model
|
||||
|
||||
@@ -4187,6 +4297,7 @@ def _build_bedrock_client(provider: str, model: Optional[str], *, raw_codex: boo
|
||||
return None, None
|
||||
# Region must match the main runtime's resolution (bedrock.region in config first, then
|
||||
# env/profile) so aux calls never leave the primary runtime's configured region.
|
||||
# See #53880, #65076.
|
||||
region = resolve_bedrock_runtime_region()
|
||||
default_model = "anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
final_model = _normalize_resolved_model(model or default_model, provider) or default_model
|
||||
@@ -4406,6 +4517,8 @@ def _resolve_custom_branch(req: _ResolveRequest) -> _ResolveResult:
|
||||
elif main_runtime:
|
||||
# Reuse main_runtime's concrete base_url + api_key for a named custom provider;
|
||||
# re-resolving from bare "custom" loses the name and lands on the wrong provider.
|
||||
# Re-resolution loses the provider name and falls back to OpenRouter or a wrong API-key provider —
|
||||
# the main agent already solved this, we just need to reuse its answer. (#45472)
|
||||
_main_base = str(main_runtime.get("base_url") or "").strip().rstrip("/")
|
||||
_main_key = str(main_runtime.get("api_key") or "").strip()
|
||||
if _main_base and _main_key:
|
||||
@@ -4487,6 +4600,7 @@ def _resolve_named_custom_branch(req: _ResolveRequest) -> Optional[_ResolveResul
|
||||
provider, final_model, entry_api_mode or "chat_completions")
|
||||
# anthropic_messages: route via AnthropicAuxiliaryClient (mirrors _try_custom_endpoint);
|
||||
# the Anthropic SDK sees the original (un-rewritten) URL.
|
||||
# Mirrors the anonymous-custom branch in _try_custom_endpoint(). See #15033.
|
||||
if entry_api_mode == "anthropic_messages":
|
||||
try:
|
||||
from agent.anthropic_adapter import build_anthropic_client
|
||||
@@ -4710,6 +4824,29 @@ def resolve_provider_client(
|
||||
# Excluded: ``auto`` (a stale main slug could pair with any picked provider) and Nous + vision (the
|
||||
# Portal's tier-aware vision recommendation must win over a text-only model).
|
||||
if not model and provider != "auto" and not (provider == "nous" and is_vision):
|
||||
# ``auto`` is intentionally excluded: `_resolve_auto(main_runtime=...)` returns the model paired
|
||||
# with the provider it actually selected. Pre-filling an auto call from `_read_main_model()` can
|
||||
# leak a stale process-global runtime into a different provider (for example Claude model slug on
|
||||
# Codex OAuth) and override that correctly resolved model. 1. ``model`` argument (caller knew what
|
||||
# they wanted) 2. Provider's catalog default — cheap/fast model the provider registered via
|
||||
# ``ProviderProfile.default_aux_model`` or the legacy ``_API_KEY_PROVIDER_AUX_MODELS_FALLBACK``
|
||||
# dict. 3. User's main model from ``model.model`` in config.yaml. This is the load-bearing step for
|
||||
# OAuth providers: an xai-oauth user with grok-4.3 configured gets grok-4.3 for title generation
|
||||
# instead of silently dropping to whatever Step-2 fallback (#31845). When the main provider is MoA,
|
||||
# ``_read_main_model_for_aux()`` substitutes the preset's aggregator model — the preset NAME is
|
||||
# never a valid wire model id, so unset aux models default to the preset's acting model instead.
|
||||
# Each provider branch below sees a non-empty ``model`` whenever the user has *anything* configured
|
||||
# — no provider-specific empty-model guards needed. When the user has NOTHING configured (fresh
|
||||
# install, main_model also empty), the branches still hit their own missing-credentials returns and
|
||||
# ``_resolve_auto`` falls through to the Step-2 chain as before. Do NOT pre-fill a blank ``auto``
|
||||
# request from the config/main default here. Claude model sent to Codex after the main lane fell
|
||||
# back to gpt-5.5). Let _resolve_auto() return the actual current runtime model when the caller did
|
||||
# not explicitly request one. (# compression-current-model) Nous + vision is the one carve-out: the
|
||||
# branch below resolves its model from the Portal's tier-aware vision recommendation
|
||||
# (``_try_nous(vision= True)``), and ``final_model = model or default`` means anything pre-filled
|
||||
# here wins over that. The main chat model is routinely text-only (e.g. a ``:free`` chat SKU), so
|
||||
# pre-filling it sends the image to a model that cannot accept one and the Portal 404s. Leave
|
||||
# ``model`` unset and let the Portal slot through; only an explicit caller model may override it.
|
||||
model = _get_aux_model_for_provider(provider) or _read_main_model_for_aux() or model
|
||||
req = _ResolveRequest(
|
||||
provider, original_provider, model, async_mode, raw_codex,
|
||||
@@ -4974,6 +5111,8 @@ def auxiliary_max_tokens_param(value: int, *, model: Optional[str] = None) -> di
|
||||
# Client cache: (provider, async_mode, base_url, api_key, api_mode, runtime_key) -> (client, default_model, loop)
|
||||
# Loop identity is NOT part of the key: stale-loop entries are replaced in place on async hits,
|
||||
# bounding growth to one entry per provider config (avoids fd accumulation in gateways).
|
||||
# This bounds cache growth to one entry per unique provider config rather than one per (config ×
|
||||
# event-loop), which previously caused unbounded fd accumulation in long-running gateway processes (#10200).
|
||||
_client_cache: Dict[tuple, tuple] = {}
|
||||
_client_cache_lock = threading.Lock()
|
||||
_CLIENT_CACHE_MAX_SIZE = 64 # safety belt — evict oldest when exceeded
|
||||
@@ -5056,6 +5195,12 @@ def _refresh_nous_auxiliary_client(
|
||||
to ``_get_cached_client`` when the stale client was acquired — so the fresh client overwrites
|
||||
the exact entry the stale one is served from. Keying on the resolved model or an empty task
|
||||
would leave the expired client immortal and every auxiliary call 401ing forever.
|
||||
|
||||
See #56889.
|
||||
For ``provider == "auto"`` the task participates in the cache key (task-specific fallback policy), so it
|
||||
MUST be carried into the key here for the same reason as ``lookup_model``; otherwise an auto-provider
|
||||
client refreshed on a 401 lands under the ``task=""`` key while the stale entry survives under the
|
||||
task-scoped key (#58894).
|
||||
"""
|
||||
runtime = _resolve_nous_runtime_api(force_refresh=True, stale_access_token=api_key)
|
||||
if runtime is None:
|
||||
@@ -5212,6 +5357,10 @@ def _get_cached_client(
|
||||
|
||||
Async clients bind to the loop they were created on, so every async hit validates the cached
|
||||
loop is the current, open loop; stale entries are replaced in place (bounded, no cross-loop reuse).
|
||||
|
||||
This keeps cache size bounded to one entry per unique provider config, preventing the fd-exhaustion that
|
||||
previously occurred in long-running gateways where recycled worker threads created unbounded entries
|
||||
(#10200).
|
||||
"""
|
||||
current_loop = _current_event_loop() if async_mode else None
|
||||
runtime = _normalize_main_runtime(main_runtime)
|
||||
@@ -5265,6 +5414,14 @@ def _get_cached_client(
|
||||
_AUX_DIRECT_API_BASE_URLS: Dict[str, str] = {"openai": "https://api.openai.com/v1"}
|
||||
|
||||
|
||||
# MoA virtual provider: an *explicit* `provider: moa` override (either the caller-passed `provider` arg or
|
||||
# `auxiliary.<task>.provider` in config.yaml) reaches this function directly — it never goes through
|
||||
# _resolve_auto(), which only unwraps the *implicit* "main provider is moa" case (#53827). Left as-is, "moa"
|
||||
# is returned verbatim and resolve_provider_client() looks it up in PROVIDER_REGISTRY (which has no "moa"
|
||||
# entry — it's not a real HTTP provider), falls to the unknown-provider dead end, and call_llm surfaces a
|
||||
# nonsensical "MOA_API_KEY environment variable" error for a provider that was never meant to be reached
|
||||
# over the wire. Auxiliary tasks don't need the reference fan-out — resolve to the preset's aggregator slot
|
||||
# instead, exactly like the implicit path does (shared helper: _resolve_moa_aggregator).
|
||||
def _unwrap_moa_provider(prov: str, mdl: Optional[str]) -> Tuple[str, Optional[str]]:
|
||||
"""Resolve an *explicit* ``provider: moa`` to its preset's aggregator slot (_resolve_auto()
|
||||
only unwraps the implicit case; "moa" isn't in PROVIDER_REGISTRY and would dead-end)."""
|
||||
@@ -5350,6 +5507,7 @@ def _resolve_task_provider_model(
|
||||
# An explicit provider without base_url adopts the task's configured endpoint (same or
|
||||
# unnamed provider) so the early return below carries it. Explicit "auto" is excluded — it
|
||||
# must keep flowing through auto-resolution.
|
||||
# See #58515.
|
||||
if provider and provider != "auto" and not base_url and cfg_base_url and cfg_provider in (None, provider):
|
||||
base_url = cfg_base_url
|
||||
if not api_key:
|
||||
@@ -5375,6 +5533,11 @@ _DEFAULT_AUX_TIMEOUT = 30.0
|
||||
# Reasoning compression models can exceed the default 120 s config timeout, falling back to the
|
||||
# deterministic marker. Bounded *floor* for config-derived compression timeouts only; never
|
||||
# overrides an explicit per-call timeout.
|
||||
# Compression summarises large conversation histories; a reasoning auxiliary model (e.g. Codex / GPT-5.5)
|
||||
# can legitimately take longer than the default ``auxiliary.compression.timeout`` (120 s), causing the
|
||||
# stream to time out and the compressor to fall back to the deterministic context marker (#54915). A floor
|
||||
# is harmless for fast compression models (they finish before the deadline) and is a minimum, so a higher
|
||||
# config value is kept unchanged.
|
||||
_COMPRESSION_TIMEOUT_FLOOR_SECONDS = 300.0
|
||||
|
||||
|
||||
@@ -5531,6 +5694,9 @@ def _get_task_extra_body(task: str) -> Dict[str, Any]:
|
||||
|
||||
# Per-task concurrency limiting: many sessions can spawn unbounded background aux calls, each
|
||||
# retrying across the fallback chain during incidents.
|
||||
# During provider incidents each call also retries / fans out across the fallback chain, multiplying request
|
||||
# volume on already-degraded endpoints. A per-task semaphore caps in-flight calls so retry amplification
|
||||
# stays bounded. See #23324.
|
||||
_aux_sync_semaphores: Dict[str, Tuple[int, threading.BoundedSemaphore]] = {}
|
||||
_aux_async_semaphores: Dict[Tuple[str, int], Tuple[int, Any]] = {}
|
||||
_aux_sem_lock = threading.Lock()
|
||||
@@ -5833,6 +5999,11 @@ def _validate_llm_response(
|
||||
|
||||
Also the single aux-usage accounting chokepoint: every successful non-streaming response
|
||||
passes here exactly once; *provider*/*base_url* are optional hints.
|
||||
|
||||
See #7264.
|
||||
Recording is best-effort and never affects validation. *provider*/*base_url* are optional accounting
|
||||
hints — fallback-path calls omit them and the row keeps the model (read from the response itself) with
|
||||
an empty route. See #23270.
|
||||
"""
|
||||
if response is None:
|
||||
raise RuntimeError(f"Auxiliary {task or 'call'}: LLM returned None response")
|
||||
@@ -6007,6 +6178,7 @@ _AFFORDABLE_TOKENS_RE = re.compile(r"can only afford\s+([0-9][0-9,]*)", re.IGNOR
|
||||
# Below the floor the affordable budget can't fit a useful aux output — treat as exhaustion;
|
||||
# the margin keeps provider-side token-count rounding from 402-ing the retry.
|
||||
_AFFORDABLE_RETRY_FLOOR_TOKENS = 512
|
||||
# See #49785.
|
||||
_AFFORDABLE_RETRY_MARGIN_TOKENS = 64
|
||||
|
||||
|
||||
@@ -6067,7 +6239,18 @@ def _create_with_progress_once(
|
||||
"""create() that streams (and re-aggregates, ticking the hook per substantive chunk) when a
|
||||
progress hook is active or the provider is stream-only; plain ``create(**kwargs)`` otherwise
|
||||
or when the adapter streams internally. Streaming rejections fall back to a plain call —
|
||||
except under ``force_stream``."""
|
||||
except under ``force_stream``.
|
||||
|
||||
Behavior is byte-for-byte identical to a plain ``create(**kwargs)`` when neither trigger applies (every
|
||||
existing caller/task) or when the client's wire adapter streams internally. With a hook + a
|
||||
chunk-capable client, the request is sent with ``stream=True`` and aggregated, ticking the hook only for
|
||||
substantive chunks. The configured ``timeout`` acts per stream read (idle) rather than as a total
|
||||
budget, and outer liveness watchdogs see tokens moving. ``force_stream=True`` (stream-only providers
|
||||
such as Tencent Copilot — credit @kudi88, PR #60686) takes the same streamed path even without a hook.
|
||||
Providers that reject the streamed request fall back to the plain non-streaming call — except under
|
||||
``force_stream``, where a stream-only provider rejects the plain call by definition, so the original
|
||||
error is surfaced to the normal recovery chains instead.
|
||||
"""
|
||||
_notify_aux_dispatch()
|
||||
_notify_aux_progress() # Preserve the watchdog's historical dispatch tick.
|
||||
if (not _aux_progress_active() and not force_stream) or _client_streams_internally(client):
|
||||
@@ -6141,6 +6324,9 @@ class _ChatStreamAccumulator:
|
||||
self._total_ceiling = total_ceiling
|
||||
# Absolute instant the waiting host gives up; checked alongside (not instead of) the
|
||||
# ceiling, and unaffected by pre-construction dispatch/TTFT.
|
||||
# Checked as well as (not instead of) the ceiling above: the ceiling still bounds callers with no
|
||||
# host deadline, and the host deadline is absolute, so it is unaffected by however long dispatch and
|
||||
# TTFT took before this accumulator was constructed. See #99692.
|
||||
self._host_deadline = host_deadline
|
||||
self.content_parts: List[str] = []
|
||||
self.reasoning_parts: List[str] = []
|
||||
@@ -6626,6 +6812,12 @@ def _ladder_provider_fallback(first_err: Exception, route: _LadderRoute):
|
||||
response) bypass the explicit-provider gate — the provider cannot serve this request
|
||||
regardless of user intent. Auth errors only fall back in auto mode."""
|
||||
task, tag, resolved_provider = route.task, route.tag, route.resolved_provider
|
||||
# Respect explicit provider choice for transient errors (auth, request validation, etc.) but allow
|
||||
# fallback when the provider clearly cannot serve the request due to capacity: payment/quota exhaustion
|
||||
# and connection failures are capacity problems, not request constraints. See #26803: daily token quota
|
||||
# (429 + "too many tokens per day") must fall back just like a 402 credit error.
|
||||
# Rate limits are included: after retries are exhausted, a 429 means the provider is at capacity. See
|
||||
# #52228. See #26803: daily token quota must fall back like a 402 credit error.
|
||||
is_auto = resolved_provider in {"auto", "", None}
|
||||
reason = next((label for predicate, label in _FALLBACK_REASONS if predicate(first_err)), None)
|
||||
is_capacity_error = any(
|
||||
@@ -6669,6 +6861,9 @@ def _ladder_provider_fallback(first_err: Exception, route: _LadderRoute):
|
||||
break
|
||||
# All fallback layers exhausted — one user-visible warning, then re-raise.
|
||||
logger.warning("Auxiliary %s%s: %s on %s and all fallbacks exhausted "
|
||||
# All fallback layers exhausted — emit a single user-visible warning so the operator
|
||||
# knows aux task is about to fail. (#26882) The error itself is re-raised below.
|
||||
# (#26882)
|
||||
"(fallback_chain + main agent model). Raising original error.",
|
||||
task or "call", tag, reason, resolved_provider)
|
||||
return None
|
||||
@@ -6705,6 +6900,10 @@ def _aux_recovery_ladder(
|
||||
return resp
|
||||
# Connection/timeout errors poison the cached client (closed transport, half-read
|
||||
# stream); evict so the next aux call rebuilds a fresh one.
|
||||
# Drop it from the cache regardless of whether we found a fallback above so the next auxiliary call
|
||||
# rebuilds a fresh client instead of reusing the dead one. See issue #23432.
|
||||
# Mirror the sync path: drop poisoned clients on connection/timeout so the next aux call rebuilds. See
|
||||
# issue #23432.
|
||||
if _is_connection_error(first_err):
|
||||
try:
|
||||
_evict_cached_client_instance(client)
|
||||
@@ -6924,6 +7123,17 @@ def _call_llm_impl(
|
||||
return _relay_sync_stream(client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode)
|
||||
|
||||
def _primary(**validate_kw: Any) -> Any:
|
||||
# Retry on the same provider for a transient transport blip (connection reset / streaming-close /
|
||||
# incomplete chunked read / 5xx / 408) before the except-chain below escalates to provider/model
|
||||
# fallback. A dropped connection shouldn't abandon an otherwise-healthy provider — this especially
|
||||
# matters for pinned auxiliary calls like MoA reference advisors, where "fallback to another
|
||||
# provider" is not a meaningful recovery (the advisor is a specific model), so a transient blip that
|
||||
# isn't retried simply loses that advisor for the turn (root of the run2 double-advisor "Connection
|
||||
# error" collapse — a genuine upstream blip hitting both parallel advisors at once). Attempts are
|
||||
# bounded and use exponential backoff. Count is configurable via auxiliary.transient_retries
|
||||
# (default 2 retries → 3 total attempts); a second/third failure or any non-transient error falls
|
||||
# through to ``first_err`` and the existing fallback handling unchanged. Unified home for the
|
||||
# transient retry every auxiliary task shares. (PR #16587)
|
||||
return _validate_llm_response(
|
||||
_relay_sync_completion(
|
||||
client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode,
|
||||
@@ -7087,6 +7297,7 @@ async def _async_call_llm_impl(
|
||||
client, kwargs, request_provider = req.client, req.kwargs, req.request_provider
|
||||
try:
|
||||
# Retry ONCE on the same provider for a transient blip before fallback (see call_llm()).
|
||||
# (PR #16587)
|
||||
_force_stream_async = (
|
||||
_provider_requires_stream(request_provider, req.base_info or req.resolved_base_url)
|
||||
and not isinstance(client, (
|
||||
|
||||
@@ -78,6 +78,9 @@ def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
(stranded failover). Same label = same configured credential; custom entries can each carry
|
||||
their own api_key, so a shared URL alone is only a weak signal when a label is missing."""
|
||||
if a.provider and b.provider:
|
||||
# Different labels = different credential config (first-class registry providers explicitly so —
|
||||
# #70893; custom entries can each carry their own api_key, so sameness is unprovable and we must not
|
||||
# skip).
|
||||
return a.provider == b.provider
|
||||
return bool(a.base_url and a.base_url == b.base_url)
|
||||
|
||||
@@ -100,6 +103,8 @@ def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
if not (a.provider and b.provider and a.provider == b.provider):
|
||||
return bool(
|
||||
a.base_url
|
||||
# Same-host different-label shims: same URL + same model IS the same deployment even when the
|
||||
# alias labels differ (#22548) — unless both labels are first-class registry providers (#70893).
|
||||
and a.base_url == b.base_url
|
||||
and a.model
|
||||
and a.model == b.model
|
||||
|
||||
@@ -119,11 +119,18 @@ def _interrupt_background_review(review_agent: Any) -> None:
|
||||
def cancel_background_review_for_live_turn(agent: Any) -> None:
|
||||
"""Cancel the current review and await its request-phase acknowledgement. Foreground priority:
|
||||
past the bounded deadline, warn and let the live turn proceed — self-improvement work must
|
||||
never block a user-facing turn."""
|
||||
never block a user-facing turn.
|
||||
|
||||
Foreground priority is preserved: if the review does not acknowledge within the bounded deadline, a
|
||||
warning is logged and the live turn proceeds anyway. See #84423.
|
||||
"""
|
||||
with _optional_lock(agent, "_background_review_lock"):
|
||||
run = getattr(agent, "_background_review_run", None)
|
||||
legacy_agent = getattr(agent, "_background_review_agent", None)
|
||||
review_agent = legacy_agent if run is None else run.cancel()
|
||||
# Attribute the review fork's usage to the PARENT session. Snapshot BEFORE unregister/close so counters
|
||||
# survive teardown. Placed in this finally so a fork that consumed tokens and THEN raised is still
|
||||
# attributed (issue #87250). Best-effort: the recorder never raises into the review thread.
|
||||
if review_agent is not None:
|
||||
_interrupt_background_review(review_agent)
|
||||
if run is None:
|
||||
@@ -594,7 +601,10 @@ def summarize_background_review_actions(
|
||||
skill-management tool results from the review agent's messages, skipping tool messages already
|
||||
present in ``prior_snapshot`` so inherited results are not re-surfaced as fresh work.
|
||||
``notification_mode``: ``off`` -> no actions; ``on`` -> generic "Memory updated"/tool messages;
|
||||
``verbose`` -> content previews from the tool-call arguments."""
|
||||
``verbose`` -> content previews from the tool-call arguments.
|
||||
|
||||
See #14944.
|
||||
"""
|
||||
mode = str(notification_mode or "on").lower()
|
||||
if mode == "off":
|
||||
return []
|
||||
@@ -814,6 +824,13 @@ def build_cache_parity_fork(
|
||||
# Same model only: share the warm cached system prompt (~26% cost cut; a rebuilt prompt misses
|
||||
# the byte-exact prefix key) and pin session_start so any re-render (compression, plugin
|
||||
# hooks) stays byte-identical.
|
||||
# Inherit the parent's cached system prompt verbatim so the review fork's outbound HTTP request hits the
|
||||
# same Anthropic/OpenRouter prefix cache the parent warmed. Without this, the fork rebuilds the system
|
||||
# prompt from scratch (fresh _hermes_now() timestamp, fresh session_id, narrower toolset → different
|
||||
# skills_prompt) and the byte-exact prefix-cache key misses. See issue #25322 and PR #17276 for the full
|
||||
# analysis + measured impact (~26% end-to-end cost reduction on Sonnet 4.5). When routed to a different
|
||||
# model the parent's cached prompt is for the wrong model/cache key and would miss anyway, so let the
|
||||
# routed fork build its own.
|
||||
if not _routed:
|
||||
review_agent._cached_system_prompt = agent._cached_system_prompt
|
||||
review_agent.session_start = agent.session_start
|
||||
@@ -824,6 +841,9 @@ def build_cache_parity_fork(
|
||||
return review_agent, _rt, _routed
|
||||
|
||||
|
||||
# Install a non-interactive approval callback on this worker thread so any dangerous-command guard the
|
||||
# review agent trips resolves to "deny" instead of falling back to input() -- which deadlocks against the
|
||||
# parent's prompt_toolkit TUI (#15216). Same pattern as _subagent_auto_deny in tools/delegate_tool.py.
|
||||
def _bg_review_auto_deny(command, description, **kwargs):
|
||||
"""Non-interactive approval: dangerous-command guards resolve to "deny" instead of input(),
|
||||
which would deadlock against the parent's TUI."""
|
||||
@@ -876,6 +896,21 @@ def _review_tool_whitelist(review_agent: Any, task_cfg: Optional[Dict[str, Any]]
|
||||
whitelist |= {"read_file", "search_files"}
|
||||
# ``extra_tools`` admits named parent tools (e.g. a human-gated proposal tool). The whitelist
|
||||
# can only admit, never advertise: a listed tool must already exist in the inherited schema.
|
||||
# Read-only file tools are whitelisted too (#61521, #39996): the model naturally reaches for
|
||||
# read_file/search_files to inspect a skill before patching it. Denying them caused a per-review denial
|
||||
# storm (~142 denials + ~204 read-before-write refusals over 2 days on one deployment) that starved the
|
||||
# self-improvement loop — the model never loaded SKILL.md the way the read-before-write guard requires,
|
||||
# so almost no patch landed. This is a DISPATCH-side change only: the advertised ``tools[]`` stays
|
||||
# byte-identical to the parent's, so prompt-cache parity is untouched. read_file registers the read with
|
||||
# the read-before-write guard (tools/file_tools.py), so a read_file → skill_manage(patch) sequence now
|
||||
# succeeds. Write tools (write_file/patch/terminal) stay denied — autonomous maintenance must go through
|
||||
# skill_manage's validation, and the deny message below names that substitute so one denial redirects
|
||||
# the model instead of a storm.
|
||||
# Profile-configured opt-in tools (#44672, salvage #82146 by @BrinShadewater):
|
||||
# ``auxiliary.background_review.extra_tools`` admits named parent tools to the review whitelist — e.g. a
|
||||
# human-gated proposal tool or a memory-provider write surface. Read from task_cfg (the
|
||||
# auxiliary.background_review block already loaded for this spawn) so no extra config I/O happens per
|
||||
# review.
|
||||
configured_extra_tools: set = set()
|
||||
try:
|
||||
extra_raw = _background_review_task_config(task_cfg).get("extra_tools", [])
|
||||
@@ -972,7 +1007,10 @@ def _run_review_in_thread(
|
||||
"""Daemon-thread worker: build the fork, run the prompt, surface the action summary via
|
||||
``agent._safe_print`` / ``background_review_callback``. ``review_run`` (from
|
||||
:func:`prepare_background_review_run`) cancelled before the first provider call aborts
|
||||
without entering ``run_conversation()``."""
|
||||
without entering ``run_conversation()``.
|
||||
|
||||
See #84423.
|
||||
"""
|
||||
if review_run is not None and review_run.cancel_requested.is_set():
|
||||
finish_background_review_run(agent, review_run)
|
||||
return
|
||||
@@ -993,11 +1031,25 @@ def _run_review_in_thread(
|
||||
try:
|
||||
# Silence stdout/stderr for THIS thread only: a process-global redirect would blank every
|
||||
# other thread's console for the whole review.
|
||||
# A process-global ``contextlib.redirect_stdout(devnull)`` here would also blank
|
||||
# ``sys.stdout``/``sys.stderr`` for every other thread — including a gateway event-loop thread
|
||||
# driving a Telegram long-poll — for the full duration of the review (tens of seconds), swallowing
|
||||
# their console output (#55769 / #55925). ``thread_scoped_silence`` routes only this thread's writes
|
||||
# to devnull and leaves all other threads on the real streams.
|
||||
with thread_scoped_silence():
|
||||
_run_review_fork(agent, messages_snapshot, prompt, task_cfg, review_run, st)
|
||||
# A buggy/legacy tool response shape must NOT take down the whole review (the outer
|
||||
# except would discard every action the fork DID complete), so coerce to an empty list.
|
||||
try:
|
||||
# Scan the review agent's messages for successful tool actions and surface a compact summary to
|
||||
# the user. Tool messages already present in messages_snapshot must be skipped, since the review
|
||||
# agent inherits that history and would otherwise re-surface stale "created"/"updated" messages
|
||||
# from the prior conversation as if they just happened (issue #14944). ``_change`` returned as a
|
||||
# list instead of a dict, #59437) must NOT take down the whole review with an AttributeError,
|
||||
# since the caller's outer except logs only "Background memory/skill review failed" and discards
|
||||
# every successful action the fork DID complete before the crash. Coerce an exception into an
|
||||
# empty actions list so the partial valid actions from earlier in the messages are returned
|
||||
# instead.
|
||||
actions = summarize_background_review_actions(
|
||||
st.review_messages, messages_snapshot,
|
||||
notification_mode=getattr(agent, "memory_notifications", "on"),
|
||||
|
||||
@@ -24,6 +24,11 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
# boto3 is not in the [all] extras; lazy_deps installs it on demand.
|
||||
try:
|
||||
# --------------------------------------------------------------------------- Ensure boto3/botocore are
|
||||
# installed before any code in this module runs. Upstream removed boto3 from [all] extras (PRs #24220,
|
||||
# #24515); lazy_deps handles on-demand installation so the Bedrock provider still works in the EKS
|
||||
# deployment without baking boto3 into the base image.
|
||||
# ---------------------------------------------------------------------------
|
||||
from tools.lazy_deps import ensure
|
||||
ensure("provider.bedrock", prompt=False)
|
||||
except Exception:
|
||||
@@ -260,7 +265,12 @@ def resolve_aws_auth_env_var(env: Optional[Dict[str, str]] = None) -> Optional[s
|
||||
|
||||
|
||||
def has_aws_credentials(env: Optional[Dict[str, str]] = None) -> bool:
|
||||
"""True if any AWS credential source (env vars or boto3 chain) is detected."""
|
||||
"""True if any AWS credential source (env vars or boto3 chain) is detected.
|
||||
|
||||
This two-tier approach mirrors the pattern from OpenClaw PR #62673: cloud environments (EC2, ECS,
|
||||
Lambda) provide credentials via instance metadata, not environment variables. The env-var check is a
|
||||
fast path for local development; the boto3 fallback covers all cloud deployments.
|
||||
"""
|
||||
return resolve_aws_auth_env_var(env) is not None or _boto3_chain_has_credentials()
|
||||
|
||||
|
||||
@@ -431,6 +441,8 @@ def convert_tools_to_converse(tools: List[Dict]) -> List[Dict]:
|
||||
|
||||
|
||||
# Converse rejects empty OR whitespace-only text blocks, so the placeholder must be non-whitespace.
|
||||
# A lone space is whitespace and is rejected too — the placeholder MUST itself be non-whitespace. Ref: issue
|
||||
# #9486.
|
||||
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
|
||||
_PLACEHOLDER_BLOCK = {"text": _EMPTY_TEXT_PLACEHOLDER}
|
||||
|
||||
@@ -447,6 +459,7 @@ def _image_block_from_data_url(url: str) -> Dict:
|
||||
header, _, data = url.partition(",")
|
||||
media_type = (header[5:].split(";")[0] if header.startswith("data:") else "") or "image/jpeg"
|
||||
try:
|
||||
# Ref: #33317.
|
||||
raw_bytes = base64.b64decode(data)
|
||||
except Exception:
|
||||
raw_bytes = data.encode("utf-8")
|
||||
|
||||
@@ -58,6 +58,13 @@ class BrowserProvider(ProviderBase):
|
||||
|
||||
# Legacy ``CloudBrowserProvider`` names still used by ``tools.browser_tool`` and out-of-tree subclasses.
|
||||
|
||||
# ------------------------------------------------------------------ Backward-compat shims for the
|
||||
# legacy CloudBrowserProvider API ------------------------------------------------------------------ The
|
||||
# pre-PR-#25214 ABC exposed ``is_configured()`` and ``provider_name()``; ``tools.browser_tool`` has ~6
|
||||
# callers that still use those names. Rather than churn every callsite (and break out-of-tree downstream
|
||||
# code that subclassed CloudBrowserProvider), we expose the old names as thin delegations to the new
|
||||
# API. Subclasses MUST implement :meth:`is_available` and :attr:`name`; they may override
|
||||
# ``is_configured`` / ``provider_name`` for compatibility with the legacy ABC but it is not required.
|
||||
def is_configured(self) -> bool:
|
||||
"""Backward-compat alias for :meth:`is_available`."""
|
||||
return self.is_available()
|
||||
|
||||
@@ -30,6 +30,9 @@ from agent.errors import EmptyStreamError
|
||||
from agent.fast_mode import effective_request_overrides
|
||||
from agent.turn_context import substitute_api_content
|
||||
from agent.gemini_native_adapter import is_native_gemini_base_url
|
||||
# Remote endpoints must never be fingerprinted: the probe waterfall is only valid for local/LM-Studio/Ollama
|
||||
# boxes. Non-Ollama remotes (sglang, vLLM, OpenAI-compat) expose Ollama-compat endpoints that can
|
||||
# misidentify and, without an api_key, return 401 on every leg (issue #89863).
|
||||
from agent.model_metadata import is_local_endpoint
|
||||
from agent.message_content import flatten_message_text
|
||||
from agent.message_metadata import append_message, stamp_message_timestamp
|
||||
@@ -1630,6 +1633,18 @@ def _assistant_tool_call_dict(agent, tool_call, index: int) -> dict:
|
||||
"function": {"name": tool_call.function.name, "arguments": tool_call.function.arguments}}
|
||||
# Preserve extra_content (Gemini thought_signature) or Gemini 3 thinking
|
||||
# models 400 on the next request.
|
||||
# Tool-call arguments are intentionally NOT redacted here. This dict enters the in-memory conversation
|
||||
# history that is replayed to the model on every subsequent turn AND persisted to state.db, which is
|
||||
# itself replayed verbatim on session resume (get_messages_as_conversation). Masking a credential to
|
||||
# `***` here poisons that replay: the model reads back its own `PGPASSWORD='***' psql ...` call and
|
||||
# copies the placeholder into the next tool call, breaking every credential-dependent command on the
|
||||
# second turn (#43083). The masking also provided no real protection — the same secret still leaks
|
||||
# verbatim through tool OUTPUT (file contents, command output, diffs, the compaction block), none of
|
||||
# which this pass ever touched. Keeping secrets out of the replayable store is a separate
|
||||
# tokenization/vault concern, not something arg-redaction can deliver without breaking replay.
|
||||
# Storage-time redaction remains governed by the `security.redact_secrets` toggle. (#19798 introduced
|
||||
# this; #43083 removed it.) Preserve extra_content (e.g. Gemini thought_signature) so it is sent back on
|
||||
# subsequent API calls. Without this, Gemini 3 thinking models reject the request with a 400 error.
|
||||
extra = getattr(tool_call, "extra_content", None)
|
||||
if extra is not None:
|
||||
tc_dict["extra_content"] = _dump_if_model(extra)
|
||||
@@ -1658,6 +1673,11 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic
|
||||
elif assistant_tool_calls and agent._needs_thinking_reasoning_pad():
|
||||
# DeepSeek v4 / Kimi thinking modes 400 on a replayed tool-call message without
|
||||
# reasoning_content; pad with a single space (empty string is rejected too).
|
||||
# Without it, replaying the persisted message causes HTTP 400 ("The reasoning_content in the
|
||||
# thinking mode must be passed back to the API"). Include streamed reasoning text when captured;
|
||||
# otherwise pad with a single space — DeepSeek V4 Pro tightened validation and rejects empty string
|
||||
# ("The reasoning content in the thinking mode must be passed back to the API"). A space satisfies
|
||||
# non-empty checks everywhere without leaking fabricated reasoning. Refs #15250, #17400, #17341.
|
||||
msg["reasoning_content"] = reasoning_text or " "
|
||||
elif reasoning_text:
|
||||
# Streaming-only providers accumulate reasoning via deltas and never set
|
||||
@@ -1665,6 +1685,16 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic
|
||||
# Promote ONLY when nothing set the field: SDK reasoning_content and the
|
||||
# tool-call pad win, and reasoning-less turns leave the field absent so
|
||||
# the replay-time leak guard and promotion tiers still apply.
|
||||
# Additive fallback (refs #16844, #16884). Streaming-only providers (glm, MiniMax, gpt-5.x via aigw,
|
||||
# Anthropic via openai-compat shims) accumulate reasoning through ``delta.reasoning_content`` chunks
|
||||
# but never land it on the message object as a top-level attribute, so neither branch above fires
|
||||
# and the chain-of-thought is stored only under the internal ``reasoning`` key. When the user later
|
||||
# replays that history through a DeepSeek-v4 / Kimi thinking model, the missing
|
||||
# ``reasoning_content`` causes HTTP 400 ("The reasoning_content in the thinking mode must be passed
|
||||
# back to the API."). Promote the already-sanitized streamed ``reasoning_text`` to
|
||||
# ``reasoning_content`` at write time, but ONLY when no prior branch already set it AND we actually
|
||||
# captured reasoning text. This preserves every existing behavior: - SDK-exposed
|
||||
# ``reasoning_content`` (OpenAI/Moonshot/DeepSeek SDK) still wins.
|
||||
msg["reasoning_content"] = reasoning_text
|
||||
|
||||
if getattr(assistant_message, "reasoning_details", None):
|
||||
@@ -1874,6 +1904,8 @@ def _should_skip_fallback_candidate(agent, fb: dict, fb_key: tuple, fb_provider:
|
||||
return True
|
||||
# Identity semantics (axes, shim aliases, credential surfaces, multi-endpoint pools)
|
||||
# are owned by agent.backend_identity — do not re-implement comparisons here.
|
||||
# Skip entries that resolve to the same backend that just failed — falling back to it loops the failure.
|
||||
# See #22548, #62984, #70893.
|
||||
from agent.backend_identity import BackendIdentity, should_skip_candidate
|
||||
current_ident = BackendIdentity.build(provider=getattr(agent, "provider", ""),
|
||||
model=getattr(agent, "model", ""), base_url=str(getattr(agent, "base_url", "") or ""))
|
||||
@@ -1937,6 +1969,8 @@ def _reresolve_fallback_reasoning_config(agent) -> None:
|
||||
"""Per-model override > global reasoning_effort (YAML False = disabled); a config load
|
||||
failure keeps the current reasoning_config rather than killing the swap."""
|
||||
try:
|
||||
# Re-resolve reasoning_config for the new fallback model (Closes #21256). Wrapped in try/except
|
||||
# because a config load failure must not kill the swap.
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_constants import resolve_reasoning_config
|
||||
agent.reasoning_config = resolve_reasoning_config(load_config() or {}, agent.model)
|
||||
@@ -2032,6 +2066,7 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool
|
||||
|
||||
# Clear the per-config context_length override so the fallback model's own context
|
||||
# window is resolved instead of the previous model's stale value.
|
||||
# See #22387.
|
||||
agent._config_context_length = None
|
||||
agent.model, agent.provider, agent.requested_provider = fb_model, fb_provider, fb_provider
|
||||
agent.base_url, agent.api_mode = fb_base_url, fb_api_mode
|
||||
@@ -2102,6 +2137,13 @@ def _iteration_summary_api_messages(agent, messages: list) -> list:
|
||||
api_msg.pop(key, None)
|
||||
# api_content holds the exact bytes the main loop sent; substituting (not popping)
|
||||
# keeps the summary's prefix identical instead of re-prefilling the largest context.
|
||||
# Strict OpenAI-compatible gateways (Fireworks-backed OpenCode Go, Mistral, Moonshot/Kimi) reject
|
||||
# any message key outside the Chat Completions schema. The main loop drops these via
|
||||
# ChatCompletionsTransport.convert_messages(), but the summary path hand-builds messages and calls
|
||||
# chat.completions.create() directly, bypassing the transport — so mirror that sanitization here:
|
||||
# tool_name (SQLite FTS bookkeeping), the codex_* reasoning carriers, timestamp (preserved on
|
||||
# gateway user replay entries for the stale-confirmation expiry check — #47868 rejection class), and
|
||||
# every Hermes-internal underscore-prefixed scaffolding key.
|
||||
substitute_api_content(api_msg)
|
||||
if needs_sanitize:
|
||||
agent._sanitize_tool_calls_for_strict_api(api_msg, model=sanitize_model)
|
||||
@@ -2243,6 +2285,8 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str:
|
||||
if getattr(agent, "suppress_status_output", False):
|
||||
# Strict machine-readable mode (-Q, oneshot): keep diagnostics off stdout. quiet_mode is
|
||||
# NOT the gate — the interactive CLI runs quiet_mode=True by default and must see this.
|
||||
# Strict machine-readable mode (hermes chat -Q, oneshot, background review): keep diagnostics out of
|
||||
# stdout so wrappers receive only the final assistant content (#93220 class).
|
||||
logger.warning(warning)
|
||||
else:
|
||||
agent._safe_print(warning)
|
||||
@@ -2797,6 +2841,8 @@ class _StreamingCall:
|
||||
as in-stream chunks (choices=None + error_type/error_message), which
|
||||
would otherwise surface as a misleading EmptyStreamError plus retries."""
|
||||
usage = chunk.usage if hasattr(chunk, "usage") and chunk.usage else None # final usage chunk
|
||||
# Without this check the error is silently dropped and the stream ends empty → EmptyStreamError →
|
||||
# misleading "empty stream" message and pointless retries on the same bad request. (#65631)
|
||||
_err_type = getattr(chunk, "error_type", None)
|
||||
_err_msg = getattr(chunk, "error_message", None)
|
||||
if _err_type or _err_msg:
|
||||
@@ -2806,6 +2852,7 @@ class _StreamingCall:
|
||||
raise ProviderStreamError(status_code=_status, body=body, raw_text=f"{_err_type}: {_err_msg}")
|
||||
# Nous Portal usage frames (choices=[] + lastOne=true, no [DONE]) are a
|
||||
# clean terminal, not a drop; relabelled upstreams send 1 / "true".
|
||||
# See #90848.
|
||||
last_one = getattr(chunk, "lastOne", None)
|
||||
if last_one is None and isinstance(getattr(chunk, "model_extra", None), dict):
|
||||
last_one = chunk.model_extra.get("lastOne")
|
||||
|
||||
@@ -117,6 +117,13 @@ class ClientLifecycleMixin:
|
||||
|
||||
``close()`` releases FDs from the calling thread while other threads may still hold the fd in an SSL BIO;
|
||||
a recycled fd then gets a TLS record written into an unrelated file (SQLite-header corruption).
|
||||
|
||||
The shared primary client has no single owning thread — worker threads from stale-killed attempts
|
||||
may still be unwinding their SSL BIOs, and the codex-direct / MoA paths stream on the shared client
|
||||
itself. If we release an FD while another thread's SSL layer still caches the raw integer fd, the
|
||||
kernel can recycle it into an unrelated ``open()`` (e.g. ``kanban.db``) and the unwinding TLS flush
|
||||
then writes an application-data record into that file — the SQLite-header corruption documented in
|
||||
#29507/#70773.
|
||||
"""
|
||||
if client is None:
|
||||
return
|
||||
@@ -134,6 +141,12 @@ class ClientLifecycleMixin:
|
||||
|
||||
The worker may be blocked in an OpenSSL read; hard-closing from the timeout thread releases FDs under a
|
||||
live BIO (native corruption / SIGSEGV). Only ``shutdown()`` so the read sees EOF and the worker closes itself.
|
||||
|
||||
See #94248.
|
||||
A delegation deadline abandons this agent's daemon worker while it may still be blocked inside an
|
||||
in-flight OpenSSL ``read`` (Codex Responses stream, httpx request). This helper only ``shutdown()``s
|
||||
pooled sockets (safe from any thread), settling blocked reads with EOF/EPIPE so the worker can
|
||||
unwind and run the real close from its own thread. See #70773, #94248.
|
||||
"""
|
||||
drained = 0
|
||||
# Shared primary client (codex-direct / MoA stream on it directly).
|
||||
@@ -194,6 +207,11 @@ class ClientLifecycleMixin:
|
||||
return False
|
||||
self.client = new_client
|
||||
# Never hard-close the replaced shared client (another thread may still be unwinding on the old pool).
|
||||
# #70773: never hard-close the replaced shared client from here — the caller may not be the thread
|
||||
# whose request is still unwinding on the old pool (credential rotation and dead-connection cleanup
|
||||
# run on the turn thread while stale-killed workers unwind; the codex-direct path streams on the
|
||||
# shared client itself). Retire it instead: sockets are shut down (FD-safe), FD release deferred to
|
||||
# GC.
|
||||
self._retire_shared_openai_client(old_client, reason=f"replace:{reason}")
|
||||
return True
|
||||
|
||||
@@ -310,6 +328,10 @@ class ClientLifecycleMixin:
|
||||
try:
|
||||
shutdown_count = self._force_close_tcp_sockets(client)
|
||||
# Zero sockets shut down means the worker stays blocked — WARN, not success.
|
||||
# tcp_force_closed=0 means the stranger-thread abort found no sockets to shut down — the worker
|
||||
# stays blocked in recv and the provider keeps the slot (#72975). Surface that as WARNING so it
|
||||
# cannot be mistaken for a successful abort in the logs.
|
||||
# See #72975.
|
||||
_log = logger.warning if shutdown_count == 0 else logger.info
|
||||
_log(
|
||||
"%s client aborted (%s, shared=False, tcp_force_closed=%d, deferred_close=stranger_thread) %s%s",
|
||||
@@ -572,7 +594,12 @@ class ClientLifecycleMixin:
|
||||
|
||||
def _try_refresh_env_client_credentials(self) -> bool:
|
||||
"""Adopt ``~/.hermes/.env`` credential/base-url edits at the turn boundary (a Settings save updates ``.env``
|
||||
but a live worker keeps init-time values). Adoption rule: ``_should_adopt_env_credentials``."""
|
||||
but a live worker keeps init-time values). Adoption rule: ``_should_adopt_env_credentials``.
|
||||
|
||||
Covers api-key registry providers and named custom providers with a ``key_env`` (#67935) — the
|
||||
latter resolve to ``provider="custom"`` with no registry entry, so they are matched through the
|
||||
runtime provider's config lookup instead.
|
||||
"""
|
||||
if self.api_mode != "chat_completions" or getattr(self, "_fallback_activated", False):
|
||||
return False
|
||||
resolved = self._resolve_env_credentials()
|
||||
|
||||
@@ -426,10 +426,38 @@ def _chat_messages_to_responses_input(
|
||||
``native_compaction_eligible``: THIS request carries ``context_management``; gates both replaying ``compaction``
|
||||
checkpoints and ``prune_pre_checkpoint_items``. Checkpoints persist across model swaps / compression flips / resume,
|
||||
so without the gate one checkpoint would erase pre-checkpoint history on a model that cannot decrypt it (lossless:
|
||||
local history is never truncated)."""
|
||||
local history is never truncated).
|
||||
|
||||
Earlier (PR #26644, May 2026) we believed xAI's OAuth/SuperGrok ``/v1/responses`` surface rejected
|
||||
replayed ``encrypted_content`` reasoning items minted by prior turns, and we stripped them. That
|
||||
decision was wrong — xAI explicitly relies on Hermes threading encrypted reasoning back across turns for
|
||||
cross-turn coherence (the whole point of their partnership integration). We now replay encrypted
|
||||
reasoning on every Responses transport (xAI, native Codex, custom relays) and let xAI tell us explicitly
|
||||
if a specific surface ever rejects a payload.
|
||||
The Copilot backend (api.githubcopilot.com/responses) binds these ids to a specific backend "connection"
|
||||
— credential-pool rotation, a gateway restart, or routine load-balancer churn between turns all
|
||||
invalidate it — and rejects a stale id with HTTP 401 "input item ID does not belong to this connection"
|
||||
even for short ids (see #32716). ``phase``/ ``status``/``content`` are still replayed; only ``id`` is
|
||||
unsafe to reuse across a Copilot connection.
|
||||
``native_compaction_eligible`` mirrors, for THIS request, the decision made by
|
||||
``native_compaction.native_compaction_context_management`` — it is True only when that gate returned a
|
||||
payload, i.e. when the request actually carries ``context_management``. It controls two things that must
|
||||
never outlive the gate: replaying ``type: "compaction"`` checkpoint items, and restructuring the wire
|
||||
around them (``prune_pre_checkpoint_items``). Checkpoints are persisted in the ``codex_reasoning_items``
|
||||
sidecar and survive a mid-session model swap, a ``compression.enabled: false`` flip, the rejection kill
|
||||
switch and a resumed session; without this flag a single captured checkpoint would keep deleting every
|
||||
pre-checkpoint item from every later request, on a model that cannot decrypt the blob (#85914). Default
|
||||
False = pre-feature wire, which is also correct for every caller that never sends ``context_management``
|
||||
(auxiliary/compression client, ad-hoc ``convert_messages``). Dropping the checkpoint costs nothing:
|
||||
Hermes' local history is never truncated by native compaction, so the full conversation is still on the
|
||||
wire.
|
||||
"""
|
||||
items: List[Dict[str, Any]] = []
|
||||
# Parallel to ``items``: source chat message per item. Pruning reads a summary
|
||||
# carrier's provenance from the source; the converted item may be a lossy shape.
|
||||
# Pruning needs this to read a canonical summary carrier's up-to-date, provenance-tagged content
|
||||
# directly — the converted `item` can be a lossy shape (stale exact-replay, or a typed
|
||||
# `function_call_output` wrapper) that no longer carries it (#90976).
|
||||
item_sources: List[Optional[Dict[str, Any]]] = []
|
||||
seen_item_ids: set = set()
|
||||
def emit(new_items: List[Dict[str, Any]], msg: Dict[str, Any]) -> None:
|
||||
@@ -470,6 +498,17 @@ def _chat_messages_to_responses_input(
|
||||
# The server renders nothing placed before a compaction item, so pre-checkpoint history is
|
||||
# dead weight and plaintext asks / merged summaries silently vanish. Keep the newest checkpoint
|
||||
# first, retain pre-checkpoint USER and SUMMARY messages within a token budget, leave the tail.
|
||||
# Native server-side compaction: when a replayed checkpoint is present, restructure the wire around it.
|
||||
# Gated on the CURRENT request's native eligibility, not merely on the presence of a checkpoint: a
|
||||
# persisted checkpoint outlives the gate, and pruning for a request that carries no
|
||||
# ``context_management`` deletes history the server never compacted. ``item_sources`` (parallel to
|
||||
# ``items``) carries the raw chat message each converted item came from. A canonical summary carrier's
|
||||
# content can be lost or gone stale by the time it becomes a Responses item — a merge-into-tail
|
||||
# tool-result carrier becomes a typed ``function_call_output`` (no ``content``/``role`` at all), and a
|
||||
# merge-into-tail assistant carrier can be shadowed by a stale exact ``codex_message_items`` replay from
|
||||
# before the merge rewrote its content. Pruning reads the source message's own up-to-date,
|
||||
# provenance-tagged content directly instead of trying to recover it from whatever shape the conversion
|
||||
# produced (#90976).
|
||||
if not native_compaction_eligible:
|
||||
return items
|
||||
from agent.native_compaction import prune_pre_checkpoint_items
|
||||
@@ -478,7 +517,12 @@ def _chat_messages_to_responses_input(
|
||||
|
||||
class ResponsesRouteFlags(NamedTuple):
|
||||
"""Which special Responses-API route an agent is talking to. Single owner of the
|
||||
codex/xai/github predicates — every site must call :func:`classify_responses_route`."""
|
||||
codex/xai/github predicates — every site must call :func:`classify_responses_route`.
|
||||
|
||||
Every site that needs these flags (request kwargs build, preflight estimation, silent- reject hints)
|
||||
must call :func:`classify_responses_route` instead of re-implementing the string comparisons inline —
|
||||
inline copies drift (backend-identity class: #22548/#70893/#59561/#72468).
|
||||
"""
|
||||
is_codex_backend: bool
|
||||
is_xai_responses: bool
|
||||
is_github_responses: bool
|
||||
@@ -505,7 +549,12 @@ def estimate_native_responses_preflight_tokens(
|
||||
agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None,
|
||||
) -> Optional[int]:
|
||||
"""Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted
|
||||
session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails."""
|
||||
session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails.
|
||||
|
||||
Automatic preflight previously counted the full durable transcript. On a natively compacted Codex
|
||||
session that overstates the wire by several times and fires local compression against history the main
|
||||
request will never send (#96155).
|
||||
"""
|
||||
if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list):
|
||||
return None
|
||||
route = classify_responses_route(agent)._asdict()
|
||||
|
||||
@@ -307,6 +307,9 @@ def make_codex_app_server_event_bridge(agent) -> Callable[[dict], None]:
|
||||
|
||||
def _fire_delta(params: dict, attr: str) -> None:
|
||||
text = params.get("delta") or params.get("text") or ""
|
||||
# Single-writer guard (#65991): a superseded stream must not pollute the turn's accumulated text
|
||||
# (which also feeds the interim-visible-text de-dup comparison), even when a caller reaches this
|
||||
# directly (the tool-suppressed content path) rather than through _fire_stream_delta.
|
||||
if isinstance(text, str) and text:
|
||||
agent_cb(attr, f"{attr} raised", args=(text,))
|
||||
|
||||
@@ -380,6 +383,11 @@ def _ensure_codex_session(agent) -> None:
|
||||
auto_approve_requests = is_approval_bypass_active()
|
||||
except Exception:
|
||||
logger.debug("codex app-server: approval-bypass lookup failed; keeping fail-closed default", exc_info=True)
|
||||
# Bridge codex JSON-RPC notifications (item/started, item/completed, item/agentMessage/delta, ...) into
|
||||
# Hermes' gateway UI callbacks (tool_progress_callback, _fire_stream_delta,
|
||||
# _emit_interim_assistant_message). Without this, Discord/Telegram users see no live tool-progress or
|
||||
# interim commentary while codex_app_server is running — only the final answer (#33200). Supersedes the
|
||||
# narrower item/started-only bridge from #38835.
|
||||
agent._codex_session = CodexAppServerSession(
|
||||
cwd=getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()), approval_callback=approval_callback,
|
||||
request_routing=_ServerRequestRouting(auto_approve_exec=auto_approve_requests, auto_approve_apply_patch=auto_approve_requests),
|
||||
|
||||
@@ -11,6 +11,18 @@ _COMPACTION_INTERNAL_FIELDS = (
|
||||
"tool_calls",
|
||||
"finish_reason",
|
||||
"reasoning",
|
||||
# Provider replay/metadata fields that ride the wire on every request but are invisible to
|
||||
# ``msg["content"]``/``msg["tool_calls"]`` accounting. Codex Responses sessions in particular carry
|
||||
# ``codex_reasoning_items`` blobs of ``encrypted_content`` that can dominate the serialized session (a
|
||||
# measured 214-turn session held ~115K tokens / 27% of its payload there — #55572).
|
||||
# ``reasoning_details`` is handled separately (see ``_reasoning_details_text_chars``): its signed/base64
|
||||
# envelope is excluded from the budget, mirroring the preflight estimator's exclusion in
|
||||
# ``model_metadata._estimate_message_tokens_without_images`` (#73298).
|
||||
# An assistant turn may carry only reasoning/thinking content with no visible text (extended-thinking
|
||||
# turns, thinking-only recovery responses). Such a turn is persisted with its reasoning fields and is
|
||||
# recallable from the transcript, but dropping it here as "empty" makes it vanish from the
|
||||
# resumed/reloaded session view while the desktop's reasoning disclosure has nothing to render. Keep it
|
||||
# when it carries reasoning so the "Thinking…" block still shows. (#44022)
|
||||
"reasoning_content",
|
||||
"reasoning_details",
|
||||
"codex_reasoning_items",
|
||||
|
||||
@@ -114,6 +114,14 @@ def _run_under_progress_timeout(
|
||||
from agent.conversation_compression import CompressionCommitFence, run_compress_context_with_progress_timeout
|
||||
|
||||
def _snapshot_worker(fence=None):
|
||||
# #76354 review F3: the pooled worker must NEVER share the caller's live transcript. Plugin/legacy
|
||||
# context engines are allowed to mutate their input list in place; after a host timeout the worker
|
||||
# stays alive, so a shared list would let a late engine rewrite the live conversation (roles,
|
||||
# ordering, persisted content) behind the caller's back. Deep-snapshot here, on the worker thread,
|
||||
# so the caller's list object is never touched by pooled code. Results are published to
|
||||
# caller-visible state only via the returned value of an ADMITTED commit (the host discards results
|
||||
# on timeout/cancel); durable SessionDB mutation is already gated behind the commit fence inside
|
||||
# compress_context.
|
||||
snapshot = copy.deepcopy(messages)
|
||||
result_msgs, result_prompt = run(fence, target_messages=snapshot)
|
||||
return (messages if result_msgs is snapshot else result_msgs), result_prompt
|
||||
@@ -185,9 +193,18 @@ class CompressionFacadeMixin:
|
||||
) -> tuple:
|
||||
"""Forwarder — see ``agent.conversation_compression.compress_context``.
|
||||
``force=True`` (manual /compress) bypasses the summary-failure cooldown; ``bypass_cooldown=True``
|
||||
(provider-proven overflow recovery) runs one real attempt while the cooldown stays armed."""
|
||||
(provider-proven overflow recovery) runs one real attempt while the cooldown stays armed.
|
||||
|
||||
``force=True`` is passed by the manual ``/compress`` slash command so users can bypass the
|
||||
summary-failure cooldown after an auto-compress abort. Auto-compress callers use the default
|
||||
``force=False``. See #100661.
|
||||
"""
|
||||
# Per-attempt timeout signal for turn-start preflight and in-loop consumers: a stalled
|
||||
# compression must not be mistaken for a structural no-op. Thread-local + per-agent lock.
|
||||
# A stalled compression must not be mistaken for a structural no-op and followed by the oversized
|
||||
# provider request it was meant to prevent. The typed helper upgrades the simple attribute to
|
||||
# thread-local state guarded by a per-agent lock so overlapping automatic/manual entrypoints cannot
|
||||
# clobber each other's outcome (#98741).
|
||||
from agent.conversation_compression import (
|
||||
CompressionCommitFence, compress_context, reset_context_compression_timeout_outcome,
|
||||
resolve_context_compression_timeouts,
|
||||
@@ -206,6 +223,9 @@ class CompressionFacadeMixin:
|
||||
root = self._conversation_root_id()
|
||||
if root:
|
||||
token = set_conversation_context(root)
|
||||
# Initialized alongside `token`: the turn-lease timeout/interrupt early returns leave the try block
|
||||
# before set_affinity_scope() runs, and the finally reads this name unconditionally
|
||||
# (UnboundLocalError otherwise — the 4 red cross-process lease tests on PR #97158).
|
||||
affinity_token = None
|
||||
if get_affinity_scope() is None:
|
||||
declared = declared_conversation_scope_safe(self)
|
||||
|
||||
+285
-9
@@ -46,6 +46,19 @@ def _safe_int(value: Any) -> int | None:
|
||||
# summary sees it while the detached stalled worker does not. A stall raises nothing, so the aux client's
|
||||
# exception-path fallback never fires; the host pins a fallback route for exactly ONE retry (the sole aux
|
||||
# call per compaction). The main-model retry must NOT re-issue the pin.
|
||||
# ── Pinned summary route ───────────────────────────────────────────────── The summary call normally
|
||||
# resolves its provider/model from ``auxiliary.compression``. One caller needs to override that for a single
|
||||
# attempt: after the host's progress-aware timeout aborts a stalled summary (#78981),
|
||||
# ``agent.conversation_compression`` re-runs compression with the route pinned to a configured
|
||||
# ``fallback_chain`` entry. Nothing raised out of the stalled call, so the auxiliary client's own fallback
|
||||
# handling — which only runs from its exception path — never saw that failure. A ContextVar, not an
|
||||
# attribute on the compressor: the aborted worker is detached and still alive on the pool, and the
|
||||
# compressor object is shared with it. Context is copied per worker (``propagate_context_to_thread``), so
|
||||
# the pin reaches the retry's whole synchronous call chain and cannot leak into the stalled attempt or any
|
||||
# unrelated auxiliary call. Coverage is the single ``_generate_summary`` LLM call only. That is one call per
|
||||
# compression run (its only non-recursive call site is the compress path; the two recursive calls are the
|
||||
# deliberate main-model retry that must NOT re-issue the pin). The summary call is the ONLY auxiliary LLM
|
||||
# call a lean compaction attempt makes (#96603) — there are no sibling digest calls.
|
||||
_SUMMARY_ROUTE_PIN: contextvars.ContextVar[Optional[Dict[str, Any]]] = (
|
||||
contextvars.ContextVar("hermes_summary_route_pin", default=None)
|
||||
)
|
||||
@@ -93,7 +106,10 @@ _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS: tuple[str, ...] = (
|
||||
def _is_hygiene_preagent_only_cooldown(error: object) -> bool:
|
||||
"""Return True for a cooldown that belongs only to pre-agent hygiene.
|
||||
Hygiene watchdog timeouts / turn-hold deferrals are not evidence of an auxiliary-model failure and
|
||||
must never block the in-agent compressor."""
|
||||
must never block the in-agent compressor.
|
||||
|
||||
See #74136, #86972.
|
||||
"""
|
||||
text = str(error or "").strip().casefold()
|
||||
return any(marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS)
|
||||
|
||||
@@ -114,6 +130,10 @@ def _response_finish_reason(response: Any) -> str:
|
||||
|
||||
# Marker for a length-stopped (PARTIAL) summary; the except-branch classifier keys
|
||||
# on this exact substring, so keep raise sites and classifier in sync.
|
||||
# RuntimeError marker raised when the summarizer's generation stopped on the output-token cap
|
||||
# (``finish_reason == "length"``). A length stop means the summary text is PARTIAL — persisting it as a
|
||||
# compaction checkpoint would silently truncate the conversation's memory and feed the cut-off text back
|
||||
# into every subsequent iterative-update prompt. (Ported from earendil-works/pi#7048 / commit 97fa14e39.)
|
||||
_TRUNCATED_SUMMARY_MARKER = "finish_reason=length"
|
||||
|
||||
|
||||
@@ -123,6 +143,11 @@ def _is_summary_access_or_quota_error(exc: Exception) -> bool:
|
||||
# No active secret scope is a missing-credential failure of our own making;
|
||||
# classify as credential so compress() preserves the session unchanged.
|
||||
try:
|
||||
# A credential read that failed closed because no profile secret scope was active (multiplexed
|
||||
# gateway, worker thread without the caller's ContextVars) is a missing-credential failure of our
|
||||
# own making: the summary model cannot be reached until the spawn site is fixed, and a placeholder
|
||||
# summary would only destroy the middle window for nothing. Classify it with the credential class so
|
||||
# compress() preserves the session unchanged (#100849 bundle: every hygiene pass truncated).
|
||||
from agent.secret_scope import UnscopedSecretError
|
||||
except Exception: # pragma: no cover - import guard
|
||||
UnscopedSecretError = () # type: ignore[assignment]
|
||||
@@ -150,6 +175,12 @@ HISTORICAL_TASK_HEADING = "## Historical Task Snapshot"
|
||||
|
||||
|
||||
SUMMARY_PREFIX = (
|
||||
# Jul 2026 (#65848 class): identical to the pre-#69619 prefix except it lacked the explicit "tools
|
||||
# remain fully active" clause — the strong REFERENCE ONLY framing bled into general tool-use suppression
|
||||
# (observed: 7 consecutive narration-only turns immediately after a compression event on a production
|
||||
# deployment).
|
||||
# Carveout era (#41607/#38364/#42812): "consistent → use as background" licensed stale-task resumption
|
||||
# on topic overlap.
|
||||
"[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted "
|
||||
"into the summary below. This is a handoff from a previous context "
|
||||
"window — treat it as background reference, NOT as active instructions. "
|
||||
@@ -193,6 +224,21 @@ COMPRESSED_SUMMARY_HAS_USER_TURN_KEY = "_compressed_summary_has_user_turn"
|
||||
# Only micro markers may be superseded/defragged/rehydrated: a batch marker's
|
||||
# content is NOT in the rolling micro summary, so rewriting one destroys history.
|
||||
MICRO_COMPACT_MARKER_KEY = "_micro_compact_marker"
|
||||
# Intrinsic marker stamped on a message dict once it has been written to the SQLite session store. Used by
|
||||
# ``_flush_messages_to_session_db`` to decide what is already durable. An object-identity (``id(msg)``)
|
||||
# dedup set cannot be trusted across turns: once a flushed message dict is dropped from the live list (e.g.
|
||||
# by scaffolding rewind or in-place compaction) and garbage- collected, CPython is free to hand its address
|
||||
# to a brand-new assistant/tool message, whose ``id()`` then collides with the stale entry and the real turn
|
||||
# is silently never persisted. A marker bound to the dict itself cannot be aliased that way. The ``_``
|
||||
# prefix is mandatory: the wire sanitizers (agent/transports/chat_completions.py,
|
||||
# agent/chat_completion_helpers.py) strip every top-level ``_``-prefixed key before the request leaves the
|
||||
# process, so this never reaches a strict OpenAI-compatible gateway. CONTRACT (#92231): the marker asserts
|
||||
# "this dict's CONTENT is durable as written". Loaded rows are stamped at materialization time
|
||||
# (hermes_state._rows_to_conversation), so any code that mutates a loaded or flushed dict's content in place
|
||||
# and needs the change persisted MUST pop the marker (and invalidate _db_flush_scan_prefix if the dict may
|
||||
# sit inside the bounded-scan prefix) — see agent/turn_finalizer.py (fill-empty-tail) and
|
||||
# agent/context_compressor.py (micro-compaction defrag) for the two canonical pop sites. Mutating without
|
||||
# popping leaves the DB silently stale.
|
||||
_DB_PERSISTED_MARKER = "_db_persisted"
|
||||
# Carried-forward tail rows archive as rewind-style (active=0, compacted=0) so
|
||||
# they don't duplicate live copies in recall; never persisted (unknown column).
|
||||
@@ -630,6 +676,12 @@ def _is_clarify_non_response_sentinel(response: Any) -> bool:
|
||||
|
||||
# Ghost-skill defense: the ONE canonical prune marker; emit sites and presence
|
||||
# checks must use the same string so they cannot drift.
|
||||
# Ghost-skill defense (#32106): when compaction reduces an old ``skill_view`` result to a 1-line metadata
|
||||
# summary, the model still believes the skill is loaded even though its instructions are gone. The marker
|
||||
# below is the ONE canonical prune signal — ``_skill_pruned_marker()`` builds it and every presence check
|
||||
# matches against the same string, so the emit side and the check side can never drift apart (the original
|
||||
# PR #44166 emitted ``[SKILL_PRUNED:`` but presence-checked ``[SKILL_PRUNED]``, making re-injection fire
|
||||
# even when the marker had survived).
|
||||
SKILL_PRUNED_MARKER_PREFIX = "[SKILL_PRUNED:"
|
||||
# Small skill_view results stay verbatim; shared by emit site and summarizer scan.
|
||||
_SKILL_VIEW_PRUNE_MIN_CHARS = 5000
|
||||
@@ -783,6 +835,13 @@ def _build_recovery_footer(session_id: str, region_len: int) -> str:
|
||||
|
||||
# Detailed session log comes from the SAME single summary request (one aux LLM
|
||||
# call per attempt); coverage via input sampling, exact needles via anchor index.
|
||||
# One flat 2-3K-token summary cannot carry a 400K+ region's specifics — the eval showed recall collapsing to
|
||||
# ~33% when the big tail (which accidentally archived restated facts) shrank. The detailed,
|
||||
# identifier-preserving session log is produced by the SAME single summary request as the narrative summary
|
||||
# (one auxiliary LLM call per compaction attempt, total — #96603: the earlier per-chunk digest loop made up
|
||||
# to 28 extra aux calls and pushed compactions to 7-11 minutes on slow aux routes). Coverage over oversized
|
||||
# regions comes from even input sampling (see ``_sample_summary_input``), and exact-needle defense comes
|
||||
# from the LLM-free anchor index below.
|
||||
_LEAN_SESSION_LOG_HEADING = "## Detailed Session Log (oldest first)"
|
||||
# Extra output-token guidance for the session-log section (single response).
|
||||
_LEAN_SESSION_LOG_BUDGET_TOKENS = 4_000
|
||||
@@ -852,6 +911,8 @@ def _build_anchor_index(turns: List[Dict[str, Any]]) -> str:
|
||||
|
||||
# Message-count window (distinct from the token-based tail boundary) in which a
|
||||
# just-loaded skill_view body must survive the Phase-1 prune.
|
||||
# A skill_view call within this many trailing messages counts as "just loaded": its full instruction body
|
||||
# must survive the Phase-1 prune even when the token-budget boundary would otherwise demote it (#32106).
|
||||
_SKILL_PRUNE_RECENT_WINDOW = 10
|
||||
|
||||
|
||||
@@ -912,12 +973,15 @@ _MAX_TAIL_MESSAGE_FLOOR = 8
|
||||
|
||||
# Skip the LLM call when the compressible middle is below this fraction of the
|
||||
# threshold (and a prior ineffectiveness strike exists); dropping alone suffices.
|
||||
# See #60451.
|
||||
_FEASIBILITY_SKIP_MIDDLE_FRACTION = 0.10
|
||||
# Under pressure, demote large tool outputs even inside the protected region but
|
||||
# keep this many trailing messages verbatim.
|
||||
_PRESSURE_KEEP_RECENT_MESSAGES = 3
|
||||
# Newest image-bearing tool results kept verbatim; older image payloads retire
|
||||
# even inside protect_last_n (matches the Anthropic adapter's keep-window).
|
||||
# Native vision_analyze / computer_use screenshots that sit inside the protected tail cannot be demoted by
|
||||
# pass 2, so they ride every later request until anti-thrash disables compression (#92699).
|
||||
_MAX_KEEP_TOOL_IMAGES = 3
|
||||
|
||||
# Below this window the threshold is floored (raise-only): at 50% the incompressible
|
||||
@@ -929,6 +993,8 @@ _SMALL_CTX_THRESHOLD_PERCENT = 0.75
|
||||
_PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+")
|
||||
|
||||
# MEDIA directives must not reach the summarizer or they get re-emitted as active.
|
||||
# MEDIA delivery directives must not reach the summarizer — if one leaks into the summary, the downstream
|
||||
# model may re-emit it as an active directive on the next turn, triggering bogus attachment sends (#14665).
|
||||
_MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+")
|
||||
_HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)")
|
||||
|
||||
@@ -1079,6 +1145,9 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) -
|
||||
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
|
||||
# Charge only thinking TEXT, never the signed/base64 envelope; skip when the
|
||||
# same text already rides in reasoning/reasoning_content.
|
||||
# When the same thinking text already rides in ``reasoning``/``reasoning_content`` (measured
|
||||
# byte-identical on Anthropic-wire sessions), skip it here entirely so the prose is not charged twice on
|
||||
# top of the envelope exclusion. See #73298.
|
||||
if not (msg.get("reasoning") or msg.get("reasoning_content")):
|
||||
tokens += _reasoning_details_text_chars(msg.get("reasoning_details")) // _CHARS_PER_TOKEN
|
||||
return tokens
|
||||
@@ -1256,6 +1325,11 @@ def _strip_historical_media(messages: List[Dict[str, Any]]) -> List[Dict[str, An
|
||||
# non-empty for the zero-user-turn guard). Rule 2: superseded tool-result image, even in the tail.
|
||||
return (
|
||||
(0 < anchor and index < anchor)
|
||||
# When the ONLY image-bearing user message is the very first one (``anchor == 0``) and newer
|
||||
# tool-result images exist, the model has moved on — but the opening base64 blob used to survive
|
||||
# every compaction forever, which is half the wedge in #89938 (the reported session opened with
|
||||
# a ~200KB poster). When nothing newer exists the opening image IS the newest image and is kept,
|
||||
# consistent with keep-newest everywhere else.
|
||||
or (anchor == 0 and index == 0 and tool_anchor > 0)
|
||||
or (message.get("role") == "tool" and index != tool_anchor)
|
||||
)
|
||||
@@ -1437,6 +1511,8 @@ def _json_dict(text: Any) -> dict:
|
||||
"""Parse ``text`` as a JSON object; ``{}`` for empty, invalid, or non-object input."""
|
||||
try:
|
||||
parsed = json.loads(text) if text else {}
|
||||
# Just-loaded / actively-referenced skills survive verbatim (#32106). Pass-4 pressure demotion overrides
|
||||
# this.
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
return {}
|
||||
return parsed if isinstance(parsed, dict) else {}
|
||||
@@ -1484,6 +1560,10 @@ def _memory_provider_section(memory_context: str) -> str:
|
||||
def _today_for_prompt() -> str:
|
||||
"""Date-only (user tz) for temporal anchoring; "" when the clock fails. Cache-safe: the summary is outside the prefix."""
|
||||
try:
|
||||
# Date-only granularity matches system_prompt.py:337 (PR #20451) and the user's configured timezone
|
||||
# via hermes_time.now(). The compaction summary is a mid-conversation message that is NOT part of
|
||||
# the cached prefix, so a date here never affects prompt-cache stability. Resolved defensively — a
|
||||
# clock failure must never block compaction.
|
||||
from hermes_time import now as _hermes_now
|
||||
return _hermes_now().strftime("%Y-%m-%d")
|
||||
except Exception: # pragma: no cover - clock resolution is best-effort
|
||||
@@ -1724,7 +1804,17 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
"""Clear all per-session compaction state at a real session boundary.
|
||||
Session end (CLI exit, gateway expiry, id rotation) — NOT /new or /reset. Every per-session
|
||||
flag/counter can contaminate the next live session (suppressed compression, stale cooldowns,
|
||||
misleading warnings), so the whole surface is reset here."""
|
||||
misleading warnings), so the whole surface is reset here.
|
||||
|
||||
Session end (CLI exit, gateway expiry, session-id rotation) goes through this method rather than
|
||||
``on_session_reset()`` (/new, /reset). The original fix (#38788) only cleared ``_previous_summary``,
|
||||
but the same cross-session contamination risk applies to every per-session variable that
|
||||
``on_session_reset()`` clears: stale ``_ineffective_compression_count`` can suppress compression in
|
||||
a subsequent live session; ``_summary_failure_cooldown_until`` can block summary generation;
|
||||
``_last_compress_aborted`` can make callers think compression is still aborted;
|
||||
``_last_aux_model_failure_*`` can surface stale error warnings; ``_last_summary_dropped_count`` /
|
||||
``_last_summary_fallback_used`` can produce misleading user warnings.
|
||||
"""
|
||||
self._reset_session_compaction_state()
|
||||
|
||||
def _reset_real_usage_pairing(self) -> None:
|
||||
@@ -1880,7 +1970,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._durable_write("set_compression_ineffective_count", "compression ineffective count", self._ineffective_compression_count)
|
||||
|
||||
def _load_anti_thrash_recovery_deadline(self) -> None:
|
||||
"""Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed."""
|
||||
"""Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed.
|
||||
|
||||
See #100185.
|
||||
"""
|
||||
self._load_durable("_anti_thrash_recovery_deadline", "get_compression_recovery_deadline", "compression recovery deadline", float, 0.0)
|
||||
|
||||
def _set_anti_thrash_recovery_deadline(self, deadline: float) -> None:
|
||||
@@ -1924,6 +2017,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._verify_compaction_cleared_threshold = True
|
||||
if feasibility_skip:
|
||||
# A pre-LLM feasibility skip is not a summary-quality verdict: it must neither extend nor reset the streak.
|
||||
# A deliberate pre-LLM feasibility skip (#60451) is not a summary-quality verdict: it must
|
||||
# neither extend a fallback streak (two skips would otherwise latch the >= 2 breaker and disable
|
||||
# compression entirely — including the cheap deterministic dropping the skip exists to reach)
|
||||
# nor reset one (a skip proves nothing about the summary model's health).
|
||||
if not self.quiet_mode:
|
||||
logger.info(
|
||||
"Compaction completed via pre-LLM feasibility skip; fallback_compression_streak unchanged (%d)",
|
||||
@@ -1980,6 +2077,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
return None
|
||||
# Hygiene-only cooldowns share the column but are not a 429/aux fault; the in-agent compressor may run.
|
||||
# A hygiene write may have overwritten an aux-model row; drop the in-memory cooldown too.
|
||||
# Hygiene watchdog timeouts and turn-hold deferrals persist the same column so the pre-agent pass
|
||||
# can skip (#74136), but they are not evidence of a 429/aux-model fault. The in-conversation
|
||||
# compressor has its own budget and must still be allowed to run (#86972).
|
||||
if _is_hygiene_preagent_only_cooldown(state.get("error")):
|
||||
self._summary_failure_cooldown_until, self._last_summary_error = 0.0, None
|
||||
return None
|
||||
@@ -1994,6 +2094,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
def _record_compression_failure_cooldown(self, cooldown_seconds: float, error: Optional[str]) -> None:
|
||||
# Never shorten a longer live deadline; record the latest error text only.
|
||||
self._summary_failure_cooldown_until = max(self._summary_failure_cooldown_until, time.monotonic() + float(cooldown_seconds))
|
||||
# A later stall or timeout records the latest error text but keeps the later of the two clocks. See
|
||||
# #96775.
|
||||
self._last_summary_error = error
|
||||
cooldown_until = time.time() + max(0.0, self._summary_failure_cooldown_until - time.monotonic())
|
||||
if not getattr(self, "_session_db", None) or not getattr(self, "_session_id", ""):
|
||||
@@ -2020,6 +2122,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
|
||||
def _compression_cancelled(self) -> bool:
|
||||
"""Read the host-owned cooperative cancellation signal, if installed."""
|
||||
# #76354 review F4: fence check BEFORE cooldown-clear. A late worker whose host already timed out
|
||||
# (and recorded a timeout cooldown) must not undo that cooldown when its summary eventually
|
||||
# succeeds. The hook is installed by compress_context for the duration of the fenced call; when it
|
||||
# reports cancellation, keep the host's cooldown.
|
||||
cancelled_check = getattr(self, "_compression_cancelled_check", None)
|
||||
if not callable(cancelled_check):
|
||||
return False
|
||||
@@ -2042,6 +2148,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._base_threshold_percent = resolve_model_threshold(model, self.model_thresholds, _config_pct)
|
||||
self.threshold_percent = self._effective_threshold_percent(context_length, self._base_threshold_percent)
|
||||
# max_tokens=None means "unspecified": keep the existing output reservation.
|
||||
# A switch that genuinely changes the output budget passes the new value explicitly. (#43547)
|
||||
if max_tokens is not None:
|
||||
self.max_tokens = self._coerce_max_tokens(max_tokens)
|
||||
self.threshold_tokens = self._compute_threshold_tokens(context_length, self.threshold_percent, self.max_tokens)
|
||||
@@ -2072,6 +2179,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
_MIN_CTX_TRIGGER_RATIO = 0.85
|
||||
|
||||
# Anti-thrash recovery: after this long blocked, allow ONE probe (counters drop to 1 strike).
|
||||
# Anti-thrash recovery window (#14694): once the ineffective/fallback breaker trips, automatic
|
||||
# compaction stays blocked for this long, then ONE probe attempt is allowed (counters drop to 1 strike,
|
||||
# so another ineffective pass re-trips immediately). Long enough that a genuinely incompressible session
|
||||
# isn't compacting in a loop; short enough that a session which has since grown real compressible
|
||||
# material recovers well before it rides into the provider's hard context limit.
|
||||
_ANTI_THRASH_RECOVERY_SECONDS = 300.0
|
||||
|
||||
# Structural no-op (nothing eligible) is not an ineffective attempt: defer retries instead of striking.
|
||||
@@ -2109,7 +2221,21 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
) -> int:
|
||||
"""Compute the compaction trigger in tokens from the effective input budget.
|
||||
Base is ``(context_length - max_tokens) * threshold_percent`` floored at MINIMUM_CONTEXT_LENGTH;
|
||||
when the floor binds it is capped at 85% of the budget so small windows can still fire."""
|
||||
when the floor binds it is capped at 85% of the budget so small windows can still fire.
|
||||
|
||||
The base value is ``effective_input_budget * threshold_percent``, floored at
|
||||
``MINIMUM_CONTEXT_LENGTH`` so large-context models don't compress prematurely at 50%. BUT that floor
|
||||
degenerates at small windows: for a model whose ``context_length`` is at/below the minimum (e.g. a
|
||||
64K local model), ``max(0.5*64000, 64000) == 64000`` makes the threshold equal the ENTIRE window —
|
||||
auto-compression can never fire because the provider rejects the request before usage reaches 100%
|
||||
(#14690).
|
||||
The provider reserves ``max_tokens`` of output space out of the same window, so the usable INPUT
|
||||
budget is ``context_length - max_tokens``. With a large ``max_tokens`` (e.g. 65536 on a custom
|
||||
provider) the input budget is materially smaller than the raw window, and a threshold based on the
|
||||
full window lets the session hit a provider 400 before compaction fires (#43547). The percentage and
|
||||
the degenerate-window check below both operate on the effective input budget. ``max_tokens=None``
|
||||
(provider default) conservatively assumes no reservation (full window).
|
||||
"""
|
||||
effective_window = context_length - (max_tokens or 0)
|
||||
if effective_window <= 0:
|
||||
effective_window = context_length
|
||||
@@ -2166,6 +2292,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
# Usable input = context_length - max_tokens; only a positive int counts as a reservation.
|
||||
self.max_tokens = self._coerce_max_tokens(max_tokens)
|
||||
# True: summary failure aborts (messages unchanged); False: insert deterministic handoff and drop middle.
|
||||
# Output-token reservation: the provider carves max_tokens out of the context window, so the usable
|
||||
# input budget is context_length - max_tokens. None = provider default => assume no reservation.
|
||||
# (#43547) Coerce defensively: only a positive int is a real reservation; any other value (None,
|
||||
# non-numeric, <=0) means "no reservation" so the threshold arithmetic never sees a non-int (e.g. a
|
||||
# test MagicMock).
|
||||
self.abort_on_summary_failure = abort_on_summary_failure
|
||||
|
||||
# Micro-compaction is OFF by default: each pass breaks the prompt-cache prefix every turn.
|
||||
@@ -2173,18 +2304,30 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._reset_micro_compact_cursor_state()
|
||||
self._micro_compact_defrag_threshold_tokens = 2000
|
||||
# Set when _defrag_rolling_summary pops _DB_PERSISTED_MARKER in place; finalize_turn resets the flush cursor.
|
||||
# Set by _defrag_rolling_summary when it pops _DB_PERSISTED_MARKER from a live dict in place;
|
||||
# consumed by finalize_turn to invalidate the agent's bounded flush-scan cursor (sibling of the
|
||||
# #75170 site).
|
||||
self._flush_scan_cursor_invalidated: bool = False
|
||||
self._micro_compact_passes = self._micro_compact_tokens_saved_total = self._micro_compact_turns_since_pass = 0
|
||||
# Cadence dial: how often the cache-breaking pass is paid. 1 = every turn.
|
||||
self._micro_compact_every_n_turns: int = 1
|
||||
# Deferred: get_model_context_length() may issue a sync HTTP probe that must not block construction.
|
||||
# Floor and cap are applied on first resolution (see _resolve_context_length / threshold_tokens).
|
||||
# The small-context threshold floor and the absolute threshold cap both need the resolved window, so
|
||||
# they are applied on first resolution (see _resolve_context_length / the threshold_tokens property)
|
||||
# instead of here. update_model() re-derives the floor for a new window from
|
||||
# _config_threshold_percent (the raw config value snapshotted above), so switching small -> large
|
||||
# correctly drops back to the configured value. See #32221.
|
||||
self._config_context_length = config_context_length
|
||||
self._configured_threshold_percent = self.threshold_percent
|
||||
self._resolved_context_length: int | None = None
|
||||
self._threshold_tokens = self._tail_token_budget = self._max_summary_tokens = None
|
||||
self.compression_count = 0
|
||||
# The init log reports resolved budgets; emit it on first resolution to keep construction non-blocking.
|
||||
# The "initialized" log reports resolved token budgets, which would force the deferred
|
||||
# get_model_context_length() probe to run inside __init__ and re-introduce the exact synchronous
|
||||
# blocking this change removes (#32221). Emit it on first context-length resolution instead so
|
||||
# construction stays non-blocking on every path (not just quiet).
|
||||
self._log_init_summary = not quiet_mode
|
||||
self._context_probed = False # True after a step-down from context error
|
||||
self.last_prompt_tokens = self.last_completion_tokens = 0
|
||||
@@ -2227,6 +2370,16 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._pending_request_rough_tokens = 0
|
||||
# Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count,
|
||||
# not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it.
|
||||
# Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's
|
||||
# real prompt count for the just-compacted conversation. Effectiveness is "did the prompt get
|
||||
# under the threshold?", not "did the message list shrink?": compaction can only shrink
|
||||
# messages, while the system prompt and tool schemas are an incompressible floor (with 50+
|
||||
# tools, 20-30K tokens — see #14695). When that floor alone meets the threshold, every pass
|
||||
# shrinks messages by a healthy margin yet leaves the prompt over the line, so the next turn
|
||||
# compacts again, forever. It must NOT live in should_compress(): that runs twice per turn with
|
||||
# two different measures (a rough preflight estimate and the real post-response count, #36718),
|
||||
# and the rough one can dip below the threshold and reset the strike every turn, re-opening the
|
||||
# loop. Keying on real usage compares like with like and fires exactly once per compaction.
|
||||
if self._verify_compaction_cleared_threshold:
|
||||
if self.last_prompt_tokens >= self.threshold_tokens:
|
||||
self._record_ineffective_compression_verdict(self._ineffective_compression_count + 1)
|
||||
@@ -2351,6 +2504,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
# probe by dropping counters to 1 strike (persisted). Deadline is armed lazily and persisted on the row.
|
||||
if self._tripped():
|
||||
# Wall clock: the deadline is persisted so a rebuilt compressor resumes the SAME window.
|
||||
# Wall clock, not monotonic: the deadline is persisted on the session row (#100185) so a fresh
|
||||
# compressor bound to the same session — the gateway rebuilds the AIAgent on every cache
|
||||
# eviction — resumes the SAME window instead of restarting it. Without that, a blocked messaging
|
||||
# session never earned its probe and stayed blocked forever.
|
||||
_now = time.time()
|
||||
if self._anti_thrash_recovery_deadline <= 0.0 or (
|
||||
# Clock jumped backwards: never wait longer than one window from now.
|
||||
@@ -2359,6 +2516,20 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._set_anti_thrash_recovery_deadline(_now + self._ANTI_THRASH_RECOVERY_SECONDS)
|
||||
elif _now >= self._anti_thrash_recovery_deadline:
|
||||
self._set_anti_thrash_recovery_deadline(0.0)
|
||||
# Anti-thrashing: back off if recent compressions were ineffective. The back-off must not be
|
||||
# permanent (#14694): the tripped state was judged against the transcript as it existed THEN
|
||||
# (e.g. a middle region too small to matter), but the conversation keeps growing and can
|
||||
# accumulate plenty of compressible material later. Without a recovery path the session
|
||||
# never auto-compacts again and rides into the provider's hard context limit. Recovery is a
|
||||
# probation probe: after _ANTI_THRASH_RECOVERY_SECONDS of continuous block, allow ONE
|
||||
# attempt by dropping the tripped counter(s) to 1 strike (persisted, so sibling agents on
|
||||
# the same session row unblock too). If the probe is ineffective again the very next verdict
|
||||
# re-trips the guard, so the worst case in the truly-incompressible state is one compaction
|
||||
# attempt per recovery window — bounded, not thrash. The clock is armed lazily on the first
|
||||
# BLOCKED evaluation and persisted on the session row (#100185): a fresh process/compressor
|
||||
# that loads a durable tripped counter (#69872) with no stored deadline starts a full window
|
||||
# blocked, preserving the restart-must-not-disarm contract (#54923) — but one that loads an
|
||||
# already-armed deadline resumes that window instead of restarting it.
|
||||
if self._ineffective_compression_count >= 2:
|
||||
self._record_ineffective_compression_verdict(1)
|
||||
if self._fallback_compression_streak >= 2:
|
||||
@@ -2548,6 +2719,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
prune_boundary = self._prune_boundary(result, protect_tail_count, protect_tail_tokens)
|
||||
pruned = self._dedupe_tool_results(result)
|
||||
# Just-loaded / tail-referenced skills keep full skill_view bodies through the ordinary passes.
|
||||
# Without this, a skill loaded moments before a compaction can be demoted to metadata while the
|
||||
# model still believes its instructions are in context. See #32106.
|
||||
protected_skills = _collect_protected_skill_names(result, prune_boundary)
|
||||
# Pass 2: summarize old tool results. Pass 3: shrink large tool_call arguments INSIDE the parsed JSON so
|
||||
# the result stays valid; otherwise providers 400 on every turn until the call leaves the window.
|
||||
@@ -2559,6 +2732,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
self._truncate_tool_call_args_at(result, i)
|
||||
# Pass 3.5: retire image payloads inside the protected tail; re-sent embeds otherwise make
|
||||
# compression look ineffective and trip anti-thrash. Newest frames stay live.
|
||||
# Newest frames stay live for follow-up QA; older ones become placeholders. See #92699.
|
||||
pruned += _retire_stale_tool_result_images(result)
|
||||
if protect_tail_tokens is not None and protect_tail_tokens > 0 and result:
|
||||
pruned += self._pressure_demote_tail(
|
||||
@@ -2645,7 +2819,18 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
object as ``(messages, 0)``. The rearm gate is measured on message bodies only, so it is
|
||||
bypassed (never the reclaim gate) when a provider-billed ``current_tokens`` reading already
|
||||
puts the request over ``threshold_tokens`` (#101889); every no-op taken while over threshold
|
||||
is logged once per distinct reason."""
|
||||
is logged once per distinct reason.
|
||||
|
||||
``_prune_old_tool_results`` runs all deterministic passes: (1) dedup byte-identical tool results —
|
||||
keeps the newest full copy and back-references older exact duplicates ANYWHERE in the list
|
||||
(including the protected tail), so no unique content is ever lost; (2) summarize non-tail tool
|
||||
results larger than ``min_prune_chars``; (3) truncate oversized tool_call arguments on non-tail
|
||||
assistant messages; (3.5) retire image payloads on all but the newest ``_MAX_KEEP_TOOL_IMAGES``
|
||||
image-bearing tool results — tail-agnostic and lossy by design (#92699). Only pass (2)'s floor is
|
||||
raised by ``proactive_prune_min_result_chars``; passes (1) and (3) keep their own fixed floors. The
|
||||
recent-tail protection applies to passes (2) and (3); pass (1) is tail-agnostic by design because
|
||||
dedup is lossless.
|
||||
"""
|
||||
if self.proactive_prune_tokens <= 0 or (
|
||||
current_tokens is not None and current_tokens < self.proactive_prune_tokens
|
||||
):
|
||||
@@ -2689,6 +2874,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine):
|
||||
logger.warning("Proactive tool-result prune DB commit failed; keeping the original transcript: %s", exc)
|
||||
return messages, 0
|
||||
# Shared post-commit stamp site with the in-place commit and micro-compaction sync.
|
||||
# See #98450.
|
||||
stamp_db_persisted_markers(pruned_msgs)
|
||||
self._proactive_prune_rearm_tokens = next_rearm_tokens
|
||||
# Reclamation just ran: let a future lockout warn again.
|
||||
@@ -2864,6 +3050,10 @@ None recoverable from deterministic fallback.
|
||||
## Critical Context
|
||||
Summary generation was unavailable, so this is a best-effort deterministic fallback for {len(turns_to_summarize)} compacted message(s).{reason_text}"""
|
||||
# Per-turn truncation cuts [SKILL_PRUNED] markers; re-derive from raw turns and re-inject.
|
||||
# Ghost-skill defense (#32106): the fallback's per-turn truncation (``_FALLBACK_TURN_MAX_CHARS``)
|
||||
# routinely cuts [SKILL_PRUNED: ...] markers out of the compacted turns. Re-derive the ghosted
|
||||
# skills from the raw turn contents and re-inject deterministically, exactly like the LLM-summary
|
||||
# path.
|
||||
_pruned_names = _collect_ghosted_skill_names(turns_to_summarize)
|
||||
del _pruned_names[_MAX_PRUNED_SKILL_MARKERS:]
|
||||
summary = self._with_summary_prefix(_redact_compaction_text(body.strip()))
|
||||
@@ -3001,6 +3191,10 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
|
||||
call_kwargs["model"] = self.summary_model
|
||||
# Pinned route (stall fallback) overrides task routing so the retry leaves the stalled backend.
|
||||
call_kwargs.update(_pinned_summary_call_kwargs())
|
||||
# Compression is atomic: protect the in-flight summary call from a mid-turn gateway interrupt.
|
||||
# Without this, an incoming user message aborts the summary and compression falls back to a degraded
|
||||
# static marker, losing the real handoff (#23975). Re-entrant: a main-model retry (_generate_summary
|
||||
# recursion) re-enters harmlessly.
|
||||
_aux_call_start = time.monotonic()
|
||||
_latency_info: Dict[str, int] = {"prompt_build_ms": max(0, int((_aux_call_start - prompt_started_at) * 1000))}
|
||||
call_kwargs["latency_info"] = _latency_info
|
||||
@@ -3026,8 +3220,25 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
|
||||
# Reasoning-field fallback (DeepSeek/Qwen/Kimi put the summary in reasoning_content); capped.
|
||||
content = extract_content_or_reasoning(response, max_reasoning_chars=8000)
|
||||
where = f"(provider={self.provider or 'auto'} model={self.summary_model or self.model})"
|
||||
# Some OpenAI-compatible proxies (e.g. cmkey.cn, one-api channels) return a well-formed HTTP 200
|
||||
# with an empty or whitespace-only ``content`` instead of an error or empty ``choices``. That
|
||||
# payload passes ``_validate_llm_response`` (a ``message`` exists), so it reaches here and would
|
||||
# otherwise be stored as a prefix-only summary with no body — silently wiping the compacted turns
|
||||
# and making the model forget the in-progress task (#11978, #11914). Treat empty content as a
|
||||
# failure so it routes through the same main-model fallback + cooldown machinery as a transport
|
||||
# error, rather than replacing real context with an empty summary.
|
||||
if not content.strip():
|
||||
raise RuntimeError(f"Context compression LLM returned empty content {where}")
|
||||
# A finish_reason of "length" means the summarizer hit its output token cap mid-generation: the text
|
||||
# present is PARTIAL. Persisting a partial summary as the compaction checkpoint silently truncates
|
||||
# the conversation's memory — the cut-off text replaces the real middle turns AND is fed back into
|
||||
# every subsequent iterative update prompt, compounding the loss across compactions. Treat it as a
|
||||
# failure so it routes through the same main-model fallback + abort machinery as other degraded
|
||||
# responses instead of becoming a checkpoint. (Ported from earendil-works/pi#7048.)
|
||||
# A length stop means the merged rolling summary is partial — persisting it would silently drop the
|
||||
# tail of the merge and feed the cut-off text into every later micro-compact pass. Leave the
|
||||
# exchange unabsorbed instead; a later pass retries it. (Same class as _generate_summary's guard;
|
||||
# pi#7048.)
|
||||
if _response_finish_reason(response) == "length":
|
||||
raise RuntimeError(
|
||||
f"Context compression summary was truncated ({_TRUNCATED_SUMMARY_MARKER}): generation hit the output "
|
||||
@@ -3046,6 +3257,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
|
||||
# bypass_cooldown: provider-proven overflow gets ONE real attempt while armed.
|
||||
if prompt_started_at < self._summary_failure_cooldown_until and not bypass_cooldown:
|
||||
logger.debug(
|
||||
# See #100661.
|
||||
"Skipping context summary during cooldown (%.0fs remaining)",
|
||||
self._summary_failure_cooldown_until - prompt_started_at,
|
||||
)
|
||||
@@ -3076,6 +3288,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
|
||||
# The summarizer may echo secrets verbatim; redact the output too.
|
||||
summary = _redact_compaction_text(content.strip())
|
||||
# Restore any [SKILL_PRUNED] marker the summarizer paraphrased away.
|
||||
# See #32106.
|
||||
summary = _reinject_pruned_skill_markers(summary, _pruned_skill_names)
|
||||
summary = self._ground_historical_task_snapshot(summary, turns_to_summarize)
|
||||
summary = self._augment_summary_lean(summary, turns_to_summarize)
|
||||
@@ -3229,6 +3442,12 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
"""Classify a summary-call failure; retry once on the main model (returning its result) or arm a cooldown (None)."""
|
||||
# Only a genuine no-provider RuntimeError gets the long cooldown; empty/invalid-response
|
||||
# RuntimeErrors are transient and must get the main-model retry below first.
|
||||
# ``call_llm`` raises ``RuntimeError`` for two very different cases: 1. 2. An empty/invalid response
|
||||
# from a configured provider (``_validate_llm_response`` empty-``choices``/``None``, or our
|
||||
# empty-``content`` guard above) — a transient/proxy fault that should fall back to the main model
|
||||
# first, exactly like the transport errors handled below. Only (1) belongs in the long no-provider
|
||||
# cooldown; (2) and every other exception flow into the generic fallback logic so they get a
|
||||
# main-model retry before any cooldown. (#11978, #11914)
|
||||
if isinstance(e, RuntimeError) and "no llm provider configured" in str(e).lower():
|
||||
self._record_compression_failure_cooldown(_SUMMARY_FAILURE_COOLDOWN_SECONDS, "no auxiliary LLM provider configured")
|
||||
self._last_summary_error = "no auxiliary LLM provider configured"
|
||||
@@ -3269,6 +3488,11 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
# Terminal network/empty-content failure after any fallback: flag so compress() ABORTS
|
||||
# and preserves the session; independent of abort_on_summary_failure.
|
||||
if kind.streaming_closed:
|
||||
# A terminal connection/network failure or empty-content response from a degraded provider (we
|
||||
# reach this branch only after any main-model fallback has already been tried or is
|
||||
# unavailable). Flag it so compress() ABORTS and preserves the session unchanged instead of
|
||||
# destroying the middle window for a placeholder marker — retrying once the provider recovers is
|
||||
# strictly better than dropping context (#29559, #25585, #94448).
|
||||
self._last_summary_network_failure = True
|
||||
elif kind.truncated:
|
||||
self._last_summary_truncated_failure = True
|
||||
@@ -3494,6 +3718,8 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
"""Find handoff summaries inside a compression window."""
|
||||
n = len(messages)
|
||||
# Clamp: callers may pass end = len(messages)+1.
|
||||
# Defensive: clamp bounds so a caller passing an out-of-range end (e.g. tail-cut returning
|
||||
# len(messages)+1 when head_end >= n) cannot trigger IndexError. (#75588)
|
||||
start = max(0, min(start, n))
|
||||
end = max(start, min(end, n))
|
||||
return [
|
||||
@@ -3578,7 +3804,12 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
|
||||
@staticmethod
|
||||
def _tool_call_id_variants(tc) -> set:
|
||||
"""Return every id variant a result might reference *tc* by (forwards to message_sanitization)."""
|
||||
"""Return every id variant a result might reference *tc* by (forwards to message_sanitization).
|
||||
|
||||
Thin forwarder — the policy owner is ``agent.message_sanitization.tool_call_id_variants``, which
|
||||
also expands ``response_item_id`` and composite ``call|item`` bridge spellings (#63000), so the
|
||||
compressor's pairing tolerance matches the pre-call sanitizer's exactly and the two can never drift.
|
||||
"""
|
||||
from agent.message_sanitization import tool_call_id_variants
|
||||
return set(tool_call_id_variants(tc))
|
||||
|
||||
@@ -3648,7 +3879,11 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
return self.protect_first_n
|
||||
|
||||
def _protect_head_size(self, messages: List[Dict[str, Any]]) -> int:
|
||||
"""Head messages to protect: the system prompt (if present) plus the decaying ``protect_first_n`` extra rows."""
|
||||
"""Head messages to protect: the system prompt (if present) plus the decaying ``protect_first_n`` extra rows.
|
||||
|
||||
The ``protect_first_n`` portion DECAYS after the first compression (see _effective_protect_first_n)
|
||||
so early user turns don't fossilize across repeated compactions (#11996).
|
||||
"""
|
||||
head = 1 if messages and messages[0].get("role") == "system" else 0
|
||||
return head + self._effective_protect_first_n(messages)
|
||||
|
||||
@@ -3762,6 +3997,16 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
from agent.conversation_compression import _is_real_user_message
|
||||
|
||||
last_user_idx = -1
|
||||
# Find the newest user message that carries at least one image part. We anchor on image-bearing user
|
||||
# messages (not all user messages) so a plain text follow-up after a big-image turn still strips the
|
||||
# old image — matching the problem kilocode#9434 set out to solve.
|
||||
# Newest tool message carrying an image. Tool-result images (``vision_analyze``,
|
||||
# screenshot-returning tools) accumulate on their own timeline and the user anchor never protects
|
||||
# the stale ones: a session whose only image-bearing user message is the FIRST one leaves ``anchor
|
||||
# <= 0`` and strips nothing at all, so twenty tool results keep multi-MB of base64 in every request
|
||||
# body until the provider answers 413 -- and the 413 handler's recovery compaction lands right back
|
||||
# here and frees nothing, which is the wedge in #89938. Keep the newest tool image, since that is
|
||||
# the one the model is reasoning about, and drop every older one wherever it sits.
|
||||
for i in range(len(messages) - 1, -1, -1):
|
||||
msg = messages[i]
|
||||
# _is_real_user_message also rejects metadata-flagged scaffolding
|
||||
@@ -3898,7 +4143,16 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
def _ensure_last_n_user_messages_in_tail(
|
||||
self, messages: List[Dict[str, Any]], cut_idx: int, head_end: int, n: int,
|
||||
) -> int:
|
||||
"""Keep the last N actionable user messages in the tail; n <= 1 delegates to the single-message method."""
|
||||
"""Keep the last N actionable user messages in the tail; n <= 1 delegates to the single-message method.
|
||||
|
||||
Only REAL actionable user turns count toward N — the collector uses the same
|
||||
``_is_actionable_user_turn`` / ``_is_synthetic_compression_user_turn`` pair as
|
||||
``_find_last_user_message_idx``, so blank platform echoes, compaction handoffs, continuation
|
||||
markers, and todo-snapshot rows never consume a slot (#69291 bug class).
|
||||
A user message is already a clean boundary — there is no tool_call/result group that spans across
|
||||
it, so ``_align_boundary_backward`` is intentionally NOT called. Calling it can pull the cut past
|
||||
the user message into the preceding assistant(tool_calls)→tool group and split it (#22566).
|
||||
"""
|
||||
if n <= 1:
|
||||
return self._ensure_last_user_message_in_tail(messages, cut_idx, head_end)
|
||||
|
||||
@@ -3957,6 +4211,8 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
cut_idx = self._align_boundary_backward(messages, cut_idx)
|
||||
# Latest user message must stay in the tail (active task). Latest assistant reply must stay too;
|
||||
# anchors only walk backward, so chaining is monotonic.
|
||||
# Ensure the most recent user message is always in the tail so the active task is never lost to
|
||||
# compression (fixes #10896).
|
||||
cut_idx = self._ensure_last_user_message_in_tail(messages, cut_idx, head_end)
|
||||
cut_idx = self._ensure_last_assistant_message_in_tail(messages, cut_idx, head_end)
|
||||
|
||||
@@ -4275,6 +4531,10 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
self.compression_count += 1
|
||||
# Replace historical image payloads with placeholders; multi-MB base64 blobs otherwise
|
||||
# exceed body limits.
|
||||
# Replace image parts in all compressed messages before the newest image-bearing user turn with a
|
||||
# short text placeholder. Without this, tail messages keep their original multi-MB base-64 image
|
||||
# payloads forever, which can push every subsequent API request past the provider's body-size limit
|
||||
# and wedge the session. Port of Kilo-Org/kilocode#9434.
|
||||
compressed = _strip_historical_media(compressed)
|
||||
|
||||
# Like-for-like savings: current_tokens includes system prompt/tool schemas, new_estimate is
|
||||
@@ -4300,6 +4560,10 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
# Compaction frees the biggest allocation: hand pages back to the OS (glibc/config-gated,
|
||||
# rate-limited, #70782). debug, not warning: compression must never fail because of a trim.
|
||||
try:
|
||||
# A successful compaction just freed the largest allocation a long session ever drops (the
|
||||
# compressed-away message dicts), which makes this the natural point to hand allocator pages
|
||||
# back to the OS. #76905's trim lifecycle covers the gateway/TUI housekeeping loops but not the
|
||||
# CLI compression path, so RSS keeps the pre-compaction high-water mark until exit. (#70782)
|
||||
from hermes_cli.mem_trim import trim_memory
|
||||
trim_memory(reason="post-compression")
|
||||
except Exception as exc:
|
||||
@@ -4317,7 +4581,19 @@ Write only the summary body. Do not include any preamble or prefix."""
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Summarize the middle turns: prune tool results and blank echoes (survives an abort), protect head and a
|
||||
token-budget tail, summarize, clean orphaned tool pairs. ``force`` clears the failure cooldown and bypasses
|
||||
the feasibility skip; ``bypass_cooldown`` runs the summary LLM without clearing the cooldown."""
|
||||
the feasibility skip; ``bypass_cooldown`` runs the summary LLM without clearing the cooldown.
|
||||
|
||||
Args: focus_topic: Optional focus string for guided compression. When provided, the summariser will
|
||||
prioritise preserving information related to this topic and be more aggressive about compressing
|
||||
everything else. Inspired by Claude Code's ``/compact``. force: If True, clear any active
|
||||
summary-failure cooldown before running so a manual ``/compress`` can retry immediately after an
|
||||
auto-compression abort, and bypass the pre-LLM feasibility skip so an explicit user request always
|
||||
exercises the full summary path. Auto-compress callers pass False. memory_context: Optional
|
||||
provider-supplied context to preserve in the summary prompt. Whitespace-only values are ignored.
|
||||
bypass_cooldown: If True, run the summary LLM even while the summary-failure cooldown is armed,
|
||||
WITHOUT clearing it (#100661). Set by provider-proven overflow recovery, which is already bounded by
|
||||
the caller's attempt budget.
|
||||
"""
|
||||
telemetry = self._begin_compress_attempt(current_tokens, force)
|
||||
n_messages = len(messages)
|
||||
# Only need head + 3 tail messages minimum (token budget decides the real tail size)
|
||||
|
||||
@@ -62,6 +62,8 @@ class ContextEngine(ABC):
|
||||
# Compaction parameters (read by run_agent.py for preflight). protect_first_n counts
|
||||
# non-system head messages kept verbatim IN ADDITION to the always-protected system
|
||||
# prompt (3 keeps the historical head shape).
|
||||
# These control the preflight compression check. Subclasses may override via __init__ or property;
|
||||
# defaults are sensible for most engines. See #13754.
|
||||
threshold_percent: float = 0.75
|
||||
protect_first_n: int = 3
|
||||
protect_last_n: int = 6
|
||||
@@ -183,6 +185,16 @@ class ContextEngine(ABC):
|
||||
|
||||
def on_session_reset(self) -> None:
|
||||
"""/new or /reset: reset per-session state (default: counters and token tracking)."""
|
||||
# Reset cross-call calibration state captured under the PREVIOUS model. These fields encode "the
|
||||
# provider proved this prompt fit" / "preflight can be deferred" decisions that are only valid for
|
||||
# the model that produced them. Carrying them across a switch to a smaller-context model would let
|
||||
# should_defer_preflight_to_real_usage() suppress a preflight compression the new model actually
|
||||
# needs — the exact oversized-send-after-switch failure in #23767. The new model's first response
|
||||
# repopulates them via update_from_response(). Setting last_prompt_tokens to 0 (NOT -1) is
|
||||
# deliberate: 0 is the documented "no real usage yet -> use the rough estimate" state, so the post-
|
||||
# response should_compress path falls back to estimate_request_tokens_rough rather than skipping
|
||||
# compression. -1 is a different sentinel (#36718, "compression just ran, await real usage") and
|
||||
# must not be set here.
|
||||
self.last_prompt_tokens = 0
|
||||
self.last_completion_tokens = 0
|
||||
self.last_total_tokens = 0
|
||||
|
||||
@@ -20,6 +20,8 @@ from hermes_cli.sizefmt import format_bytes
|
||||
|
||||
# ── Plugin context-reference provider API ────────────────────────────────────
|
||||
|
||||
# --------------------------------------------------------------------------- Plugin context-reference
|
||||
# provider API (Issue #26193) ---------------------------------------------------------------------------
|
||||
BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"})
|
||||
|
||||
_context_reference_providers: dict[str, "ContextReferenceProvider"] = {}
|
||||
|
||||
@@ -53,6 +53,8 @@ _TERMINAL_COMPRESSION_PROVENANCES = frozenset(
|
||||
|
||||
# Split failures are usually transient lease/DB conditions, so use the FIRST
|
||||
# timeout-ladder rung (60s), not the 600s summary-provider cooldown.
|
||||
# Cooldown armed when a compression SPLIT fails (session_split_failed / rotation rollback, #97948 symptom
|
||||
# B).
|
||||
_SPLIT_FAILURE_COOLDOWN_SECONDS = 60
|
||||
|
||||
# Marker tui_gateway/server.py::_status_update matches to tag kind="compacting" for drivers' "Summarizing…" UI. Keep
|
||||
@@ -108,6 +110,11 @@ COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE = (
|
||||
|
||||
# FAILURE-class notice: compression blocked, so the session grows until the provider limit kills it. Must stay visible
|
||||
# on gateways: never add it to ROUTINE_COMPRESSION_STATUS_SAMPLES or _TELEGRAM_NOISY_STATUS_RE.
|
||||
# FAILURE-CLASS notice — a deliberate carve-out from routine-compression silence (#16775 class): the context
|
||||
# is over the compression threshold but compression is blocked (summary-LLM cooldown / anti-thrash breaker),
|
||||
# so the session will keep growing until the hard provider token limit kills it. Do NOT add it to
|
||||
# ROUTINE_COMPRESSION_STATUS_SAMPLES or the gateway noise regex (_TELEGRAM_NOISY_STATUS_RE); it is pinned
|
||||
# un-swallowed in tests/gateway/test_telegram_noise_filter.py::VISIBLE_COMPRESSION_MESSAGES.
|
||||
CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = (
|
||||
"⚠ Context is over the compression threshold (~{tokens:,} tokens >= {threshold:,}) "
|
||||
"but compression is currently blocked ({reason}). The model may stop responding. Run /new to start a fresh "
|
||||
@@ -188,6 +195,19 @@ def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]:
|
||||
# Attempt ownership: stall-fallback detaches a timed-out worker and reuses the compressor, so its late unwind could
|
||||
# restore a stale snapshot or clear the fallback's cancel check. Generation guards ATTRIBUTE writes; fence, COMMITs.
|
||||
|
||||
# --------------------------------------------------------------------------- Attempt ownership (#96634
|
||||
# follow-up). The stall-fallback path deliberately DETACHES a timed-out primary worker (fence cancel wins;
|
||||
# the future stays on the shared pool) and immediately starts a fallback attempt against the SAME
|
||||
# ContextCompressor. Two races follow from that overlap: 1. The late primary's unwind still calls
|
||||
# _restore_compressor_attempt_state with the PRIMARY's pre-attempt snapshot. Landing after the fallback's
|
||||
# commit, it rolls _previous_summary / cooldown / provenance / telemetry back to pre-primary values —
|
||||
# silently discarding fallback-owned state. 2. _compression_cancelled_check is one shared attribute: the
|
||||
# late primary's ``finally`` clears the callback the fallback just installed, so the fallback's F4
|
||||
# cancellation consult reads None. Both are fixed with a monotonic per-compressor attempt generation,
|
||||
# claimed under one module lock. Restores and callback set/clear are keyed to the claiming generation and
|
||||
# no-op when a newer attempt owns the compressor. The commit fence still owns COMMIT admission; the
|
||||
# generation owns compressor-ATTRIBUTE writes — two different boundaries.
|
||||
# ---------------------------------------------------------------------------
|
||||
_COMPRESSOR_ATTEMPT_LOCK = threading.Lock()
|
||||
|
||||
|
||||
@@ -268,7 +288,11 @@ def _restore_compressor_attempt_state(
|
||||
) -> None:
|
||||
"""Restore the per-attempt snapshot after a pre-commit hard cancel.
|
||||
A restore stamped with a stale ``attempt_generation`` no-ops so a timed-out primary's late unwind cannot
|
||||
roll back state owned by the fallback attempt."""
|
||||
roll back state owned by the fallback attempt.
|
||||
|
||||
``attempt_generation`` (when provided) is the claim the calling attempt took via
|
||||
:func:`_claim_compressor_attempt`. See #96634.
|
||||
"""
|
||||
if attempt_generation is not None and not _compressor_attempt_is_current(compressor, attempt_generation):
|
||||
logger.warning(
|
||||
"Skipping stale compressor attempt-state restore: attempt "
|
||||
@@ -348,10 +372,25 @@ class CompressionCommitFence:
|
||||
self._cancelled = False
|
||||
self._commit_started = False
|
||||
# Readable WITHOUT the lock (begin_commit holds it until finish_commit): hosts see a hung commit.
|
||||
# Lock-free commit-phase marker (#76354 review F1). ``begin_commit`` RETAINS ``self._lock`` until
|
||||
# ``finish_commit``, so any host-side observation that needs the lock (``try_cancel_before_commit``)
|
||||
# blocks/space-outs for the whole commit. This Event is set inside ``begin_commit`` while the lock
|
||||
# is held but is READABLE WITHOUT the lock, so a host can observe "a commit was admitted and may be
|
||||
# in flight" even while the commit itself is hung — which is exactly when the overrun warning must
|
||||
# be able to fire.
|
||||
self._commit_phase = threading.Event()
|
||||
# Set on ANY host unwind without the fence lock so FUTURE commits are blocked; bool store is atomic.
|
||||
# Lock-free admission revocation (#76354 review F2). Set by :meth:`revoke_commit_admission` on ANY
|
||||
# host unwind (KeyboardInterrupt, cancellation, unexpected exception) without touching the fence
|
||||
# lock, so a host that cannot afford to block behind an in-flight commit can still guarantee no
|
||||
# FUTURE commit is admitted.
|
||||
self._admission_revoked = False
|
||||
# Holder-scoped release published by the worker once it owns the durable lock (no ABA on a NEW holder).
|
||||
# Holder-qualified durable-lock release hook (#76354 review F4; transplanted from PR #71569 by
|
||||
# @ciabata-git). The worker publishes an idempotent, holder-scoped release callable once it owns the
|
||||
# durable compression lock; a timed-out host invokes it to free the lease without racing a NEW
|
||||
# holder (DB release is holder-qualified, so a stale release can never delete a replacement's row —
|
||||
# no ABA).
|
||||
self._lock_release_guard = threading.Lock()
|
||||
self._cancelled_lock_release: Optional[Callable[[], None]] = None
|
||||
self._cancelled_lock_release_requested = False
|
||||
@@ -389,7 +428,14 @@ class CompressionCommitFence:
|
||||
|
||||
@property
|
||||
def deadline_monotonic(self) -> float | None:
|
||||
"""Armed deadline (absolute monotonic); the worker's stream consumer stops when the host stops waiting."""
|
||||
"""Armed deadline (absolute monotonic); the worker's stream consumer stops when the host stops waiting.
|
||||
|
||||
:meth:`set_total_ceiling_seconds` documents this deadline as "shared by the host and worker", but
|
||||
until #99692 only the host could read it — ``deadline_exceeded`` answers "is it past?" for a caller
|
||||
that is already polling, which is useless to a worker blocked inside a provider stream. Publishing
|
||||
the instant itself lets the worker's stream consumer stop at exactly the moment the host stops
|
||||
waiting (see ``auxiliary_client.aux_stream_deadline``).
|
||||
"""
|
||||
return self._deadline
|
||||
|
||||
def seconds_since_progress(self) -> float:
|
||||
@@ -454,7 +500,14 @@ class CompressionCommitFence:
|
||||
self._retain_cancelled_lock_until_worker_done = True
|
||||
|
||||
def mark_commit_watermark_fenced(self) -> None:
|
||||
"""Record a watermark-bounded commit (later rows survive as tail); a detached worker may keep admission."""
|
||||
"""Record a watermark-bounded commit (later rows survive as tail); a detached worker may keep admission.
|
||||
|
||||
Called by the compression worker right after it captures ``get_active_message_watermark()`` under
|
||||
the durable compression lock (#75316/#87484). A watermark-fenced commit archives ONLY rows at or
|
||||
below the watermark; rows appended later — e.g. the user turn the host released at the turn-hold
|
||||
boundary (#97963) — are cloned as live concurrent tail. That is exactly the property a host needs
|
||||
before letting a detached worker keep its commit admission.
|
||||
"""
|
||||
self._commit_watermark_fenced = True
|
||||
|
||||
@property
|
||||
@@ -481,6 +534,11 @@ class CompressionCommitFence:
|
||||
# ── Holder-qualified durable-lease cancellation: release is DELETE WHERE
|
||||
# holder = ?, so a stale release can never free a NEW holder's lease (no ABA).
|
||||
|
||||
# ── Holder-qualified durable-lease cancellation (#76354 F4) ────────── Transplanted from PR #71569
|
||||
# (@ciabata-git): the worker publishes an idempotent, holder-scoped release hook once it owns the
|
||||
# durable compression lock, and the host invokes it after winning cancellation. ABA safety comes from
|
||||
# SessionDB.release_compression_lock being holder-qualified (DELETE ... WHERE holder = ?), so a stale
|
||||
# release can never free a NEW holder's lease.
|
||||
def begin_lock_setup(self) -> bool:
|
||||
"""Hold the fence across lock acquisition + release-hook publication so a timeout cannot win between."""
|
||||
self._lock.acquire()
|
||||
@@ -524,6 +582,9 @@ DEFAULT_CONTEXT_TIMEOUT_SECONDS = 120.0
|
||||
DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0
|
||||
|
||||
# Unlike explicit_interrupt, a /stop after the stall window arms the durable backoff (no automatic re-entry).
|
||||
# Distinct from ``explicit_interrupt``: a /stop that arrived after the summary stream had already crossed
|
||||
# the no-progress stall window (#96775). Ordinary early /stop stays cooldown-neutral; this class arms the
|
||||
# durable backoff so the next automatic turn does not re-enter the same stalled strategy.
|
||||
STALL_INTERRUPTED_FAILURE_CLASS = "stall_interrupted"
|
||||
|
||||
# Daemon pool so a fence-cancelled hung worker cannot block interpreter exit; never shut down per call.
|
||||
@@ -535,6 +596,8 @@ _COMMIT_OVERRUN_WAIT_SLICE_SECONDS = 30.0
|
||||
|
||||
# A worker exiting within the grace proves no provider call is in flight, so its lease may be released even
|
||||
# on the total-ceiling path; one that doesn't exit is orphaned behind the poison fence and keeps its lease.
|
||||
# Bounded grace given to a fence-cancelled compression worker to actually exit before the host moves on
|
||||
# (#97488).
|
||||
_CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS = 5.0
|
||||
|
||||
|
||||
@@ -561,6 +624,16 @@ def _join_cancelled_worker(future: Any, grace_seconds: float) -> bool:
|
||||
|
||||
# The executor queue is unbounded and a queued job would run stale, so admission is capped at the worker
|
||||
# count (fail fast, continue uncompressed). Slots free via done-callback; a never-returning worker loses one.
|
||||
# Bounded admission for the shared compress-timeout pool (#76354 review F6). The stdlib executor queue is
|
||||
# unbounded: with all four workers wedged in hung summaries, a fifth compression would queue silently, wait
|
||||
# out its whole timeout without ever starting, and remain eligible to run as a stale job whenever a worker
|
||||
# recovered. Admission is therefore capped at the worker count — when every worker slot is occupied (running
|
||||
# OR admitted-not-started) submission FAILS FAST and the caller continues without compression. Recovery
|
||||
# contract when all workers are wedged: new compressions fail fast (no queue growth, conversation continues
|
||||
# uncompressed, a warning is logged each attempt); wedged workers are fence-cancelled so they cannot publish
|
||||
# anything when they eventually return, and each recovery frees its admission slot via the future
|
||||
# done-callback, restoring normal service. If a worker NEVER returns, its slot is lost for the process
|
||||
# lifetime — bounded, observable degradation instead of an unbounded stale-job queue.
|
||||
_COMPRESS_EXECUTOR_MAX_WORKERS = 4
|
||||
_compress_admission_lock = threading.Lock()
|
||||
_compress_admitted_count = 0
|
||||
@@ -632,7 +705,12 @@ def compression_attempt_stalled(
|
||||
) -> bool:
|
||||
"""Return whether a pre-commit cancel landed after the stall window.
|
||||
An early ``/stop`` stays cooldown-neutral; an interrupt after the inactivity budget counts as a stall so
|
||||
the next automatic turn does not blindly retry."""
|
||||
the next automatic turn does not blindly retry.
|
||||
|
||||
When the fence (or, without a fence, the attempt clock) has already sat idle for the configured
|
||||
compression inactivity budget, the interrupt is a stalled attempt — the same condition the host timeout
|
||||
uses — and the next automatic turn must not blindly retry that strategy (#96775).
|
||||
"""
|
||||
idle = idle_timeout_seconds
|
||||
if idle is None:
|
||||
idle, _ceiling = resolve_context_compression_timeouts()
|
||||
@@ -674,6 +752,8 @@ def _record_stall_interrupted_backoff(
|
||||
if not compression_attempt_stalled(commit_fence=commit_fence, started_at=started_at):
|
||||
return False
|
||||
compressor = getattr(agent, "context_compressor", None)
|
||||
# Same timeout cooldown ladder as summary-LLM timeouts (#62452): avoid re-burning the full idle budget
|
||||
# every turn.
|
||||
record = getattr(compressor, "record_timeout_failure", None)
|
||||
if not callable(record):
|
||||
return False
|
||||
@@ -741,7 +821,17 @@ def _retry_compression_on_fallback_chain(
|
||||
"""Re-run an aborted compression once with the summary route pinned.
|
||||
Returns ``(messages, system_prompt)`` on real compression, else ``None`` and the caller degrades as
|
||||
before. The entry's ``timeout`` sets the idle window. Re-runs the whole worker, so pre-compression
|
||||
callbacks must be idempotent."""
|
||||
callbacks must be idempotent.
|
||||
|
||||
The retry is bounded the same way the primary was: silence for one idle window ends it, while a fallback
|
||||
that is streaming keeps its ceiling. The entry's own ``timeout`` (when declared) sets that idle window,
|
||||
so a fallback tuned for a slower-but-healthy backend is not held to a deadline the stalled primary
|
||||
defined (#62452 semantics, applied to the stall path).
|
||||
Known limitation (accepted, #96634 review): the retry re-runs the COMPLETE worker, which repeats
|
||||
memory/plugin pre-compression callbacks. Built-in callbacks are idempotent (re-reads and overwrites of
|
||||
attempt-scoped state); third-party plugin callbacks are advised to be. Splitting the worker to resume
|
||||
mid-pipeline would couple this path to every host's callback ordering — deliberately out of scope.
|
||||
"""
|
||||
# An explicit stop is not a stalled route. The retry worker would abort on
|
||||
# the same event anyway, but starting one at all makes /stop look ignored.
|
||||
hard_cancel = getattr(telemetry_agent, "_hard_interrupt_requested", None)
|
||||
@@ -930,6 +1020,9 @@ def run_compress_context_with_progress_timeout(
|
||||
executor = _get_compress_timeout_executor()
|
||||
# Refuse rather than queue when the pool is full: a queued job would wait out
|
||||
# its budget unstarted and run stale later. Skip compression this cycle.
|
||||
# A queued job would silently wait out its whole budget without starting and stay eligible to run as a
|
||||
# stale cancelled job when a worker recovers. Fail fast: continue without compression this cycle. See
|
||||
# #76354.
|
||||
if not _try_admit_compression_job():
|
||||
logger.warning(
|
||||
"Context compression pool saturated (%d workers busy) — refusing new compression this cycle and continuing without "
|
||||
@@ -980,6 +1073,14 @@ def run_compress_context_with_progress_timeout(
|
||||
# cancel() is a no-op for a running worker (fence handles that path).
|
||||
future.cancel()
|
||||
total_exhausted = time.monotonic() - wait_started >= ceiling or fence.deadline_exceeded
|
||||
# #97488 teardown (total-ceiling path only): give the cancelled worker a bounded grace to actually
|
||||
# exit before this host moves on. The worker checks the poison fence between provider phases, so a
|
||||
# cooperative worker exits quickly; an uninterruptible provider call is orphaned behind the fence
|
||||
# after the grace elapses (its late result is discarded and cannot touch session state). The
|
||||
# idle-stall path intentionally skips the join: its worker is by definition silent/hung, the
|
||||
# stall-fallback retry below needs a prompt host return (pinned by the #76354 S3 latency contract),
|
||||
# and the fence poison + attempt-generation supersession already protect state against its late
|
||||
# unwind.
|
||||
if total_exhausted:
|
||||
# A total-ceiling candidate may be unwinding a healthy provider call; keep its
|
||||
# lease until it exits so no other attempt overlaps the unchanged source.
|
||||
@@ -999,6 +1100,9 @@ def run_compress_context_with_progress_timeout(
|
||||
handled_exit = True
|
||||
_release_cancelled_worker(future, fence, total_exhausted=total_exhausted, ceiling=ceiling)
|
||||
waited = time.monotonic() - wait_started
|
||||
# #76354 S3 analogue for this wait: charge the idle budget from the LAST PROGRESS event, not from
|
||||
# the start of this wait slice. Waiting a full ``idle`` after progress that landed early in the
|
||||
# previous slice would allow silence to approach 2x the budget.
|
||||
since_progress = fence.seconds_since_progress()
|
||||
# Lease is free, so run the fallback BEFORE on_timeout: that callback records
|
||||
# the summary-failure cooldown, which would no-op the retry's summary call.
|
||||
@@ -1222,7 +1326,15 @@ def compression_blocked_transiently(agent: Any) -> bool:
|
||||
"""Type-pinned read of the transient-block signal.
|
||||
Set when an automatic pass no-ops on a TRANSIENT guard (summary-failure cooldown or structural backoff).
|
||||
Consumers must defer, not count it toward ``compression_exhausted``, or an overflow auto-reset wipes a
|
||||
session that was merely cooling down. The permanent ``ineffective`` breaker never sets it."""
|
||||
session that was merely cooling down. The permanent ``ineffective`` breaker never sets it.
|
||||
|
||||
See #97488.
|
||||
Consumers (the overflow-recovery loops in ``conversation_loop``) must treat such a no-op as a temporary
|
||||
defer, NOT as evidence the session is incompressible: counting it toward ``compression_exhausted`` lets
|
||||
a real upstream ``context_length_exceeded`` auto-reset (wipe) a session whose compression was merely
|
||||
cooling down (#97488). The permanent ``ineffective`` breaker intentionally does NOT set this signal — a
|
||||
genuinely incompressible session must still be able to exhaust.
|
||||
"""
|
||||
_sig = getattr(agent, "_compression_blocked_transient", None)
|
||||
return isinstance(_sig, str) and bool(_sig)
|
||||
|
||||
@@ -1265,7 +1377,14 @@ def _adopt_live_compression_child(
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Move a stale compression contender onto the live continuation tip.
|
||||
Resolve and load first, then mutate the agent, so ambiguous lineage or an unreadable handoff fails closed.
|
||||
Uses the transitive ``get_compression_tip`` walk; a tip is adopted only while its row is still live."""
|
||||
Uses the transitive ``get_compression_tip`` walk; a tip is adopted only while its row is still live.
|
||||
|
||||
Resolution uses the canonical transitive walk ``get_compression_tip`` so a lineage with >=2 compression
|
||||
hops (root -> mid -> tip) recovers to the live tip — the depth-1 ``find_live_compression_child`` lookup
|
||||
this used to call finds no live *direct* child in that shape and skipped recovery (#82001). The tip walk
|
||||
returns the input id when no continuation exists, and a resolved tip is adopted only while its row is
|
||||
still live — both cases fail closed exactly as before.
|
||||
"""
|
||||
resolver = getattr(type(session_db), "get_compression_tip", None)
|
||||
row_getter = getattr(type(session_db), "get_session", None)
|
||||
loader = getattr(type(session_db), "get_messages_as_conversation", None)
|
||||
@@ -1591,6 +1710,12 @@ def _lower_threshold_to_aux_context(
|
||||
safe_pct = int((aux_context / main_ctx) * 100) if main_ctx else 50
|
||||
# Mirror the compressor's threshold math (percent floor, output reservation, 64K floor): a suggestion it
|
||||
# would override is silently ignored and this warning reappears every session. External engines: keep it plain.
|
||||
# The "lower the threshold" suggestion must survive the built-in trigger recomputation (#67422):
|
||||
# _effective_threshold_percent() raises sub-75% values back up for main windows under 512K, and
|
||||
# _compute_threshold_tokens() further applies the output-token reservation, the 64K floor, and the
|
||||
# degenerate-window guard. Recommending a value those would override is silently ignored and this
|
||||
# warning would reappear every session — so mirror the compressor's own math and only offer the option
|
||||
# when the recomputed trigger actually fits the auxiliary model's context.
|
||||
from agent.context_compressor import ContextCompressor as _CC
|
||||
recomputed_threshold = None
|
||||
if main_ctx and isinstance(compressor, _CC):
|
||||
@@ -1979,6 +2104,12 @@ def _ensure_compressed_has_user_turn(original_messages: list, compressed: list)
|
||||
"""Preserve human intent, not merely a synthetic user-role placeholder."""
|
||||
if any(_is_real_user_message(message) for message in compressed) or _compressed_has_busy_steer(compressed):
|
||||
return "already_present"
|
||||
# Post-commit contract (#98450, mirrors _sync_micro_compact_to_db): archive_and_compact just durably
|
||||
# wrote every dict in `compressed` as the new active set, but compress() returned marker-swept COPIES
|
||||
# (_strip_persistence_markers, #57491). These exact dict instances become the live message list the
|
||||
# caller keeps, so without the stamp the next _persist_session → _flush_messages_to_session_db_unlocked
|
||||
# walk treats the whole compacted transcript as unpersisted and re-INSERTs it — the live set doubles on
|
||||
# every compaction (~58K → ~512K tokens in production).
|
||||
from agent.context_compressor import (
|
||||
_INFLIGHT_REPLAY_MERGED_KEY, COMPRESSION_CONTINUATION_USER_CONTENT, _fresh_compaction_message_copy,
|
||||
)
|
||||
@@ -1987,6 +2118,9 @@ def _ensure_compressed_has_user_turn(original_messages: list, compressed: list)
|
||||
return "already_present"
|
||||
# One reversed scan over BOTH kinds: scanning steer then user would let an older
|
||||
# consumed steer outrank a newer real user request and replay it.
|
||||
# One reversed positional scan: the anchor is whichever intent-bearing row is LAST in the original
|
||||
# transcript — a real ``role=user`` turn or a steer marker riding inside a ``role=tool`` result. See
|
||||
# #100053.
|
||||
for message in reversed(original_messages):
|
||||
if _is_real_user_message(message):
|
||||
return _insert_real_user_anchor(compressed, _fresh_compaction_message_copy(message))
|
||||
@@ -2421,6 +2555,13 @@ def _adopt_grown_durable_parent(agent: Any, lease: _CompressionLease, messages:
|
||||
return None
|
||||
# In-memory carries this turn's un-persisted user tail; flush it via the normal
|
||||
# rotation-boundary path before adopting, else skip adoption (would drop input).
|
||||
# The in-memory transcript carries the CURRENT turn's un-persisted user tail (anchored by
|
||||
# _persist_user_message_idx) that the durable snapshot read above does not contain yet. Flush that tail
|
||||
# through the normal rotation-boundary path (conversation_history = the already-durable prefix, #68196
|
||||
# boundary) BEFORE adopting, then re-read the durable parent so the adopted snapshot includes the live
|
||||
# input. If the flush fails (or the anchor is unknown), skip adoption entirely: replacing the in-memory
|
||||
# transcript with a snapshot that lacks the user's input would silently drop it from the summarized and
|
||||
# rotated history (#adopt-live-tail).
|
||||
_preflush_idx = getattr(agent, "_persist_user_message_idx", None)
|
||||
# No un-persisted tail means the transcript is fully durable: adopting the longer parent cannot drop input.
|
||||
_preflush_ok = True
|
||||
@@ -2523,6 +2664,9 @@ def _run_summary_dispatch(
|
||||
# A LATE successful summary must not undo the host's timeout cooldown: the
|
||||
# compressor checks cancellation before clearing; removed in finally (no leak).
|
||||
if commit_fence is not None:
|
||||
# Install a cancellation check the compressor consults BEFORE clearing the failure cooldown; removed
|
||||
# in the finally below so it cannot leak into later attempts (e.g. a manual /compress force-clear).
|
||||
# See #76354.
|
||||
_install_compression_cancelled_check(
|
||||
agent.context_compressor, lambda: commit_fence.is_cancelled, attempt_generation
|
||||
)
|
||||
@@ -2594,11 +2738,20 @@ def _fold_todo_snapshot(agent: Any, compressed: list) -> None:
|
||||
if todo_snapshot:
|
||||
# If this boundary pruned skill bodies, the policy behind the todos is gone:
|
||||
# add a reload notice after TODO_INJECTION_HEADER so both strip together.
|
||||
# Retention parity (#84718): the snapshot below re-injects the imperative verbatim. If this same
|
||||
# boundary pruned skill bodies to [SKILL_PRUNED: ...] markers, the policy that governed those tasks
|
||||
# is gone — couple a reload instruction to the snapshot so the imperative never crosses the boundary
|
||||
# alone.
|
||||
_reload_notice = _pruned_skill_reload_notice(compressed)
|
||||
if _reload_notice:
|
||||
todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}"
|
||||
# Fold the snapshot into a trailing REAL user msg (no synthetic user/user pair);
|
||||
# strip old snapshots first. Scaffolding tails must not absorb it (provenance).
|
||||
# Any snapshot merged at an earlier boundary is stripped first so repeated compactions refresh
|
||||
# rather than accumulate todo state (#26981). Scaffolding tails (continuation marker, summary
|
||||
# handoff, a bare stale snapshot row) must never absorb the snapshot: merging would upgrade them to
|
||||
# "real user" evidence and break zero-user provenance (#69292), so those keep the flagged standalone
|
||||
# append and the real-user preservation pass continues to see todo scaffolding, not human intent.
|
||||
from agent.context_compressor import _append_text_to_content
|
||||
merged = False
|
||||
_tail = compressed[-1] if compressed and isinstance(compressed[-1], dict) else None
|
||||
@@ -2630,6 +2783,12 @@ def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str:
|
||||
# Refresh tool schemas at the commit boundary: forever-sessions never restart,
|
||||
# so config reaches agent.tools here. Keep list identity if byte-equal (cache).
|
||||
try:
|
||||
# Refresh dynamic tool schemas at the same admitted-commit boundary that rebuilds the system prompt
|
||||
# (maintainer-directed, #95681 arc): forever-sessions (Bot Mode chats, gateway channels) never
|
||||
# restart, so compaction is the ONLY point where a config change — image model swap, delegation
|
||||
# depth, code_execution mode — can reach agent.tools. The prompt cache is already broken here, so
|
||||
# the refresh is free; when nothing changed the snapshot is byte-equal and we keep the existing list
|
||||
# object (identity matters to provider-side tool-block caching on some backends).
|
||||
_refresh_agent_tool_definitions(agent)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.warning(
|
||||
@@ -2638,6 +2797,13 @@ def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str:
|
||||
|
||||
# ALWAYS rebuild the prompt here: keeping old bytes meant prompt-builder changes
|
||||
# never reached long sessions. Equal bytes keep KV; preserve object identity.
|
||||
# ALWAYS rebuild the prompt at the admitted-commit boundary (maintainer-directed, #95681 arc). The
|
||||
# previous "keep-prompt" containment branch put the OLD bytes back whenever the reloaded memory blocks
|
||||
# were already embedded — which meant prompt-builder changes (guidance diets, new blocks, renames) NEVER
|
||||
# reached a long-lived session. The cache argument for keeping bytes was hollow: when nothing changed,
|
||||
# the rebuild is byte-identical and local KV prefixes survive on equality; when something changed, the
|
||||
# cache was stale by definition and propagation is the point. Preserve OBJECT identity on byte-equality
|
||||
# for backends that key on it.
|
||||
rebuilt_system_prompt = agent._build_system_prompt(system_message)
|
||||
if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt:
|
||||
new_system_prompt = agent._cached_system_prompt = cached_system_prompt
|
||||
@@ -2662,6 +2828,13 @@ def _salvage_or_refuse_grown_transcript(
|
||||
Compares like-for-like rough estimates; on growth tries one mechanical salvage pass, else treats the
|
||||
attempt as a refused no-op. Returns ``(compressed, None)`` to proceed or ``(None, prompt)`` when refused
|
||||
(caller releases the lease)."""
|
||||
# Anti-growth guard at the COMMIT SITE: never persist a compression that makes the transcript larger
|
||||
# (observed: 379K -> 687K when the generated summary plus retained reasoning exceeded what it replaced).
|
||||
# Compare like-for-like (both rough estimates of the same message shape) so an "actual vs estimate"
|
||||
# measurement mismatch cannot produce a false verdict. The gateway has a rotation-path-only guard
|
||||
# (#83339), but in-place compaction commits inside this method via archive_and_compact — before the
|
||||
# gateway can inspect the result — so the guard must live here to protect both paths. On growth, treat
|
||||
# the attempt as a no-op: the original transcript stays untouched and durable.
|
||||
_rough_in = estimate_messages_tokens_rough(messages)
|
||||
_rough_out = estimate_messages_tokens_rough(compressed)
|
||||
if _rough_out > _rough_in:
|
||||
@@ -2699,6 +2872,9 @@ def _salvage_or_refuse_grown_transcript(
|
||||
# Count the refusal as an ineffective-compaction strike so the anti-thrash
|
||||
# breaker latches; otherwise auto-compress retries the same summary every turn.
|
||||
with _swallow('could not record rejected-compaction strike', exc_info=True):
|
||||
# Without this, the unchanged transcript stays over the compression threshold and automatic
|
||||
# compression retries the identical summary request on every turn (#88568). Manual /compress
|
||||
# keeps bypassing the latch (force=True skips the guards).
|
||||
agent.context_compressor.record_rejected_compaction()
|
||||
_restore_prune_rearm_tokens(agent.context_compressor, attempt_snapshot)
|
||||
return None, _existing_sp
|
||||
@@ -2726,6 +2902,9 @@ def _carry_session_state_to_child(agent: Any, old_session_id: str, old_title: An
|
||||
transfer clears the ancestor's row, then restored so an inherited auto-title stays upgradeable.
|
||||
"""
|
||||
with _swallow('Could not migrate goal on compression: %s'):
|
||||
# Carry a persistent /goal onto the continuation session. Compression mints a fresh child id;
|
||||
# load_goal does a flat per-session lookup with no parent walk, so without this an active goal
|
||||
# silently dies at the boundary (#33618).
|
||||
from hermes_cli.goals import migrate_goal_to_session
|
||||
migrate_goal_to_session(old_session_id, agent.session_id, reason="compression")
|
||||
with _swallow('Could not migrate heartbeat on compression: %s'):
|
||||
@@ -2980,6 +3159,10 @@ def _candidate_rejected(
|
||||
|
||||
# Compare semantic state, not identity: engines may return an equal copy or
|
||||
# mutate the live list. ``==`` first (subclass __eq__), then marker-insensitive.
|
||||
# Neither case may rotate or rewrite the session. The raw ``==`` leg runs FIRST so a list subclass
|
||||
# returned by an engine keeps its ``__eq__`` semantics (tests seam on this); the marker-insensitive leg
|
||||
# (#92231) then covers the cold-resume shape where the stamped snapshot differs from the marker-swept
|
||||
# compress() output only by ``_db_persisted``.
|
||||
if compressed == messages_before_compression or (
|
||||
_strip_marker_for_comparison(compressed) == _strip_marker_for_comparison(messages_before_compression)
|
||||
):
|
||||
@@ -3095,6 +3278,19 @@ def _commit_compaction(
|
||||
agent._last_flushed_db_idx = 0
|
||||
else:
|
||||
# Bind old_session_id first: it is the rollback key in the handler below.
|
||||
# ── Rotation (legacy): end this session, fork a continuation ─ Flush any un-persisted
|
||||
# current-turn messages to the OLD session before ending it, so they survive in the
|
||||
# preserved parent transcript (#47202). (In-place skips this — see above.) Pass the
|
||||
# already-durable prefix as conversation_history so the flush skips it by identity (#68196).
|
||||
# Preflight compression runs BEFORE the normal turn flush has stamped the cold-resumed
|
||||
# history dicts with _DB_PERSISTED_MARKER, so without a boundary
|
||||
# _flush_messages_to_session_db treats every restored row as new and re-appends the whole
|
||||
# transcript to the parent. turn_context anchors _persist_user_message_idx at the
|
||||
# current-turn user message before preflight runs, so messages[:idx] is exactly the
|
||||
# persisted prefix; only the current turn's new messages get written. Bound to
|
||||
# old_session_id, hoisted above the flush: the ``except`` handler below keys its in-memory
|
||||
# rollback off this name, so anything that fails from here on rolls the transcript back
|
||||
# instead of leaving the failed attempt's compacted snapshot in place.
|
||||
old_session_id = agent.session_id
|
||||
_publish_rotated_compaction(
|
||||
agent, messages, compressed, new_system_prompt=new_system_prompt, lease=lease,
|
||||
@@ -3117,6 +3313,23 @@ def _commit_compaction(
|
||||
):
|
||||
if rotation_rollback:
|
||||
old_session_id = None
|
||||
# In-place sibling of the rotation rollback above (#99477). archive_and_compact() is atomic,
|
||||
# so a raise before it returned means EVERY pre-compaction row is still ``active = 1`` in
|
||||
# state.db — nothing was archived and the compacted set was never inserted. But
|
||||
# ``compressed`` is the marker-swept output of compress() (_strip_persistence_markers,
|
||||
# #57491) and the post-commit ``stamp_db_persisted_markers`` never ran, so handing it back
|
||||
# makes the next append-only flush treat the whole compacted transcript as new and INSERT it
|
||||
# ON TOP of the rows it was supposed to replace. The active set then holds the summary AND
|
||||
# the turns it summarized; the next resume reloads both, the token count goes UP, preflight
|
||||
# fires again, and each failed attempt appends another copy of the protected head + tail
|
||||
# (#99477: ~15 real turns stored as 3,814 rows, the first user message repeated 893 times).
|
||||
# Gate on ``split_status`` rather than ``compacted_in_place``: it is assigned on the
|
||||
# statement immediately after the atomic commit returns, so a committed compaction can never
|
||||
# be rolled back into a live/durable mismatch of the opposite sign. The deepcopy carries
|
||||
# each row's _DB_PERSISTED_MARKER from the pre-compression snapshot, so the restored
|
||||
# transcript is correctly skipped by the flush, and replacing every dict breaks
|
||||
# _db_flush_scan_prefix identity (same reasoning as the rotation branch — no explicit clear
|
||||
# needed).
|
||||
messages[:] = copy.deepcopy(messages_before_compression)
|
||||
compressed = messages
|
||||
made_progress = False
|
||||
@@ -3134,6 +3347,7 @@ def _commit_compaction(
|
||||
# Arm the failure cooldown so the next turn can't rerun the doomed compression;
|
||||
# try/except so a stub compressor can't mask the original error in this handler.
|
||||
with _swallow('could not record split-failure cooldown', exc_info=True):
|
||||
# See #97948.
|
||||
agent.context_compressor._record_compression_failure_cooldown(
|
||||
_SPLIT_FAILURE_COOLDOWN_SECONDS, f"session_split_failed: {e}"
|
||||
)
|
||||
@@ -3270,6 +3484,13 @@ def _begin_compression_attempt(agent: Any, *, force: bool, defer_notification: b
|
||||
agent._last_compression_attempt_recorded = True
|
||||
agent._last_compression_attempt_in_place = None
|
||||
agent._compression_skipped_due_to_lock = None
|
||||
# Clear the lock-skip signal at the VERY TOP, before the codex route and the breaker gates below can
|
||||
# early-return (per-attempt state rule, #58630/#69853). A stale ``True``/holder value from a prior
|
||||
# lock-skip must never make a later breaker/codex no-op look like lock contention to the automatic-path
|
||||
# consumers (compression_deferred, #49874) — the second clear before lock acquisition below stays for
|
||||
# the same reason it was added in #69870 and is simply idempotent now.
|
||||
# Transient-block signal (#97488): cleared with the same per-attempt rule; set by the breaker gates
|
||||
# below when a TRANSIENT guard (cooldown / structural backoff) no-ops this pass.
|
||||
agent._compression_blocked_transient = None
|
||||
started_at = time.monotonic()
|
||||
attempt_id = uuid.uuid4().hex
|
||||
@@ -3327,7 +3548,24 @@ def compress_context(
|
||||
"""Compress conversation context and split the session in SQLite.
|
||||
``force`` (manual /compress) clears the summary-failure cooldown; ``bypass_cooldown`` (provider-proven
|
||||
overflow) skips it once, breakers still apply. ``commit_fence`` stops a timed-out worker mutating session
|
||||
state. Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split."""
|
||||
state. Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split.
|
||||
|
||||
Args: agent: The owning :class:`AIAgent`. messages: Current message history (will be summarised).
|
||||
system_message: Current system prompt; used when compression needs a rebuilt cached prompt.
|
||||
approx_tokens: Pre-compression token estimate, logged for ops. task_id: Tool task scope (used for
|
||||
clearing file-read dedup state). focus_topic: Optional focus string for guided compression — the
|
||||
summariser will prioritise preserving information related to this topic. Inspired by Claude Code's
|
||||
``/compact <focus>``. force: If True, bypass any active summary-failure cooldown. Set by the manual
|
||||
``/compress`` slash command so users can retry immediately after an auto-compress abort. Auto-compress
|
||||
callers use the default ``False``. bypass_cooldown: If True, the automatic breaker gates ignore ONLY the
|
||||
summary-failure cooldown for this attempt (#100661). Set by the provider-proven overflow recovery path:
|
||||
the provider already rejected the request, so deferring until the cooldown lapses wedges the session.
|
||||
Unlike ``force`` it does not clear the cooldown, and the ineffective/structural breakers still apply; a
|
||||
failed attempt records its cooldown normally. defer_context_engine_notification: Delay the existing
|
||||
context-engine hook until a manual host commits its outer history transaction. commit_fence: Optional
|
||||
cooperative fence for executor callers that may time out. It prevents a late worker from mutating
|
||||
session state after its caller has moved on.
|
||||
"""
|
||||
attempt = _begin_compression_attempt(agent, force=force, defer_notification=defer_context_engine_notification)
|
||||
|
||||
# Codex owns the real thread; route compaction to its own compact (config
|
||||
@@ -3395,6 +3633,11 @@ def compress_context(
|
||||
|
||||
# Interrupts/redirects must not tear a summary in half. Use the explicit stop
|
||||
# Event (message fields race) + fence timeout so pool slots free promptly.
|
||||
# Explicit stop surfaces set a separate Event atomically; never infer cause from the racy message
|
||||
# fields. A host timeout also cancels the attempt's commit fence. Feed BOTH into the protected
|
||||
# auxiliary-call seam so the compression owner unwinds promptly while an isolated provider stream
|
||||
# finishes or closes in its daemon worker. Otherwise four timed-out streams retain all four shared
|
||||
# compression-pool slots until the auxiliary stream's longer absolute ceiling expires. See #23975.
|
||||
_hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None)
|
||||
phase = _run_summary_phase(
|
||||
agent, messages, lease=lease, in_place=in_place, checkpoint_required=checkpoint_required,
|
||||
|
||||
@@ -54,6 +54,14 @@ from hermes_logging import set_session_context
|
||||
# patch them here, so they must stay bound in this namespace.
|
||||
from agent.conversation_compression import conversation_history_after_compression # noqa: F401
|
||||
from agent.model_metadata import ( # noqa: F401
|
||||
# ----------------------------------------------------------------- Session hygiene: auto-compress
|
||||
# pathologically large transcripts Long-lived gateway sessions can accumulate enough history that every
|
||||
# new message rehydrates an oversized transcript, causing repeated truncation/context failures. Detect
|
||||
# this early and compress proactively — before the agent even starts. (#628) Token source priority: 1.
|
||||
# Actual API-reported prompt_tokens from the last turn (stored in session_entry.last_prompt_tokens) 2.
|
||||
# Rough char-based estimate (str(msg)//4). Overestimates by 30-50% on code/JSON-heavy sessions, but that
|
||||
# just means hygiene fires a bit early — safe and harmless.
|
||||
# -----------------------------------------------------------------
|
||||
estimate_messages_tokens_rough,
|
||||
estimate_request_tokens_rough,
|
||||
save_context_length,
|
||||
@@ -91,7 +99,13 @@ def _midturn_request_pressure_tokens(
|
||||
"""Token figure the mid-turn pre-API compression guard compares: the pruned
|
||||
native-Responses estimate when native compaction eligibility is proven (the generic
|
||||
estimate overstates the wire on compacted sessions, #96995), else messages+tools.
|
||||
The system prompt is counted exactly once."""
|
||||
The system prompt is counted exactly once.
|
||||
|
||||
When the upcoming request is eligible for native Responses compaction the transport will
|
||||
checkpoint-prune the payload before sending, so the generic durable-history estimate overstates the wire
|
||||
by orders of magnitude on a compacted session and fires a 600s local compression the main request never
|
||||
needed (#96995).
|
||||
"""
|
||||
try:
|
||||
from agent.codex_responses_adapter import estimate_native_responses_preflight_tokens
|
||||
native = estimate_native_responses_preflight_tokens(
|
||||
@@ -182,11 +196,18 @@ def _should_skip_model_call_for_reference_handoff(
|
||||
|
||||
# Fallback final_response for the sole-handoff skip (#80622); finalize_turn appends it as a
|
||||
# fresh assistant row, so it must not replay the last assistant text.
|
||||
# Deliberately NOT a replay of the last assistant text: finalize_turn's non-assistant-tail chokepoint
|
||||
# (#43849) appends final_response as a fresh assistant row, so recovering the previous turn's prose here
|
||||
# would duplicate it in the durable transcript AND re-deliver it to the user as if it were this turn's
|
||||
# answer. A short status is honest and idempotent.
|
||||
_HANDOFF_SKIP_FINAL_RESPONSE = (
|
||||
"Context was compacted. The previous response is complete — awaiting your next message."
|
||||
)
|
||||
|
||||
# Terminal final_response when compression timed out while the request was still oversized (#98722).
|
||||
# Terminal final_response for a turn ended because context compression hit its host progress-aware timeout
|
||||
# while the request was still oversized (#98722, salvaged from #98741). Sending the unchanged request would
|
||||
# only bounce off the provider's overflow error and re-enter compression in the same turn.
|
||||
_COMPRESSION_TIMEOUT_FINAL_RESPONSE = (
|
||||
"Context compression timed out without reducing this conversation. No messages were "
|
||||
"dropped. Start a fresh session with /new, or check auxiliary.compression before retrying /compress."
|
||||
@@ -228,6 +249,12 @@ def _is_interpreter_shutdown_error(exc: Exception) -> bool:
|
||||
"""True for a fatal interpreter-shutdown RuntimeError. The RuntimeError type gate
|
||||
stays here: a ValueError carrying similar text must not match (#93269)."""
|
||||
if isinstance(exc, RuntimeError):
|
||||
# ── Interpreter finalization: abandon immediately ── The process is exiting (TUI quit, SIGTERM,
|
||||
# one-shot done) while this turn — typically the post-turn review fork's daemon thread — is
|
||||
# mid-flight. Retries, credential rotation, and fallbacks are all futile ("cannot schedule new
|
||||
# futures..."), and the buffered ⚠️/❌ retry trace spams the shell after the TUI already exited. End
|
||||
# the turn with a single log line: no print, no traceback, no debug dump, no retry. Same class as
|
||||
# cron delivery (#55924/#58720) and concurrent tool submission — shared predicate.
|
||||
from tools.interpreter_shutdown import interpreter_shutting_down
|
||||
return interpreter_shutting_down(exc)
|
||||
return False
|
||||
@@ -296,6 +323,12 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text
|
||||
# Transcript shows the user's own words; the provider replays the scaffolded form.
|
||||
append_message(messages, {"role": "user", "content": text, "api_content": correction})
|
||||
|
||||
# Stateful scrubber for <memory-context> spans split across stream deltas (#5719). sanitize_context()
|
||||
# alone can't survive chunk boundaries because the block regex needs both tags in one string.
|
||||
# Stateful scrubber for reasoning/thinking tags in streamed deltas (#17924). Replaces the per-delta
|
||||
# _strip_think_blocks regex that destroyed downstream state (e.g. MiniMax-M2.7 streaming '<think>' as
|
||||
# delta1 and 'Let me check' as delta2 — the regex erased delta1, so downstream state machines never
|
||||
# learned a block was open and leaked delta2 as content).
|
||||
agent._current_streamed_assistant_text = ""
|
||||
agent._stream_needs_break = True
|
||||
|
||||
@@ -991,6 +1024,14 @@ def _provider_overflow_exhausted_result(
|
||||
"remains over threshold at ~%s tokens.",
|
||||
agent.log_prefix, max_compression_attempts, f"{request_pressure_tokens:,}",
|
||||
)
|
||||
# Host progress-aware timeout (#98722, salvaged from #98741): the provider proved the request does not
|
||||
# fit, but this recovery pass spent the full wait budget without a committed summary. Re-sending the
|
||||
# unchanged request would bounce off the same overflow error and re-enter compression in the same turn.
|
||||
# End the turn with the typed recovery contract instead — transcript intact, no further doomed provider
|
||||
# sends.
|
||||
# Prior <3 retries (or an earlier successful tool batch) leave a tool-result tail. Closing it here
|
||||
# matches interrupt aborts (#48879 / #52592) so the next user turn is not tool→user for strict
|
||||
# providers.
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return _partial_turn_result(
|
||||
"Context length exceeded: compression could not reduce the rebuilt request below the safe threshold.",
|
||||
|
||||
@@ -119,6 +119,7 @@ def _build_subprocess_env() -> dict[str, str]:
|
||||
|
||||
# Copilot ACP drives a model and needs LLM provider credentials; the central helper still
|
||||
# strips Tier-1 secrets (bot tokens, GitHub auth, infra).
|
||||
# See #29157.
|
||||
env = hermes_subprocess_env(inherit_credentials=True)
|
||||
env["HOME"] = _resolve_home_dir()
|
||||
apply_subprocess_home_env(env)
|
||||
@@ -303,6 +304,8 @@ class CopilotACPClient:
|
||||
try:
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags # hide the Windows console flash (#56747); pipes intact for the ACP wire
|
||||
|
||||
# Hide the console the CLI child would otherwise flash on Windows (#56747). Hide-only — stdio
|
||||
# pipes stay intact for the ACP wire.
|
||||
proc = subprocess.Popen(
|
||||
[self._acp_command] + self._acp_args, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||
text=True, encoding='utf-8', errors='replace', bufsize=1, cwd=self._acp_cwd, env=_build_subprocess_env(),
|
||||
|
||||
@@ -146,6 +146,14 @@ FAILURE_REASON_BILLING_UNVERIFIED = "billing_unverified"
|
||||
# every model call; on Windows several processes share one rotating log behind
|
||||
# a cross-process lock, and per-selection logging stormed that lock, pegged a
|
||||
# core, and stalled the event loop (Desktop backend readiness timeouts).
|
||||
# Credential selection runs on a hot path (every model call, plus auxiliary tasks like
|
||||
# compression/moa/titles), so when a pool is empty or fully exhausted the un-throttled log fires on *every*
|
||||
# selection. On Windows several Hermes processes share one rotating log guarded by concurrent-log-handler's
|
||||
# cross-process lock; that per-selection volume storms the lock (``RuntimeError: Cannot acquire lock after
|
||||
# 20 attempts``), pegs a core, and stalls the asyncio event loop long enough to fail the Desktop backend
|
||||
# readiness handshake ("Timed out connecting to Hermes backend after 15000ms"). Logging the condition at
|
||||
# most once per window preserves the signal while removing the storm — same class of fix as the warn-once
|
||||
# dedup in #58265.
|
||||
NO_AVAILABLE_ENTRIES_LOG_THROTTLE_SECONDS = 60.0
|
||||
|
||||
# Pool key prefix for custom OpenAI-compatible endpoints: all share
|
||||
@@ -731,6 +739,8 @@ def _write_through_provider_state_to_global_root(
|
||||
a failed write-through degrades to root-stale and must never break the
|
||||
profile's own successful save. Mirrors
|
||||
``hermes_cli.auth._write_through_xai_oauth_to_global_root``.
|
||||
|
||||
See #48415.
|
||||
"""
|
||||
try:
|
||||
global_path = _guarded_global_root(auth_mod._global_auth_file_path())
|
||||
@@ -2717,6 +2727,7 @@ def _seed_custom_pool(pool_key: str, entries: List[PooledCredential]) -> Tuple[b
|
||||
# The pool may be keyed under the durable ``providers.<key>``
|
||||
# slug or legacy ``custom:<name>``; accept any candidate, or
|
||||
# seeding is skipped when the pool holds the other identity.
|
||||
# Check if this model's base_url matches our custom provider. See #100413.
|
||||
matched_keys = {
|
||||
str(key).strip().lower() for key in custom_provider_pool_key_candidates(model_base_url)
|
||||
}
|
||||
|
||||
@@ -33,6 +33,7 @@ DEFAULT_KEEP = 5
|
||||
# is the backup dir itself; .git is repository metadata — rolling it back breaks git tracking, and snapshots that include it grow
|
||||
# with the full history (once backups are committed back, each snapshot contains the prior ones: 38MB of skills inflated to 24GB
|
||||
# in weeks). The tar filter in ``snapshot_skills`` applies the same set to nested paths, so a nested ``.git`` is skipped too.
|
||||
# See #91449.
|
||||
_EXCLUDE_TOP_LEVEL = {".curator_backups", ".hub", ".git"}
|
||||
|
||||
# Snapshot id: UTC ISO with colons replaced by dashes (Windows-safe filename); optional ``-NN`` suffix for same-second snapshots.
|
||||
|
||||
+12
-1
@@ -41,6 +41,8 @@ _LOOP_BLOCKED_DUMP_GRACE_S = 5.0
|
||||
|
||||
# ``Event.wait`` is a C-level block: KeyboardInterrupt / SetAsyncExc only land when the
|
||||
# thread returns to Python, so the sync wait is sliced to observe /stop or SIGINT promptly.
|
||||
# Slice the wait so a /stop or SIGINT during a bounded sync call is observed within this window rather than
|
||||
# at the full deadline (#94285, tools/test_local_interrupt_cleanup).
|
||||
_BOUNDED_SYNC_WAIT_SLICE_S = 0.2
|
||||
|
||||
|
||||
@@ -165,6 +167,12 @@ def resolve_timeout(key: str, *, default: Optional[float], env_var: Optional[str
|
||||
# second timer dumps all thread stacks when the loop provably failed to process the expiry.
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- Bounded execution — async
|
||||
# flavor. Generalizes plugins/platforms/telegram/adapter.py:_await_with_thread_deadline (the #63309 fix):
|
||||
# the deadline is driven by a daemon threading.Timer so a blocked event loop cannot disable it, and a second
|
||||
# timer dumps all thread stacks when the loop provably failed to process the expiry — the one piece of
|
||||
# information loop-blocked hangs otherwise never surface.
|
||||
# ---------------------------------------------------------------------------
|
||||
def _consume_abandoned(task: "asyncio.Future[Any]") -> None:
|
||||
"""Observe an abandoned task's outcome so it never logs 'never retrieved'."""
|
||||
try:
|
||||
@@ -282,7 +290,10 @@ def run_bounded_sync(
|
||||
"""Run ``fn`` in a daemon worker thread under a wall-clock deadline; exceptions re-raise in
|
||||
the caller. On expiry the worker is **abandoned** (every timeout leaks one daemon thread, so
|
||||
do NOT use per-item in hot loops) and ``on_timeout`` runs best-effort in the caller's thread.
|
||||
The worker runs under ``contextvars.copy_context()`` so secret scope / session id survive."""
|
||||
The worker runs under ``contextvars.copy_context()`` so secret scope / session id survive.
|
||||
|
||||
See #94285.
|
||||
"""
|
||||
timeout_s = clamp_timeout(timeout)
|
||||
start = time.monotonic()
|
||||
if timeout_s is None:
|
||||
|
||||
@@ -140,6 +140,9 @@ _USAGE_LIMIT_TRANSIENT_SIGNALS = (
|
||||
# Anthropic's "request_too_large" type without one).
|
||||
_PAYLOAD_TOO_LARGE_PATTERNS = (
|
||||
"request entity too large", "payload too large", "error code: 413", "request_too_large",
|
||||
# Normally arrives with an HTTP 413 status (handled by the status path), but aggregators/proxies can
|
||||
# re-wrap it into a plain message with no status attribute — route it to the same compression recovery.
|
||||
# (port of anomalyco/opencode#37848)
|
||||
"request exceeds the maximum size",
|
||||
)
|
||||
|
||||
@@ -181,6 +184,8 @@ _CONTEXT_OVERFLOW_PATTERNS = (
|
||||
"超过最大长度", "上下文长度",
|
||||
"tokens in request more than max tokens allowed",
|
||||
"input is too long", "max input token", "input token", "exceeds the maximum number of input tokens",
|
||||
# Together/Fireworks-style: "Input length 131393 exceeds the maximum allowed input length of 131040
|
||||
# tokens." No other pattern in this list matches that wording. (port of anomalyco/opencode#37848)
|
||||
"maximum allowed input length",
|
||||
)
|
||||
|
||||
@@ -548,6 +553,16 @@ def _by_transport(c: _Ctx) -> Optional[Verdict]:
|
||||
if any(p in msg for p in _SERVER_DISCONNECT_PATTERNS) and not c.status_code:
|
||||
# Reasoning models: far more likely the gateway idle-killed a long
|
||||
# thinking stream — never compress on a phantom overflow (#52310).
|
||||
# Reasoning-model override: a transport disconnect on a reasoning model is much more likely the
|
||||
# upstream proxy idle-killing a long thinking stream than a true context overflow — even on large
|
||||
# sessions. The default disconnect+large-session routing below would otherwise send the user into
|
||||
# the compression branch (should_compress=True) and silently delete conversation history on a
|
||||
# phantom context-length error. Reasoning models have multi-minute thinking phases that routinely
|
||||
# exceed the cloud gateway's idle window (NVIDIA NIM ~120s — first-party repro at
|
||||
# NVIDIA/NemoClaw#4846; OpenAI worker / Anthropic stream-idle similar). The per-reasoning-model
|
||||
# stale-timeout floor in agent/reasoning_timeouts.py raises the stale-detector threshold to tolerate
|
||||
# long thinking, so a true transport-layer failure here is recoverable via the retry path — not via
|
||||
# context compression. Reclassify as timeout. (Part 1 of Fixes #52310.)
|
||||
from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor
|
||||
if get_reasoning_stale_timeout_floor(c.model) is not None:
|
||||
return _V_TIMEOUT
|
||||
|
||||
@@ -84,7 +84,13 @@ def bare_gemini_model_id(model: str) -> str:
|
||||
|
||||
def gemini_requires_tool_call_ids(model: str) -> bool:
|
||||
"""Gemini 3+ needs explicit functionCall/functionResponse ids so replayed parallel tool calls
|
||||
pair with their responses; 2.x rejects the field."""
|
||||
pair with their responses; 2.x rejects the field.
|
||||
|
||||
Gemini 3+ models require explicit tool call IDs in replayed history — without them, multi-tool turns can
|
||||
be rejected or mismatched. Older Gemini models (2.x) reject unexpected ``id`` fields, so this is gated
|
||||
on the major version. Mirrors earendil-works/pi#7494 (their fix for the same class of bug in the
|
||||
google-shared converter).
|
||||
"""
|
||||
match = re.match(r"gemini-(\d+)", bare_gemini_model_id(model).lower())
|
||||
return match is not None and int(match.group(1)) >= 3
|
||||
|
||||
@@ -242,6 +248,10 @@ def _translate_tool_result_to_gemini(
|
||||
parsed = json.loads(content) if content.strip().startswith(("{", "[")) else None
|
||||
except json.JSONDecodeError:
|
||||
parsed = None
|
||||
# Gemini 3 resolves JSON-Schema ``$ref`` pointers inside a functionResponse.response payload and rejects
|
||||
# unknown references with HTTP 400 INVALID_ARGUMENT ("referenced name '#/$defs/...' does not match a
|
||||
# display_name"; see vercel/ai#14369). A tool result that is itself a JSON Schema (e.g. tool_describe
|
||||
# output for an MCP tool) must therefore be forwarded as opaque text, not as a structured response.
|
||||
structured = isinstance(parsed, dict) and not _looks_like_json_schema(parsed)
|
||||
function_response: Dict[str, Any] = {"name": name, "response": parsed if structured else {"output": content}}
|
||||
if include_ids and tool_call_id:
|
||||
@@ -264,6 +274,17 @@ def _merge_alternating(contents: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
functionResponse + functionResponse still merge); 3) the split pair stays API-valid via an interposed
|
||||
placeholder model turn."""
|
||||
merged: List[Dict[str, Any]] = []
|
||||
# Compatibility contract for native Gemini generateContent: 1) Same-role adjacent contents still merge
|
||||
# in general (strict user/model alternation for ordinary text turns and parallel tool-result grouping;
|
||||
# consecutive same-role contents are rejected with HTTP 400 "Please ensure that multiturn requests
|
||||
# alternate between user and model"). 2) Exception: do NOT fuse a human user text turn into a preceding
|
||||
# user content that only carries functionResponse parts (or vice versa). Gemini 3 accepts that fold with
|
||||
# HTTP 200 but then reads the trailing text as a continuation of the tool result — it returns an empty
|
||||
# candidate or "finishes the user's sentence" instead of answering (same defect gemini-cli fixed in
|
||||
# google-gemini/gemini-cli#28700). 3) Because rule 1's HTTP 400 makes two consecutive user contents
|
||||
# unsafe to emit (#55125 — the reason this merge exists), the split pair is kept API-valid by
|
||||
# interposing a placeholder model turn between the functionResponse content and the human text content,
|
||||
# mirroring gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER repair.
|
||||
for content in contents:
|
||||
prev = merged[-1] if merged else None
|
||||
same_role = prev is not None and prev["role"] == content["role"]
|
||||
@@ -677,6 +698,10 @@ class AsyncGeminiNativeClient:
|
||||
self.api_key, self.base_url = sync_client.api_key, sync_client.base_url
|
||||
self.chat = SimpleNamespace(completions=SimpleNamespace(create=self._create_chat_completion))
|
||||
|
||||
# Expose the underlying sync client as _real_client so the auxiliary cache's eviction-by-leaf-client
|
||||
# helper (#23482) can find and drop this async entry when the sync GeminiNativeClient is poisoned.
|
||||
# GeminiNativeClient is itself the leaf (no OpenAI client beneath it), so we point at the sync_client
|
||||
# directly.
|
||||
async def _create_chat_completion(self, **kwargs: Any) -> Any:
|
||||
result = await asyncio.to_thread(self._sync.chat.completions.create, **kwargs)
|
||||
return self._async_stream(result) if kwargs.get("stream") else result
|
||||
|
||||
+10
-1
@@ -212,7 +212,11 @@ def _resolve_inference_base_url(cfg: Optional[Dict[str, Any]], provider: str) ->
|
||||
def _resolve_inference_api_key(cfg: Optional[Dict[str, Any]], provider: str) -> str:
|
||||
"""Best-effort API key, resolved like :func:`_resolve_inference_base_url` so it
|
||||
matches the base URL actually probed; otherwise the local server-type probe hits
|
||||
a keyed remote endpoint without Authorization and sprays 401s on every image turn."""
|
||||
a keyed remote endpoint without Authorization and sprays 401s on every image turn.
|
||||
|
||||
Mirrors :func:`_resolve_inference_base_url`'s resolution order (runtime value, then ``model.api_key``,
|
||||
then the providers blocks) so the key matches the base URL actually being probed. See #89863.
|
||||
"""
|
||||
return _resolve_inference_value(cfg, provider, "api_key", runtime_ok=lambda _: True)
|
||||
|
||||
|
||||
@@ -270,6 +274,11 @@ def _probe_models_dev(provider: str, model: str, cfg: Optional[Dict[str, Any]])
|
||||
The fetch is cached (4h TTL) and backoff-limited."""
|
||||
from agent.models_dev import get_model_capabilities
|
||||
|
||||
# allow_network=True on purpose: vision-capability lookup runs when an image actually needs routing (not
|
||||
# per turn), and the #31179 text-only-main guard depends on catalog data — a cold cache returning
|
||||
# "unknown" would fall back to attempting the call and reintroduce the bug. This preserves the
|
||||
# historical network-on-cold-cache behavior for this one path; the fetch is cached (4h TTL) and
|
||||
# backoff-limited after failures.
|
||||
caps = get_model_capabilities(provider, model, allow_network=True)
|
||||
return None if caps is None else bool(caps.supports_vision)
|
||||
|
||||
|
||||
+7
-1
@@ -16,7 +16,11 @@ _SKILL_TOOLS = {"skill_view", "skill_manage"}
|
||||
|
||||
|
||||
def _fmt_est_cost(est_cost: float) -> str:
|
||||
"""Shared label helper so sub-cent totals render at 4dp, not "~$0.00"."""
|
||||
"""Shared label helper so sub-cent totals render at 4dp, not "~$0.00".
|
||||
|
||||
Routes through ``format_cost_label`` so sub-cent aggregates render at 4dp instead of collapsing to
|
||||
"~$0.00" (#79220 bug class — the same dishonesty this module's cost buckets exist to fix, #77223).
|
||||
"""
|
||||
return format_cost_label(Decimal(str(est_cost)))
|
||||
|
||||
|
||||
@@ -450,6 +454,8 @@ class InsightsEngine:
|
||||
@staticmethod
|
||||
def _cost_lines(o: Dict, templates: tuple) -> List[str]:
|
||||
"""One formatted line per non-zero cost bucket (estimated, included, unknown)."""
|
||||
# Cost breakdown — surface the three buckets so subscription-included and unknown-cost sessions are
|
||||
# visible instead of silently collapsing to $0. See #77223.
|
||||
est_cost = o.get("estimated_cost", 0.0)
|
||||
values = (_fmt_est_cost(est_cost) if est_cost > 0 else "", o.get("included_cost_sessions", 0), o.get("unknown_cost_sessions", 0))
|
||||
return [tpl.format(v) for tpl, v in zip(templates, values) if v]
|
||||
|
||||
@@ -111,6 +111,11 @@ def delete_node(node_id: str) -> dict[str, Any]:
|
||||
|
||||
def _delete_skill(name: str) -> dict[str, Any]:
|
||||
from tools import skill_usage
|
||||
# Pin must be respected by autonomous maintenance. The curator already skips pinned skills from every
|
||||
# auto-transition; the background review fork is the same kind of autonomous, no-user-present actor, so
|
||||
# it must not write to a pinned skill either (issue #25839). This is stricter than the foreground
|
||||
# ``_pinned_guard`` (which only blocks deletion) precisely because there is no user in the loop to
|
||||
# consent to an edit here.
|
||||
if skill_usage.get_record(name).get("pinned"):
|
||||
return {"ok": False, "message": f"'{name}' is pinned — unpin it first (hermes curator unpin {name})"}
|
||||
ok, message = skill_usage.archive_skill(name)
|
||||
|
||||
@@ -127,6 +127,7 @@ def inject_memory_provider_tools(agent: Any) -> int:
|
||||
if not memory_provider_tools_exposed(agent):
|
||||
# Say so once: a silent 0 leaves the provider looking "half on" with no clue which
|
||||
# config key (platform_toolsets / disabled_toolsets) gated it.
|
||||
# See #81014.
|
||||
_providers = [p for p in getattr(memory_manager, "providers", None) or []
|
||||
if getattr(p, "name", "") != "builtin"]
|
||||
if _providers:
|
||||
@@ -344,6 +345,9 @@ class MemoryManager:
|
||||
|
||||
# Core tool names are reserved: built-ins always win at agent init, so a shadowing
|
||||
# provider tool would linger in ``_tool_to_provider`` and hijack dispatch.
|
||||
# ``clarify``, ``delegate_task``). Reject it here, at the door, so it never enters the routing table
|
||||
# at all — matching the built-ins-always-win invariant used by the TTS/browser/search provider
|
||||
# registries. See #40466.
|
||||
from toolsets import _HERMES_CORE_TOOLS
|
||||
|
||||
for raw_schema in provider.get_tool_schemas():
|
||||
@@ -592,6 +596,16 @@ class MemoryManager:
|
||||
|
||||
``on_session_end`` (LLM-bound, seconds) must run strictly BEFORE ``on_session_switch`` rebinds
|
||||
provider state; an ad-hoc thread raced the inline switch and misattributed transcripts.
|
||||
|
||||
Running extraction inline blocked the /new command for the whole LLM round-trip (#16454); running it
|
||||
on an ad-hoc thread raced the inline switch — providers key off internal state, so a late
|
||||
``on_session_end`` ran against post-switch bindings (transcript misattributed to the new session id,
|
||||
double-ingest of the old turn buffer, new-session buffers cleared).
|
||||
Submitting BOTH hooks as one task on the manager's single background worker gives both properties at
|
||||
a single chokepoint: the caller returns immediately, and the worker's FIFO order serializes
|
||||
end→switch against every other provider write (per-turn ``sync_all``, prefetches), which already
|
||||
share the same worker. If the executor is unavailable, ``_submit_background`` degrades to inline
|
||||
execution — the pre-#16454 synchronous behavior, slow but correct.
|
||||
"""
|
||||
if not self._providers:
|
||||
return
|
||||
|
||||
@@ -253,6 +253,7 @@ _IMAGE_REJECTION_PHRASES = (
|
||||
"does not support images", "does not support image input", "does not support multimodal",
|
||||
"does not support vision", "model does not support image",
|
||||
# DashScope-style gateways reject non-text blocks with this generic body.
|
||||
# Some OpenAI-compatible endpoints (e.g. (issue #57948)
|
||||
"unexpected item type in content",
|
||||
# ChatGPT-account Codex backend rejects data:image URLs in input_image; keyed on the
|
||||
# field-path apostrophe so other URL errors don't false-trip. Second: its wording for
|
||||
@@ -262,8 +263,17 @@ _IMAGE_REJECTION_PHRASES = (
|
||||
"unknown variant `image_url`, expected `text`", "unknown variant image_url, expected text",
|
||||
# OpenRouter HTTP 404 when no upstream endpoint accepts image input (passes the 4xx
|
||||
# gate; without this the gateway queue wedges behind the stuck turn).
|
||||
# Without this phrase the agent never strips the images, the retry loop re-sends the same rejected
|
||||
# request until exhaustion, and the gateway leaves every subsequent message queued behind the stuck turn
|
||||
# — the P1 in issue #21160.
|
||||
"no endpoints found that support image input",
|
||||
# Kimi/Moonshot et al. reject truncated/corrupt image bytes baked into history.
|
||||
# Kimi / Moonshot / other OpenAI-compatible Chinese providers reject truncated or corrupt image bytes
|
||||
# with HTTP 400 "Invalid request: prepare image failed ... failed to decode image: invalid or
|
||||
# unsupported image format". Like the Codex case above, the bad bytes are baked into immutable
|
||||
# conversation history and re-sent on every retry, wedging the session. Strip the images so the turn
|
||||
# recovers instead of exhausting retries. (issue #76884; complements the proactive full-decode
|
||||
# validation in tools/vision_tools._normalize_to_supported_image)
|
||||
"failed to decode image",
|
||||
)
|
||||
|
||||
@@ -305,6 +315,17 @@ def _tc_set(tc: Any, key: str, value: Any) -> None:
|
||||
tc.__setitem__(key, value) if isinstance(tc, dict) else setattr(tc, key, value)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- call_id policy — single owner
|
||||
# (audit F4, incident chain I4) ---------------------------------------------------------------------------
|
||||
# Three forked policy sites converged here: * agent/codex_responses_adapter.py `_deterministic_call_id` —
|
||||
# hash synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2). *
|
||||
# run_agent.AIAgent._get_tool_call_id_static — `call_id or id` coalescing for dicts and SDK objects. *
|
||||
# run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with deterministic `_d<n>` suffixes
|
||||
# (#58327 loss class). NOT consolidated (different scheme on purpose):
|
||||
# agent/transports/codex_event_projector._deterministic_call_id maps codex app-server ITEM ids
|
||||
# (`codex_<type>_<item_id>`), not chat tool-call content; merging the two would change ids and invalidate
|
||||
# prompt caches. HARD INVARIANT: everything here must stay deterministic (never uuid4) and byte-identical
|
||||
# for existing inputs — these ids feed prompt-cache prefixes.
|
||||
def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
|
||||
"""Deterministic call_id fallback when the API omits one (random ids would break caching)."""
|
||||
seed = f"{fn_name}:{arguments}:{index}"
|
||||
@@ -392,6 +413,19 @@ def uniquify_tool_call_ids(tool_calls: list) -> list:
|
||||
# empty-string pads → " ". Strict side (400/422 "Extra inputs are not permitted"): everyone
|
||||
# else — Mistral, Cerebras, Groq, SambaNova, … Strip the key entirely, even a one-space pad.
|
||||
|
||||
# --------------------------------------------------------------------------- reasoning_content policy —
|
||||
# single owner (audit F4) --------------------------------------------------------------------------- The
|
||||
# strip-vs-repad decision was previously forked across the wire files in separate incident commits
|
||||
# (2b3a4f0af8 strip for strict providers, b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi
|
||||
# pad). The POLICY — which provider direction gets which treatment — lives here as one rule table + apply
|
||||
# functions; adapters keep only SYNTAX mapping (e.g. anthropic_adapter turning reasoning_content into a
|
||||
# thinking block). Direction table: require-side (echo-back enforced; replays 400 without the field): kimi
|
||||
# — provider kimi-coding/kimi-coding-cn, or host api.kimi.com / moonshot.ai / moonshot.cn. Host-driven on
|
||||
# purpose: aggregators re-exporting kimi models reject the echo. deepseek — provider "deepseek", model
|
||||
# contains "deepseek", or host api.deepseek.com (#15250; V4 rejects empty-string pads, hence the " "
|
||||
# single-space pad, #17341). mimo — provider "xiaomi", model contains "mimo", or host *.xiaomimimo.com.
|
||||
# strict side (field rejected with 400/422 "Extra inputs are not permitted"): everyone else — Mistral,
|
||||
# Cerebras, Groq, SambaNova, … (#45655). Strip the key entirely, even a single-space pad.
|
||||
_REASONING_ECHO_RULES: tuple = (
|
||||
# (family, exact providers (raw), exact providers (lowered), model substrings (lowered), hosts)
|
||||
("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (), ("api.kimi.com", "moonshot.ai", "moonshot.cn")),
|
||||
@@ -449,6 +483,15 @@ def apply_reasoning_content_policy(source_msg: dict, api_msg: dict, needs_thinki
|
||||
api_msg.pop("reasoning_content", None)
|
||||
return
|
||||
existing, reasoning = source_msg.get("reasoning_content"), source_msg.get("reasoning")
|
||||
# 1. Explicit reasoning_content already set. When the active provider enforces the thinking-mode
|
||||
# echo-back (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their own space-placeholder
|
||||
# written at creation time and any valid reasoning from the same provider. Sessions persisted BEFORE
|
||||
# #17341 have empty-string placeholders pinned at creation time; DeepSeek V4 Pro rejects those with
|
||||
# HTTP 400, so upgrade "" → " " on replay. When the active provider does NOT enforce echo-back, strip
|
||||
# the field entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq, SambaNova, …)
|
||||
# reject ANY reasoning_content key in input messages with HTTP 400/422 ("Extra inputs are not
|
||||
# permitted"), even an empty string or a single-space pad. Stripping here covers the rebuild path;
|
||||
# ``reapply_reasoning_echo`` covers the already-built api_messages path. Refs #45655.
|
||||
if isinstance(existing, str):
|
||||
# Explicit value: preserve verbatim, upgrading legacy "" to " " (DeepSeek V4 400s on "").
|
||||
api_msg["reasoning_content"] = existing or " "
|
||||
@@ -475,6 +518,16 @@ def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int:
|
||||
for api_msg in api_messages:
|
||||
if api_msg.get("role") != "assistant":
|
||||
continue
|
||||
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content' for providers that use the
|
||||
# internal 'reasoning' key. This must happen before the unconditional empty-string fallback so
|
||||
# genuine reasoning content is not overwritten (#15812 regression in PR #15478). Only promote for
|
||||
# providers that enforce echo-back — strict providers reject the field (refs #45655).
|
||||
# 4. DeepSeek / Kimi thinking mode: all assistant messages need reasoning_content. Inject a single
|
||||
# space to satisfy the provider's requirement when no explicit reasoning content is present.
|
||||
# Covers both tool-call turns (already-poisoned history with no reasoning at all) and plain text
|
||||
# turns. Space (not "") because DeepSeek V4 Pro tightened validation and rejects empty string with
|
||||
# HTTP 400 ("The reasoning content in the thinking mode must be passed back to the API"). Refs
|
||||
# #17341.
|
||||
if needs_thinking_pad:
|
||||
if not api_msg.get("reasoning_content"):
|
||||
apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad)
|
||||
|
||||
@@ -180,6 +180,11 @@ class MicroCompactionMixin:
|
||||
# in-place pop on a live dict would be identity-skipped by the bounded flush scan;
|
||||
# flag the finalizer.
|
||||
entry.pop(_cc()._DB_PERSISTED_MARKER, None)
|
||||
# Sibling of the finalize_turn pop site (#75170): this pop also strips the marker from a LIVE
|
||||
# dict in place, so the bounded flush-scan cursor would identity-skip the rewritten marker and
|
||||
# the defragged summary would never reach state.db. The compressor holds no agent reference, so
|
||||
# raise a flag the finalizer consumes to invalidate agent._db_flush_scan_prefix. (The pop sites
|
||||
# at module scope — fresh copies in strip-marker helpers — break identity and need no flag.)
|
||||
self._flush_scan_cursor_invalidated = True
|
||||
logger.info(
|
||||
"Micro-compaction defrag: rolling summary re-summarized (%d -> %d chars)",
|
||||
@@ -313,6 +318,9 @@ class MicroCompactionMixin:
|
||||
self._micro_compact_tokens_saved_total -= delta or 0
|
||||
self._micro_compact_passes += 1
|
||||
# Cached reads only: the lazy properties can fire a synchronous /models probe.
|
||||
# The ``threshold_tokens`` / ``context_length`` properties resolve lazily and can fire a
|
||||
# synchronous /models probe on first access (#32221) — telemetry must never be the thing that
|
||||
# blocks a turn. Unresolved simply reports null.
|
||||
threshold = self._threshold_tokens
|
||||
has_occupancy = threshold and tokens_after is not None and threshold > 0
|
||||
occupancy = round(tokens_after / threshold * 100, 1) if has_occupancy else None
|
||||
@@ -343,6 +351,7 @@ class MicroCompactionMixin:
|
||||
# Every row except the marker is a carried-forward original: archive rewind-style.
|
||||
session_db.archive_and_compact(session_id, compacted_messages, tail_count=max(0, len(compacted_messages) - 1))
|
||||
# Shared post-commit stamp site with batch commit and proactive prune.
|
||||
# See #98450.
|
||||
_cc().stamp_db_persisted_markers(compacted_messages)
|
||||
except Exception:
|
||||
logger.info(
|
||||
|
||||
@@ -30,6 +30,19 @@ logger = logging.getLogger(__name__)
|
||||
# Privacy filter (moa.privacy_filter: '' | display | full): PII classes agent.redact
|
||||
# leaves alone. The phone pattern requires explicit delimiters so line numbers,
|
||||
# dates, times, SHAs, IPs and versions never match.
|
||||
# Advisor (reference) outputs can echo PII from the conversation — emails, phone numbers, credentials pasted
|
||||
# by the user — into surfaces the user may not expect: the labelled reference blocks rendered in the UI,
|
||||
# saved MoA trace files, and (in `full` mode) the guidance block injected into the aggregator prompt (issue
|
||||
# #59959). Secret/credential shapes (API-key prefixes, JWTs, private keys, DB connection strings, E.164
|
||||
# phone numbers) are handled by the repo's central redactor, ``agent.redact .redact_sensitive_text`` — the
|
||||
# MoA filter never re-implements those. The two patterns below cover the PII classes the central redactor
|
||||
# deliberately leaves alone for log/tool output (emails and formatted phone numbers). Pattern safety:
|
||||
# advisory text is frequently code-review-shaped — line numbers, timestamps, git SHAs, IDs, IP addresses. A
|
||||
# bare 10-digit match would mangle all of those, so the phone pattern requires clearly delimited formatting:
|
||||
# a parenthesized area code and/or explicit `-`/`.` separators between groups ((555) 123-4567, 555-123-4567,
|
||||
# 555.123.4567, +1 555-123-4567). Undelimited digit runs (5551234567), dates (2026-07-12), times (12:34:56),
|
||||
# hex IDs, and dotted quads never match. International numbers in E.164 form (+14155551234) are already
|
||||
# masked by the central redactor.
|
||||
_MOA_EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")
|
||||
_MOA_PHONE_RE = re.compile(
|
||||
r"(?<![\w.+-])" # no leading word char / dot / + / - (kills IPs, IDs, versions)
|
||||
@@ -99,6 +112,9 @@ def _redact_trace_accounting(acct: Any) -> Any:
|
||||
|
||||
|
||||
# Cold-start caches: preset and per-(provider, model) runtime are immutable for a turn.
|
||||
# A MoA preset switch used to re-resolve the full config + preset + every slot's provider runtime on EACH
|
||||
# create() call (once per tool-loop iteration), serially before the parallel fan-out could start — adding
|
||||
# 5-30s of "frozen" latency on complex presets (#66793).
|
||||
_preset_cache_lock = threading.Lock()
|
||||
_preset_cache: dict[tuple, Any] = {}
|
||||
|
||||
@@ -286,6 +302,9 @@ def _maybe_apply_moa_cache_control(
|
||||
legacy system-and-3 fallback is used. ``cache_disabled`` is stamped onto the
|
||||
stub so ``cache_ttl: off`` is honored; ``cache_ttl`` is clamped per destination.
|
||||
Returns the messages unchanged on any error.
|
||||
|
||||
``cache_disabled`` (or the live config when omitted) is stamped onto the policy stub so
|
||||
``prompt_caching.cache_ttl: off`` is not bypassed by the blank-agent pattern (#76085).
|
||||
"""
|
||||
try:
|
||||
from agent.agent_runtime_helpers import anthropic_prompt_cache_policy, blank_cache_policy_stub
|
||||
@@ -353,6 +372,10 @@ def _run_reference(
|
||||
# Trim to THIS model's window (advisors may be smaller than the aggregator); the
|
||||
# advisory view is append-only across iterations, so cache_control lets
|
||||
# iteration N+1 replay N's cached prefix.
|
||||
# Reference models may have a smaller window than the aggregator (e.g. kimi-k2.7-code @ 262K
|
||||
# advising a glm-5.2 @ 1M conversation); without this trim the provider returns a hard HTTP 400
|
||||
# which the except below silently converts to a [failed: …] note (issue #60345). Estimated AFTER the
|
||||
# advisory system prompt is prepended so its tokens count against the budget too.
|
||||
trimmed = _trim_messages_for_reference(
|
||||
messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache,
|
||||
)
|
||||
@@ -421,6 +444,11 @@ def _trim_messages_for_reference(
|
||||
body and the trailing user turn plus one preceding turn (even if still over
|
||||
budget). ``context_length_cache`` memoizes the window per (provider, model);
|
||||
unresolvable windows leave messages unchanged.
|
||||
|
||||
Reference models may have a smaller context window than the aggregator or the main conversation. Without
|
||||
this trim, a reference whose window is exceeded gets a hard HTTP 400 from the provider, which
|
||||
``_run_reference``'s try/except silently converts to a ``[failed: …]`` note — the MoA turn silently
|
||||
degrades to fewer references (issue #60345).
|
||||
"""
|
||||
if not messages or not slot.get("model"):
|
||||
return messages
|
||||
@@ -480,6 +508,13 @@ def _settle_interrupted(
|
||||
for future, idx in futures.items():
|
||||
if results[idx] is not None:
|
||||
continue
|
||||
# #38922: a slow confirmation does NOT necessarily mean the send failed — but we must distinguish
|
||||
# two cases via future.cancel()'s return value: cancel() == False -> the coroutine was already
|
||||
# running on the gateway loop when the timeout fired; the request is in flight on the wire and
|
||||
# cannot be un-sent. Re-sending via standalone would be a guaranteed DUPLICATE, so treat it as
|
||||
# delivered (assume-delivered). cancel() == True -> the scheduled callback never started executing
|
||||
# (loop wedged/backlogged for the full 60s), so nothing was sent. We MUST fall through to the
|
||||
# standalone path or the message is silently dropped (worse than a duplicate).
|
||||
cancelled = future.cancel()
|
||||
if not cancelled and future.done():
|
||||
results[idx] = future.result()
|
||||
@@ -753,6 +788,12 @@ def aggregate_moa_context(
|
||||
Failures become model-specific notes instead of aborting the loop.
|
||||
``reference_max_tokens`` caps ONLY the fan-out (capping the aggregator truncated
|
||||
long syntheses). ``agent`` makes the fan-out interruptible.
|
||||
|
||||
``reference_max_tokens`` applies ONLY to the reference fan-out — the aggregator's own synthesis call is
|
||||
never capped, so it always uses its model's own maximum. ``call_llm`` omits the parameter entirely when
|
||||
it is ``None`` (see its docstring), which also sidesteps providers that reject ``max_tokens`` outright.
|
||||
A hardcoded cap on the aggregator call previously truncated long aggregator syntheses (#53580) — passing
|
||||
``reference_max_tokens`` to both calls here would silently reintroduce that regression.
|
||||
"""
|
||||
reference_models = [slot for slot in reference_models if slot.get("enabled", True)]
|
||||
reference_outputs = _run_references_parallel(
|
||||
@@ -1015,6 +1056,9 @@ class MoAChatCompletions:
|
||||
planning_messages = peel_reference_guidance(agg_messages, str(guidance)) if guidance else agg_messages
|
||||
# Tri-state cache_disabled: facades built via __new__ have no _agent; forcing
|
||||
# False would suppress the planner's config fallback.
|
||||
# plan_cache_sections_for_destination never mutates its inputs and always returns request-local
|
||||
# copies, so the prepared state stays canonical. Tri-state: only pass a bool when a live agent
|
||||
# snapshot exists. See #76085.
|
||||
_agent = getattr(self, "_agent", None)
|
||||
cache_disabled, cache_ttl = _agent_cache_opts(_agent)
|
||||
# Agent TTL + stable system prefix so MoA does not regress 1h → 5m.
|
||||
@@ -1090,6 +1134,19 @@ class MoAChatCompletions:
|
||||
view changes. "every_n:<N>": iteration 1 of a turn, then every Nth; in-between
|
||||
iterations return the pinned last on-cadence key (HIT: no calls, no re-emit).
|
||||
"""
|
||||
# "user_turn" (default — cheapest cadence, #67199): advisors run ONCE per user turn; subsequent tool
|
||||
# iterations reuse that turn's advice and the aggregator acts alone (the original MoA shape:
|
||||
# synthesize at the start, then let the acting model work). Implemented by hashing only the prefix
|
||||
# up to the LAST USER message so mid-turn growth doesn't change the signature — iteration 2+ becomes
|
||||
# a cache HIT. "per_iteration": advisors re-run whenever the advisory view changes — i.e. every tool
|
||||
# iteration, since the view grows with each tool result; advice tracks live task state at the cost
|
||||
# of multiplying advisor latency/spend by tool-loop depth. "every_n:<N>" (N >= 2): the middle ground
|
||||
# (issue #63393 — advisor fan-out multiplies latency/cost by the tool-iteration count). Advisors run
|
||||
# on iteration 1 of a user turn and then every Nth tool iteration; the iterations in between REUSE
|
||||
# the cached guidance from the last on-cadence run (same mechanism as user_turn's cache HIT — the
|
||||
# aggregator still gets advice every iteration, it's just not refreshed against the very latest tool
|
||||
# results). The iteration counter is scoped per user turn and resets on a new user message, so every
|
||||
# turn starts with fresh advice.
|
||||
fanout_mode = str(preset.get("fanout") or "user_turn").strip().lower()
|
||||
every_n = 0
|
||||
if fanout_mode.startswith("every_n:"):
|
||||
|
||||
+39
-1
@@ -112,6 +112,11 @@ _ENDPOINT_MODEL_CACHE_TTL = 300
|
||||
# server swap on the same port is re-detected; None gets the short TTL so a
|
||||
# transient failure recovers in minutes without re-running the waterfall each turn.
|
||||
_ENDPOINT_PROBE_TTL_SECONDS = 3600.0
|
||||
# A failed probe verdict (server_type is None — no known endpoint answered) is cached for a much shorter
|
||||
# window: the in-memory entry exists only to keep one image-bearing turn from re-running the 5-request
|
||||
# waterfall on every subsequent turn (#89863 — a keyed remote endpoint answered 401 to each leg and the None
|
||||
# verdict was never cached, so every turn re-probed). Short TTL keeps a transient failure (server starting
|
||||
# up, key being fixed) recoverable within minutes instead of pinning "undetected" for an hour.
|
||||
_ENDPOINT_PROBE_FAILURE_TTL_SECONDS = 300.0
|
||||
_endpoint_probe_path_cache: Dict[str, tuple] = {}
|
||||
# Routable-but-dead endpoints (corp LAN off-VPN) blackhole TCP: once ANY probe paid
|
||||
@@ -674,6 +679,9 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
return result
|
||||
|
||||
|
||||
# Cache the negative verdict in memory only (never on disk — a failure is often transient: server starting,
|
||||
# key being fixed) so the very next turn does not re-run the whole waterfall against an endpoint that just
|
||||
# answered nothing (#89863).
|
||||
def _iter_nested_dicts(value: Any):
|
||||
if isinstance(value, dict):
|
||||
yield value
|
||||
@@ -778,6 +786,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any
|
||||
try:
|
||||
_ensure_requests()
|
||||
# (connect, read) tuple: a flat timeout lets urllib3 block per retry stage through proxies that 403 CONNECT.
|
||||
# See #46620.
|
||||
response = requests.get(OPENROUTER_MODELS_URL, timeout=(5, 10), verify=_resolve_requests_verify())
|
||||
response.raise_for_status()
|
||||
cache = {}
|
||||
@@ -1186,6 +1195,8 @@ def _ollama_show(server_url: str, api_key: str, bare_model: str, timeout: float
|
||||
|
||||
def _is_ollama_server(base_url: str, api_key: str) -> bool:
|
||||
try:
|
||||
# Forward the API key: a remote API-keyed endpoint answers the probe waterfall with 401s without it,
|
||||
# and an unauthorized probe can never produce a positive verdict (#89863).
|
||||
return detect_local_server_type(base_url, api_key=api_key) == "ollama"
|
||||
except Exception:
|
||||
return False
|
||||
@@ -1447,7 +1458,11 @@ _CODEX_900K_SNAPSHOT_RE = re.compile(r"^\d{4}-\d{2}-\d{2}$")
|
||||
|
||||
|
||||
def _bare_codex_slug(model: Optional[str]) -> str:
|
||||
"""Lowercased slug without ``vendor/`` (display/auxiliary callers pass ``openai/gpt-5.6-sol-900k``)."""
|
||||
"""Lowercased slug without ``vendor/`` (display/auxiliary callers pass ``openai/gpt-5.6-sol-900k``).
|
||||
|
||||
Display/auxiliary callers pass ids like ``openai/gpt-5.6-sol-900k``; the main-agent path normalizes the
|
||||
namespace away earlier, but this resolver must accept both shapes (#92797 review).
|
||||
"""
|
||||
return (model or "").strip().lower().rsplit("/", 1)[-1]
|
||||
|
||||
|
||||
@@ -1567,6 +1582,10 @@ def _resolve_codex_oauth_context_length_with_source(model: str, access_token: st
|
||||
return bumped, source
|
||||
return ctx, source
|
||||
# The Codex catalog only knows the base slug (no -900k, no vendor/).
|
||||
# ``-900k`` variants are Hermes picker aliases — the Codex catalog only knows the base slug, so resolve
|
||||
# against the stripped id. Also drop any ``vendor/`` namespace (``openai/gpt-5.6-sol-900k``): the
|
||||
# main-agent path normalizes it away before reaching here, but display/auxiliary callers pass it through
|
||||
# (#92797 review).
|
||||
lookup_bare = _bare_codex_slug(strip_codex_context_variant_suffix(model_bare))
|
||||
if access_token:
|
||||
live, fresh_probe = _fetch_codex_oauth_context_lengths_with_source(access_token)
|
||||
@@ -1650,6 +1669,10 @@ def _validate_cached_context_length(model: str, base_url: str, cached: int, is_b
|
||||
_invalidate_cached_context_length(model, base_url)
|
||||
return bedrock_ctx
|
||||
return cached
|
||||
# For local endpoints, run the probe that respects configured Modelfile context values first.
|
||||
# _query_local_context_length prefers num_ctx from Modelfile, while _query_ollama_api_show returns the
|
||||
# GGUF training max first which can be larger and would create a false-safe window for compression
|
||||
# (#63122). Non-local endpoints preserve the existing GGUF-first behavior.
|
||||
if is_local_endpoint(base_url):
|
||||
return _reconcile_local_cached_context_length(model, base_url, cached, api_key=api_key)
|
||||
return cached
|
||||
@@ -1740,12 +1763,17 @@ def _config_override_context_length(model: str, base_url: str, provider: str, cu
|
||||
"""Steps 0b-0c: config-only overrides (never touch the network). 0b: EXPLICIT model_overrides
|
||||
only — fill-gap _default entries apply inside lookup_models_dev_context once the catalog has
|
||||
missed, so a _default can never preempt custom_providers or live probes. 0c: custom_providers."""
|
||||
# This is the supported self-unblock path for models with wrong context in models.dev (#84482) and for
|
||||
# custom/local models (#8731).
|
||||
if provider and model:
|
||||
with contextlib.suppress(Exception): # fall through to other resolution paths
|
||||
from agent.models_dev import _override_context_window
|
||||
mo_ctx = _override_context_window(provider, model)
|
||||
if mo_ctx is not None and mo_ctx > 0:
|
||||
return mo_ctx
|
||||
# 0c. custom_providers per-model override — check before any probe. This closes the gap where /model
|
||||
# switch and display paths used to fall back to 128K despite the user having a per-model context_length
|
||||
# set. See #15779.
|
||||
if custom_providers and base_url and model:
|
||||
with contextlib.suppress(Exception): # fall through to probing
|
||||
from hermes_cli.config import get_custom_provider_context_length
|
||||
@@ -1981,6 +2009,16 @@ def _strip_stale_thinking_for_estimate(messages: List[Dict[str, Any]]) -> List[D
|
||||
# pinned (strong ref in the entry, so the id can't be reused and immutability makes id-equality
|
||||
# value-equality); numbers/bools/None by value; dicts/lists structurally in key order (``str(shadow)``
|
||||
# depends on it); any other type aborts the memo. api_messages shallow-copies dicts but shares the strings.
|
||||
# ``estimate_messages_tokens_rough`` is called on the full history every loop iteration (conversation_loop
|
||||
# preflight), repeatedly during compaction telemetry, and inside an O(n^2) shrink loop in moa_loop. The
|
||||
# per-message helpers are pure functions of the message's value, so a memo keyed on a fingerprint that
|
||||
# uniquely determines the value is exactly equivalent. Fingerprint design (soundness argument): While the
|
||||
# entry lives, that id cannot be reused by another object, so id-equality implies object-equality — strings
|
||||
# are immutable, so value-equality too (no #50372-style aliasing). Equal fingerprints therefore imply
|
||||
# deep-equal messages built from identical immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical
|
||||
# estimate. Because the api_messages build shallow-copies history dicts each iteration, the copies share the
|
||||
# same content strings — so unchanged history messages hit the memo even though the outer dicts are fresh
|
||||
# objects every turn.
|
||||
_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {}
|
||||
_MSG_TOKENS_CACHE_MAX = 4096
|
||||
|
||||
|
||||
+20
-3
@@ -498,7 +498,10 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
|
||||
"""Context window in tokens for provider+model, or None if not found. An EXPLICIT ``model_overrides``
|
||||
entry wins over the catalog; ``_default`` fills the gap only when the catalog has no answer (the
|
||||
self-unblock path for wrong/missing context in models.dev). Catalog entries with context=0 are
|
||||
skipped in favour of later candidates. ``allow_network`` defaults to False — runs every turn."""
|
||||
skipped in favour of later candidates. ``allow_network`` defaults to False — runs every turn.
|
||||
|
||||
See #84482.
|
||||
"""
|
||||
override_ctx = _override_context_window(provider, model)
|
||||
if override_ctx is not None:
|
||||
return override_ctx
|
||||
@@ -513,6 +516,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
|
||||
# catalog. ``<provider>._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to
|
||||
# models the catalog does not know and never displace catalog data. Provider keys accept the Hermes
|
||||
# or models.dev id; model ids match exactly, then case-insensitively (mirroring catalog lookup).
|
||||
# Resolution semantics: 1. 2. See #84482, #8731.
|
||||
_OVERRIDE_WARNED_KEYS: set = set()
|
||||
# Safe defaults for models absent from the catalog (tools on, vision/reasoning off, 200K context);
|
||||
# shared by get_model_capabilities and get_model_info so the two unknown-model paths agree.
|
||||
@@ -589,6 +593,7 @@ def _override_context_window(provider: str, model: str) -> Optional[int]:
|
||||
return _override_int(ov, "context_window") if ov is not None else None
|
||||
|
||||
|
||||
# Catalog miss — a _default override may fill the gap (#84482).
|
||||
def _default_override_context(provider: str) -> Optional[int]:
|
||||
"""Fill-gap context from a ``_default`` override, for catalog misses."""
|
||||
default = _default_model_override(provider)
|
||||
@@ -655,7 +660,13 @@ def _entry_supports_vision(entry: Dict[str, Any]) -> bool:
|
||||
def get_model_capabilities(provider: str, model: str, *, allow_network: bool = False) -> Optional[ModelCapabilities]:
|
||||
"""Capability metadata from the models.dev cache, or None if unresolvable. EXPLICIT ``model_overrides``
|
||||
patch catalog fields; ``_default`` fills the gap only for models the catalog does not know. Unspecified
|
||||
fields fall through to the catalog, or to safe defaults. ``allow_network`` defaults to False (hot path)."""
|
||||
fields fall through to the catalog, or to safe defaults. ``allow_network`` defaults to False (hot path).
|
||||
|
||||
EXPLICIT ``model_overrides`` entries (per-provider+model) win over catalog values for the fields they
|
||||
set. ``_default`` entries fill the gap only for models the catalog does not know — the supported
|
||||
self-unblock path for custom/local models (#8731) and for models with wrong metadata in models.dev
|
||||
(#84482).
|
||||
"""
|
||||
models = _get_provider_models(provider, allow_network=allow_network)
|
||||
entry = _find_model_entry(models, model) if models is not None else None
|
||||
raw = _apply_overrides(provider, model, entry)
|
||||
@@ -751,7 +762,13 @@ def get_provider_info(provider_id: str, *, allow_network: bool = True) -> Option
|
||||
def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = False) -> Optional[ModelInfo]:
|
||||
"""Full model metadata by Hermes or models.dev provider ID (exact match, then case-insensitive), or
|
||||
None if not found. EXPLICIT ``model_overrides`` patch known catalog models; ``_default`` fills the gap
|
||||
only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths."""
|
||||
only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths.
|
||||
|
||||
``model_overrides`` entries use the SAME canonical schema as every other consumer (``context_window``,
|
||||
``max_output_tokens``, ``supports_*``, ``model_family``) — they are translated into the catalog shape at
|
||||
this boundary, and sub-dicts (``limit``, ``modalities``) are merged rather than clobbered. See #84482,
|
||||
#8731.
|
||||
"""
|
||||
mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id)
|
||||
models = _registry_models(mdev_id, allow_network=allow_network)
|
||||
mid, entry = next(_iter_model_entries(models, model_id, suffix_fallback=False), (model_id, None)) if models is not None else (model_id, None)
|
||||
|
||||
@@ -206,6 +206,18 @@ def prune_pre_checkpoint_items(
|
||||
that is itself a canonical summary carrier is read from the SOURCE and retained as a
|
||||
synthesized ``role="assistant"`` message.
|
||||
- ``enable_summary_retention`` is a test override, not a config surface.
|
||||
|
||||
The server drops every input item that precedes a replayed ``compaction`` item (live-verified Aug 2026),
|
||||
so sending pre-checkpoint history is dead weight AND silently erases the user's plaintext asks —
|
||||
including any local-compression summary the agent already produced, which previously vanished here
|
||||
because it carries ``role="assistant"``, not ``"user"`` (#90975).
|
||||
A summary is never byte/character-sliced: Hermes summaries carry structural framing (handoff prefix, end
|
||||
marker, merge-into-tail delimiters) that a blind slice can corrupt, so one that doesn't fit whole is
|
||||
dropped instead. A summary already retained once (identical text) is never duplicated, so repeated
|
||||
checkpoints stay idempotent. - ``enable_summary_retention`` is a function-level override (used by tests
|
||||
and callers that need the pre-#90975 behavior back); it is not wired to a user-facing config surface.
|
||||
Without ``item_sources`` (default), retention only sees what survived conversion, matching pre-#90976
|
||||
behavior (#90976).
|
||||
"""
|
||||
if not isinstance(items, list) or not items:
|
||||
return items
|
||||
@@ -241,6 +253,9 @@ def prune_pre_checkpoint_items(
|
||||
continue
|
||||
# Source-based detection sees past a lossy conversion; it only fires
|
||||
# when the source itself is a provenance-tagged summary carrier.
|
||||
# Canonical source-based summary detection: reads the ORIGINAL chat message's own content, so it
|
||||
# sees past a lossy conversion (a typed `function_call_output` wrapper, or a stale exact-replay
|
||||
# message) that erased the summary from `item` itself (#90976).
|
||||
if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source):
|
||||
text = flatten_message_text(source.get("content"))
|
||||
_src_role = source.get("role")
|
||||
@@ -291,6 +306,13 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool:
|
||||
Drives one-shot recovery (strip, disable for the session, retry), so matching is narrow:
|
||||
a transient 5xx that merely ECHOES the request must not downgrade native compaction.
|
||||
Requires ``status_code`` 400 (or unknown) AND the field name with rejection language.
|
||||
|
||||
See #82777.
|
||||
* ``status_code`` is 400 (or unknown/None — some transports surface only a message string; field-name
|
||||
matching alone is then the best available signal, preserving pre-#82777 behavior for them), and * the
|
||||
error text names ``context_management`` / ``compact_threshold`` alongside rejection language ("unknown",
|
||||
"unsupported", "invalid", "unexpected", "not permitted"...). A bare field-name echo without rejection
|
||||
language does not match.
|
||||
"""
|
||||
text = str(error or "").lower()
|
||||
if "context_management" not in text and "compact_threshold" not in text:
|
||||
|
||||
@@ -56,6 +56,18 @@ def opencode_session_headers(
|
||||
from agent.transports.codex import _cache_scope_from_session_id
|
||||
|
||||
key = _cache_scope_from_session_id(
|
||||
# Top-level session_id → OpenRouter's sticky routing key. Per their prompt-caching docs it is
|
||||
# used directly as the routing key instead of hashing the opening messages, and it activates
|
||||
# stickiness on the first successful request rather than only after a cache hit. Resolve it from
|
||||
# the declared routing scope first (set only by a host that names its own conversation, #96811),
|
||||
# then the ambient conversation contextvar, with the explicit argument as fallback. The gap this
|
||||
# closes is the auxiliary call sites — compression, title generation, vision, web_extract,
|
||||
# session_search, MoA slots — which funnel through ``agent.auxiliary_client``. That module has
|
||||
# no session handle and passes no ``session_id``, so those calls sent NO sticky key at all and
|
||||
# each routed independently of the conversation it belonged to (#70820). Mirrors the Nous Portal
|
||||
# profile, which resolves the same way (f2f4df064d). The ambient value is the session-lineage
|
||||
# ROOT, so it also stays stable for installs that opt out of the default ``compression.in_place:
|
||||
# true`` and across delegate-subagent trees.
|
||||
get_affinity_scope() or get_conversation_context() or session_id
|
||||
)
|
||||
except Exception:
|
||||
|
||||
@@ -133,7 +133,12 @@ def flush(timeout: float = 5.0) -> bool:
|
||||
def re_register_config_hooks() -> None:
|
||||
"""Re-register outbound webhooks after a plugin force-reload cleared ``_hooks``. Only the
|
||||
current home's idempotence keys are cleared so a force-reload in one profile cannot
|
||||
invalidate another profile's still-live registration."""
|
||||
invalidate another profile's still-live registration.
|
||||
|
||||
Mirrors ``agent.shell_hooks.re_register_config_hooks``: config-owned outbound-webhook callbacks live in
|
||||
the same ``_hooks`` dict that ``PluginManager.discover_and_load(force=True)`` clears via ``unload()``,
|
||||
so without this the force-reloaded profile's outbound webhooks go silently inert (#92682 review).
|
||||
"""
|
||||
from hermes_cli.config import load_config
|
||||
_forget_home_registrations(_registered, _registered_lock)
|
||||
register_from_config(load_config())
|
||||
@@ -235,6 +240,7 @@ def _serialize_payload(event: str, kwargs: Dict[str, Any], delivery_id: str) ->
|
||||
(also the ``X-Hermes-Delivery`` header) and ``timestamp`` live inside the HMAC-signed
|
||||
body, so they double as replay protection."""
|
||||
# Profile resolved at fire time so a multiplexed gateway's receivers can tell which profile emitted.
|
||||
# See #92674.
|
||||
from hermes_cli.profiles import get_active_profile_name
|
||||
payload = {
|
||||
"hook_event_name": event, "profile": get_active_profile_name(), **_payload_fields(kwargs),
|
||||
|
||||
@@ -58,7 +58,10 @@ Interaction style:
|
||||
|
||||
|
||||
def build_plan_prompt(task: str = "") -> str:
|
||||
"""Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context."""
|
||||
"""Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context.
|
||||
|
||||
See #36821.
|
||||
"""
|
||||
task = (task or "").strip()
|
||||
task_block = f"Task to plan:\n{task}\n" if task else (
|
||||
"No explicit task was given with /plan — infer the task from the "
|
||||
|
||||
+5
-1
@@ -211,7 +211,11 @@ def _check_task(policy: _TrustPolicy, *, plugin_id: str, requested_task: Optiona
|
||||
registered itself → allowed; a built-in key → only with ``allow_task_override``;
|
||||
anything else raises + logs. Never silently downgraded to ``auto``: that would
|
||||
mask the misconfiguration and could route to a main model the user steered
|
||||
elsewhere on purpose."""
|
||||
elsewhere on purpose.
|
||||
|
||||
A foreign/unknown key raises :class:`PluginLlmTrustError` and logs a warning naming the offending plugin
|
||||
and key. See #64174, #64182.
|
||||
"""
|
||||
task = (requested_task or "").strip()
|
||||
if not task or task.lower() == "auto":
|
||||
return None
|
||||
|
||||
@@ -209,6 +209,11 @@ def enable_happy_eyeballs_on_client(client) -> None:
|
||||
For callers that build clients inline (Codex OAuth/device-login). Proxy-backed
|
||||
pools are skipped (TCP connect goes to the proxy host); async clients need
|
||||
nothing (anyio already races per RFC 8305). Best-effort.
|
||||
|
||||
Proxy-backed transports (``httpcore.HTTPProxy`` / SOCKS pools) are left untouched: with a proxy in play
|
||||
the TCP connect goes to the proxy host, which is out of scope for the direct-transport racing added in
|
||||
#94388. Async clients are also left untouched — httpcore's async backend already performs RFC 8305
|
||||
racing natively via ``anyio.connect_tcp(happy_eyeballs_delay=0.25)``.
|
||||
"""
|
||||
try:
|
||||
import httpcore
|
||||
@@ -312,6 +317,8 @@ def _shared_transport_cls():
|
||||
mounted object absorbs that close while the shared pool keeps serving other clients.
|
||||
``handle_request`` stamps the owning view into ``request.extensions`` so socket-abort
|
||||
sweeps target only this client's in-flight connections on the shared pool.
|
||||
|
||||
See #10933.
|
||||
"""
|
||||
|
||||
__slots__ = ("_inner", "_closed")
|
||||
@@ -391,6 +398,9 @@ def build_keepalive_http_client(base_url: str = "", *, async_mode: bool = False,
|
||||
``HTTPTransport`` through a ``_SharedTransport`` view, so N delegated children share one
|
||||
connection pool + SSL context. Async clients are never shared: an httpcore async pool is
|
||||
bound to the event loop that first used it. Proxy-backed clients keep httpx's own transport.
|
||||
|
||||
See #12952, #54049.
|
||||
See #10933.
|
||||
"""
|
||||
try:
|
||||
import httpx
|
||||
|
||||
+86
-1
@@ -150,6 +150,10 @@ HERMES_AGENT_HELP_GUIDANCE_NO_SKILLS = (
|
||||
)
|
||||
|
||||
|
||||
# Memory guidance (#95681, consolidated): ONE block from ONE builder. The opening frame adapts to which
|
||||
# stores config enables; everything else is written exactly once. Leads with the positive posture (save
|
||||
# proactively, replace when full) — the routing rules come after, as refinements, not as the headline. WHAT
|
||||
# belongs in memory is the memory tool schema's job and is never re-taught here.
|
||||
def build_memory_guidance(memory_enabled: bool = True, profile_enabled: bool = True) -> str:
|
||||
"""ONE memory-guidance block whose opening frame adapts to the enabled store(s); "" when both are off.
|
||||
|
||||
@@ -199,6 +203,17 @@ SESSION_SEARCH_GUIDANCE = (
|
||||
# ("After completing a complex task (5+ tool calls)... save the approach as a skill...") on subscription OAuth
|
||||
# credentials, surfacing as a billing-shaped HTTP 400. If you rewrite it, re-verify with a subscription OAuth
|
||||
# token — sk-ant-api keys do not hit the filter. The safety-rule heading is referenced by tests and compaction summaries.
|
||||
# Anthropic's server-side content filter rejects the previous phrasing ("After completing a complex task (5+
|
||||
# tool calls), fixing a tricky error, or discovering a non-trivial workflow, save the approach as a skill
|
||||
# with skill_manage so you can reuse it next time.") on subscription OAuth credentials, and surfaces that
|
||||
# rejection as a billing-shaped HTTP 400 ("You're out of extra usage"), which sends users to buy quota they
|
||||
# do not need. Bisected against the live API: that sentence alone reproduces the 400 and removing it alone
|
||||
# clears it; size and the system[0] identity gate were both ruled out. The reword is empirically validated,
|
||||
# not understood — if you rewrite this sentence, re-verify against a subscription OAuth token, not an
|
||||
# sk-ant-api… key, which does not hit the filter. Dieted (#95681, maintainer-directed): the record-it /
|
||||
# patch-it coaching that used to open this block duplicated the ## Skills section (which teaches both "offer
|
||||
# to save as a skill" and "fix it with skill_manage(action='patch')") and skill_manage's own schema. Only
|
||||
# the compaction-pruning contract lives here — nothing else teaches it.
|
||||
SKILLS_GUIDANCE = (
|
||||
"When you work out a non-trivial workflow, record it with skill_manage for future reuse.\n\n"
|
||||
"## Skill Safety Rule\n"
|
||||
@@ -308,6 +323,14 @@ TOOL_USE_ENFORCEMENT_MODELS = ("gpt", "codex", "gemini", "gemma", "grok", "glm",
|
||||
# traces showed the same failure modes; Muse Spark stops after a chat-only turn on defaults). Gemini/Gemma get
|
||||
# GOOGLE_MODEL_OPERATIONAL_GUIDANCE instead; Claude does not exhibit these modes. Any model can opt in via
|
||||
# config.yaml (`true` or a substring list).
|
||||
# Model name substrings whose sessions receive OPENAI_MODEL_EXECUTION_GUIDANCE (execution discipline: tool
|
||||
# persistence, mandatory tool use for arithmetic, external-write read-back, count reconciliation, literal
|
||||
# preservation, verification-gated completion) when agent.execution_guidance is "auto". gpt/codex/grok are
|
||||
# the historical set; deepseek/kimi/qwen/glm/minimax/ mimo/mistral were added after Composio agentic-eval
|
||||
# traces showed the same failure modes on those families (financial math in prose, no read-back after
|
||||
# external writes, identifier "repair", completeness claims despite count mismatches). GLM's
|
||||
# tool-calls-as-plain-text stall (#53847) and MiMo (#41874) are covered here too. Gemini/Gemma are excluded
|
||||
# — they get the more specific GOOGLE_MODEL_OPERATIONAL_GUIDANCE block instead.
|
||||
EXECUTION_GUIDANCE_MODELS = (
|
||||
"gpt", "codex", "grok",
|
||||
"deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral", "muse",
|
||||
@@ -329,6 +352,20 @@ TASK_COMPLETION_GUIDANCE = (
|
||||
|
||||
# Universal parallel-tool-call guidance (ALL models): the runtime already executes independent calls
|
||||
# concurrently. Supersedes the former Google-only bullet so no model receives the steer twice.
|
||||
# Why this matters for cost: every assistant turn resends the entire accumulated conversation (and, on
|
||||
# cache-friendly providers, re-reads the cached prefix and pays for the newly-appended turn). A model that
|
||||
# issues one tool call per turn multiplies the number of round-trips — and therefore the resent context —
|
||||
# for any task that needs several independent reads, searches, or safe lookups. Batching independent calls
|
||||
# into a single assistant response collapses N turns into one, cutting both latency and the resent-context
|
||||
# cost that compounds over a long conversation. The hermes-agent runtime already executes a batch of tool
|
||||
# calls concurrently when they are independent (read-only tools always; path-scoped file ops when their
|
||||
# targets don't overlap — see run_agent._execute_tool_calls / tool_dispatch_helpers). The missing piece was
|
||||
# telling the *model* to emit those calls together in the first place. Until now the only batching steer in
|
||||
# the prompt lived in GOOGLE_MODEL_OPERATIONAL_GUIDANCE — Gemini/Gemma got it, every other model got
|
||||
# nothing. Short on purpose — shipped in the cached system prompt to every user, every session. Token cost
|
||||
# is paid once at install and amortised across all sessions via prefix caching. Keep it tight. Ported from
|
||||
# cline/cline#11514 ("encourage parallel tool calls"), adapted from Cline's TypeScript tool-surface guidance
|
||||
# to hermes-agent's Python prompt-assembly architecture.
|
||||
PARALLEL_TOOL_CALL_GUIDANCE = (
|
||||
"# Parallel tool calls\n"
|
||||
"When you need several pieces of information that don't depend on each other, request them together in a "
|
||||
@@ -342,6 +379,15 @@ PARALLEL_TOOL_CALL_GUIDANCE = (
|
||||
# Execution-discipline guidance for models that abandon partial results, skip prerequisite lookups, answer
|
||||
# from memory, or declare "done" unverified. Body is family-agnostic (OPENAI_ prefix reflects origin).
|
||||
# Injection gate: system_prompt.py via config.yaml ``agent.execution_guidance`` (auto/true/false/list).
|
||||
# OpenAI GPT/Codex-specific execution guidance. Addresses known failure modes where GPT models abandon work
|
||||
# on partial results, skip prerequisite lookups, hallucinate instead of using tools, and declare "done"
|
||||
# without verification. Inspired by patterns from OpenAI's GPT-5.4 prompting guide & OpenClaw PR #38953.
|
||||
# Also applied to xAI Grok — same failure modes in practice (claims completion without tool calls, suggests
|
||||
# workarounds instead of using existing tools, replies with plans/suggestions instead of executing). As of
|
||||
# the Composio agentic-eval follow-up, the block is no longer fenced to gpt/codex/grok: eval traces showed
|
||||
# DeepSeek/Kimi doing financial math in prose, skipping read-back verification after external writes,
|
||||
# "repairing" malformed identifiers, and claiming completeness despite count mismatches — exactly the
|
||||
# failure modes this block targets.
|
||||
OPENAI_MODEL_EXECUTION_GUIDANCE = (
|
||||
"# Execution discipline\n"
|
||||
"<tool_persistence>\n"
|
||||
@@ -462,6 +508,13 @@ def format_steer_marker(steer_text: str) -> str:
|
||||
|
||||
STEER_CHANNEL_NOTE = (
|
||||
# Only what the marker cannot say about itself: it is the ONLY trusted shape and carries full user authority.
|
||||
# Dieted (#95681, maintainer-directed). History: #40240 added this note when the marker was bare and
|
||||
# models refused steers as prompt injection (screenshot-verified). The marker has since become
|
||||
# self-describing — it declares its own provenance ("a direct message from the user...") and its own
|
||||
# replay rule ("not a new delivery when replayed from conversation history") at delivery time — so the
|
||||
# prompt-side briefing keeps only what the marker cannot say about itself: it is the ONLY trusted shape
|
||||
# (anti-lookalike), and it carries full user authority. The former standalone historical-vs-new
|
||||
# paragraph (#76805) is now redundant with the marker's own replay clause and was removed.
|
||||
"## Mid-turn user steering\n"
|
||||
"Mid-turn, the user can steer you: Hermes appends their message to the end of a tool result, wrapped exactly as:\n"
|
||||
f"{STEER_MARKER_OPEN}\n<their message>\n{STEER_MARKER_CLOSE}\n"
|
||||
@@ -681,6 +734,12 @@ PLATFORM_HINTS = {
|
||||
|
||||
# Telegram rich-messages extension — injected only with
|
||||
# ``platforms.telegram.extra.rich_messages: true`` (gateway.* or top-level).
|
||||
# NOTE: a "webui" hint lived here until 2026-08-29. It was a ghost (verified in the all-platform hint audit,
|
||||
# PR #97873): no code path constructs platform="webui" — the dashboard chat resolves to 'desktop' or 'tui'
|
||||
# (tui_gateway/server.py:_resolve_session_platform), and the browser chat tab is an xterm.js PTY hosting the
|
||||
# TUI, not an HTML chat renderer. Its content (tables/LaTeX/Mermaid, MEDIA: rich previews incl. Excalidraw)
|
||||
# described a renderer that does not exist anywhere in web/. If a real WebUI chat surface ships, write a
|
||||
# hint from its actual renderer — do not resurrect this text.
|
||||
TELEGRAM_RICH_MESSAGES_HINT = (
|
||||
"Telegram now supports rich Markdown, so lean into it: whenever it makes the answer clearer or easier to scan, "
|
||||
"actively reach for real Markdown tables (pipe `| col | col |` syntax), bullet and numbered lists, task lists (`- "
|
||||
@@ -737,7 +796,13 @@ def _plugin_backend_is_remote(backend: str) -> bool:
|
||||
|
||||
|
||||
def _windows_marketing_version() -> str:
|
||||
""""10"/"11" (``platform.release()`` says 10 for both; 11 is build >= 22000)."""
|
||||
""""10"/"11" (``platform.release()`` says 10 for both; 11 is build >= 22000).
|
||||
|
||||
``platform.release()`` reports the kernel version, which is ``10`` for BOTH Windows 10 and Windows 11 —
|
||||
the prompt then claims "Windows (10)" on Windows 11 hosts and misleads the model about the OS (#51755).
|
||||
Windows 11 is distinguished by build number: >= 22000 is 11. Falls back to ``platform.release()`` on any
|
||||
lookup failure.
|
||||
"""
|
||||
try:
|
||||
return "11" if sys.getwindowsversion().build >= 22000 else "10" # type: ignore[attr-defined]
|
||||
except Exception:
|
||||
@@ -974,6 +1039,9 @@ def drain_truncation_warnings() -> list:
|
||||
|
||||
# Skills index (two-layer cache: in-process LRU, then disk snapshot).
|
||||
# One entry per profile × platform (key carries skills_dir); a multiplexing gateway needs more than a handful.
|
||||
# Sized for multi-profile processes: since #86313 the cache key carries a per-profile skills_dir (one entry
|
||||
# per profile × platform), so the old cap of 8 could thrash on a gateway multiplexing default + several bots
|
||||
# (each miss = full os.walk manifest rebuild). ~32 costs low single-digit MB worst case.
|
||||
_SKILLS_PROMPT_CACHE_MAX = 32
|
||||
_SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict()
|
||||
_SKILLS_PROMPT_CACHE_LOCK = threading.Lock()
|
||||
@@ -1361,6 +1429,11 @@ def load_soul_md(context_length: Optional[int] = None, home_override: "Path | No
|
||||
|
||||
Callers must pass ``skip_soul=True`` to ``build_context_files_prompt`` so it isn't injected twice.
|
||||
``home_override`` pins the profile home (a thread that lost the HERMES_HOME ContextVar reads the wrong one).
|
||||
|
||||
``home_override`` scopes the read to an explicit profile home (the agent knows its own home from its
|
||||
session_db path). Without it, resolution is ambient — which on a thread that lost the HERMES_HOME
|
||||
ContextVar falls back to the launch home and reads the wrong profile's SOUL.md (#50233, same class as
|
||||
the skills-index leak fixed in #86313).
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import ensure_hermes_home
|
||||
@@ -1423,6 +1496,14 @@ def _load_agents_md(cwd_path: Path, context_length: Optional[int] = None) -> str
|
||||
|
||||
Per directory the first of ``AGENTS.override.md`` / ``AGENTS.md`` / ``agents.md`` wins (a gitignored
|
||||
personal override shadows the committed file); identical content seen again down the chain is skipped.
|
||||
|
||||
Each directory on the chain (see ``_agents_md_directory_chain``) contributes its ``AGENTS.override.md``
|
||||
/ ``AGENTS.md`` / ``agents.md`` (first name wins per directory) as its own provenance-labelled section.
|
||||
``AGENTS.override.md`` wins over ``AGENTS.md`` so a developer can keep a personal, typically-gitignored
|
||||
override next to the committed project instructions without editing the tracked file (same convention as
|
||||
earendil-works/pi#7681). Identical content encountered again further down the chain (copied or symlinked
|
||||
files) is deduplicated. With a single match — the common case, and always the case outside a git repo —
|
||||
output is identical to the historical single-file behavior.
|
||||
"""
|
||||
cwd_resolved = cwd_path.resolve()
|
||||
sections: list[str] = []
|
||||
@@ -1483,6 +1564,10 @@ def build_context_files_prompt(
|
||||
cwd_path = Path(cwd if cwd is not None else os.getcwd()).resolve()
|
||||
# A FALLBACK-picked cwd inside the Hermes install tree must not gain system-prompt authority (the desktop
|
||||
# default would load this repo's contributor AGENTS.md). An explicit cwd is honored verbatim.
|
||||
# An explicitly configured cwd is honored verbatim — the Hermes tree is a legitimate workspace when the
|
||||
# user deliberately points a session at it — and CLI-style surfaces pass
|
||||
# allow_install_tree_fallback=True because their launch dir IS the user's shell cwd (developing Hermes
|
||||
# in-tree). See #64590.
|
||||
from agent.runtime_cwd import _is_install_tree
|
||||
if cwd is None and not allow_install_tree_fallback and _is_install_tree(cwd_path):
|
||||
logger.warning(
|
||||
|
||||
@@ -73,6 +73,8 @@ def _conversation_generation(session_key: str, source: str, session_db: Any) ->
|
||||
The declared key survives ``/new``, so hashing it alone would reuse one scope across
|
||||
conversations. The counter advances with each reset boundary, independent of prunable
|
||||
rows and wall-clock; compression does not advance it.
|
||||
|
||||
The declared key names a chat and deliberately survives `/new` and policy resets. See #79017, #86733.
|
||||
"""
|
||||
reader = getattr(session_db, "latest_conversation_boundary", None)
|
||||
if not callable(reader):
|
||||
@@ -99,6 +101,11 @@ def declared_conversation_scope(agent: Any) -> Optional[str]:
|
||||
if sid and db is not None:
|
||||
try:
|
||||
# One read for both halves of the row identity (fork verdict + source).
|
||||
# One read for both halves of the row's identity: the fork verdict and the source the peer
|
||||
# queries match on live on the same ``sessions`` row, and asking for them separately read it
|
||||
# twice per resolution (@teknium1 on #98811). A SessionDB without the combined view keeps the
|
||||
# original call, so nothing that predates it — including the doubles that certify the
|
||||
# fail-closed contract below — changes behaviour.
|
||||
identity = getattr(db, "declared_scope_identity", None)
|
||||
if callable(identity):
|
||||
is_fork, row_source = identity(sid)
|
||||
@@ -158,7 +165,14 @@ def declared_conversation_scope_safe(agent: Any) -> Optional[str]:
|
||||
|
||||
def resolve_prompt_cache_scope_safe(agent: Any) -> Optional[str]:
|
||||
"""Never-raising variant of :func:`resolve_prompt_cache_scope` (None = use the physical id).
|
||||
At turn_context an exception inside ``set_runtime_main(...)`` would skip the whole binding."""
|
||||
At turn_context an exception inside ``set_runtime_main(...)`` would skip the whole binding.
|
||||
|
||||
Returns None on any failure (or when there is no scope). Consumers treat None/empty as "fall back to the
|
||||
physical session_id", so a resolution failure degrades to pre-#79017 behavior instead of blocking the
|
||||
caller — important at turn_context's call site, where an exception raised inside the
|
||||
``set_runtime_main(...)`` argument list would otherwise skip the whole runtime binding, not just the
|
||||
cache scope.
|
||||
"""
|
||||
try:
|
||||
return resolve_prompt_cache_scope(agent) or None
|
||||
except Exception:
|
||||
|
||||
@@ -280,6 +280,12 @@ def apply_anthropic_cache_control(
|
||||
marker and the remaining two go to the latest cacheable non-system messages; otherwise
|
||||
the legacy system-and-3 layout applies. Idempotent: pre-existing markers are stripped from
|
||||
a per-message copy first. Returns a shallow list copy with deep copies of modified messages.
|
||||
|
||||
Idempotent: pre-existing ``cache_control`` markers are stripped from a per-message copy before new ones
|
||||
are placed, so calling this twice (or handing it messages a prior call already marked) can never
|
||||
accumulate past 4 markers. Only messages that already carry a marker pay the copy cost — a shallow
|
||||
top-level copy suffices because :func:`strip_anthropic_cache_control` is copy-on-write on content parts
|
||||
— and the rest of the copy-on-write contract is unchanged (#90971).
|
||||
"""
|
||||
if not api_messages:
|
||||
return api_messages
|
||||
|
||||
@@ -71,6 +71,13 @@ _BEARER_PROVIDERS: Dict[str, Tuple[str, ...]] = {
|
||||
# two require-rules on one host would reject each other's requests; the sandbox gets the token
|
||||
# under every name. Authorization is also matched for Anthropic/Azure (SDKs may send Bearer);
|
||||
# Gemini's ``?key=<token>`` style is covered by match_query.
|
||||
# Providers whose API authenticates with a NON-Authorization header. iron-proxy v0.39's
|
||||
# ``secrets.replace.match_headers`` targets arbitrary header names (case-insensitive; confirmed by the
|
||||
# iron-proxy author on PR #30179 and verified in the pinned v0.39.0 source — ``swapHeaders`` +
|
||||
# ``parseHeaderMatchers``), so these are first-class swapped providers, not "uncovered". ``aliases`` are
|
||||
# interchangeable env-var names for the SAME upstream credential (Hermes' auth.py keys Google on both
|
||||
# GEMINI_API_KEY and GOOGLE_API_KEY). The sandbox receives the minted token under the canonical name AND
|
||||
# every alias so SDKs reading either work.
|
||||
_HEADER_AUTH_PROVIDERS: Dict[str, Dict[str, Tuple[str, ...]]] = {
|
||||
"ANTHROPIC_API_KEY": {"hosts": ("api.anthropic.com",), "match_headers": ("x-api-key", "Authorization"), "aliases": ()},
|
||||
"AZURE_OPENAI_API_KEY": {"hosts": ("*.openai.azure.com", "*.cognitiveservices.azure.com", "*.services.ai.azure.com"),
|
||||
|
||||
@@ -16,6 +16,7 @@ import re
|
||||
from typing import Optional, Sequence
|
||||
|
||||
#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``), never K2-era names (``kimi-k2.6``).
|
||||
# From #76427 by @ruizanthony.
|
||||
_KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)")
|
||||
|
||||
# Canonical low→high ordering for nearest-level clamping. Includes "none" so an explicit
|
||||
@@ -32,6 +33,7 @@ CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh"
|
||||
CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh")
|
||||
|
||||
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
|
||||
# : Backward-compat alias (pre-#68365-verification name).
|
||||
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
|
||||
XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
@@ -59,6 +61,9 @@ SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
#: widens it to a graded scale (live-verified, monotonic). ``xhigh`` requests the top tier.
|
||||
GLM52_EFFORTS: tuple[str, ...] = ("high", "max")
|
||||
GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
# : GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified : live on
|
||||
# api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every : level accepted with monotonic
|
||||
# reasoning-token scaling (low=4, medium=11, : high=98, max=125 on the probe prompt).
|
||||
GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
|
||||
GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
|
||||
@@ -80,7 +85,12 @@ def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
|
||||
|
||||
|
||||
def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
|
||||
"""Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3)."""
|
||||
"""Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3).
|
||||
|
||||
K3 is served as the bare slug ``k3``, plan variants like ``k3-256k``, and the ``kimi-k3*`` aliases; its
|
||||
documented set is low/high/max. Everything earlier speaks low/medium/high. Boundary-matched so K2-era
|
||||
names (``kimi-k2.6``) never match (detection regex from #76427 by @ruizanthony).
|
||||
"""
|
||||
m = (model or "").strip().lower().split("/")[-1]
|
||||
return KIMI_K3_EFFORTS if _KIMI_K3_SLUG_RE.search(m) else KIMI_K2_EFFORTS
|
||||
|
||||
|
||||
@@ -130,7 +130,12 @@ class ReasoningParamsMixin:
|
||||
def _needs_thinking_reasoning_pad(self) -> bool:
|
||||
"""True when the provider enforces ``reasoning_content`` echo-back on tool-call replays (DeepSeek, Kimi,
|
||||
MiMo thinking all 400 without it). Cached per (provider, model, base_url), invalidated by
|
||||
``switch_model()`` / ``_try_activate_fallback()`` — called ~16× per turn."""
|
||||
``switch_model()`` / ``_try_activate_fallback()`` — called ~16× per turn.
|
||||
|
||||
DeepSeek v4 thinking and Kimi / Moonshot thinking both reject replays of assistant tool-call
|
||||
messages that omit ``reasoning_content`` (refs 15250, #17400). Xiaomi MiMo thinking mode has the
|
||||
same requirement.
|
||||
"""
|
||||
key = (self.provider, self.model, getattr(self, "_base_url_lower", self.base_url))
|
||||
cached = getattr(self, "_thinking_pad_cache", None)
|
||||
if cached is not None and cached[0] == key:
|
||||
@@ -162,7 +167,11 @@ class ReasoningParamsMixin:
|
||||
return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url)
|
||||
|
||||
def _needs_deepseek_tool_reasoning(self) -> bool:
|
||||
"""True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400)."""
|
||||
"""True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400).
|
||||
|
||||
DeepSeek V4 thinking mode requires ``reasoning_content`` on every assistant tool-call turn; omitting
|
||||
it causes HTTP 400 when the message is replayed in a subsequent API request (#15250).
|
||||
"""
|
||||
return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url)
|
||||
|
||||
def _needs_mimo_tool_reasoning(self) -> bool:
|
||||
|
||||
+97
-1
@@ -19,6 +19,8 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
# Sensitive query-string param names (case-insensitive): opaque tokens / OAuth
|
||||
# codes / pre-signed signatures with no vendor prefix.
|
||||
# Ported from nearai/ironclaw#2529 — catches tokens whose values don't match any known vendor prefix regex
|
||||
# (e.g. opaque tokens, short OAuth codes).
|
||||
_SENSITIVE_QUERY_PARAMS = frozenset({
|
||||
"access_token", "refresh_token", "id_token", "token", "api_key", "apikey",
|
||||
"client_secret", "password", "auth", "jwt", "session", "secret", "key",
|
||||
@@ -28,6 +30,11 @@ _SENSITIVE_QUERY_PARAMS = frozenset({
|
||||
# Snapshot at import time so runtime env mutations (e.g. an LLM-generated
|
||||
# `export HERMES_REDACT_SECRETS=false`) cannot disable redaction mid-session.
|
||||
# ON by default; `security.redact_secrets: false` bridges to this env var.
|
||||
# ON by default — secure default per issue #17691. Users who need raw credential values in tool output (e.g.
|
||||
# working on the redactor itself) can opt out via `security.redact_secrets: false` in config.yaml (bridged
|
||||
# to this env var in hermes_cli/main.py, gateway/run.py, and cli.py) or `HERMES_REDACT_SECRETS=false` in
|
||||
# ~/.hermes/.env. An opt-out warning is logged at gateway and CLI startup so operators see the downgrade —
|
||||
# see `_log_redaction_status()` in gateway/run.py and cli.py.
|
||||
_REDACT_ENABLED = os.getenv("HERMES_REDACT_SECRETS", "true").lower() in {"1", "true", "yes", "on"}
|
||||
|
||||
# Known API key prefixes -- match the prefix + contiguous token chars.
|
||||
@@ -76,6 +83,7 @@ _PREFIX_PATTERNS = [
|
||||
r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key
|
||||
r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key
|
||||
# GitLab token families (each keeps a full literal prefix for the pre-screen).
|
||||
# Ported from openclaw/openclaw#112954; follow-up invited in #4541.
|
||||
r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token
|
||||
r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret
|
||||
r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token
|
||||
@@ -97,12 +105,17 @@ _PREFIX_PATTERNS = [
|
||||
# tolerate spaces around "=" and allow the keyword embedded anywhere
|
||||
# (``MYTOKEN=…``) — an all-caps key is almost never prose. Bare ``KEY``/``PASS``/
|
||||
# ``PW`` suffixes are included; _key_has_secret_keyword rejects ``KEYBOARD=``.
|
||||
# The regex is IGNORECASE so lowercase env names (``openai_key=…``) are caught here too. The secret name
|
||||
# must sit at a word boundary (``_``-delimited or whole-word) so generic prose words (``password=``,
|
||||
# ``token=``, ``KEYBOARD=``, ``PASSAGE=``) do not match — those are handled by the config/form/URL paths,
|
||||
# and a bare ``password=…`` in a form body must not be swallowed greedily by ``\S+``. See #77484.
|
||||
_SECRET_ENV_NAMES = r"(?:API_?KEY|KEY|TOKEN|SECRET|PASSWORD|PASSWD|PASS|PW|CREDENTIAL|AUTH)"
|
||||
_ENV_ASSIGN_RE = re.compile(rf"([A-Z0-9_]{{0,50}}{_SECRET_ENV_NAMES}[A-Z0-9_]{{0,50}})\s*=\s*(['\"]?)(\S+)\2")
|
||||
# Lowercase env names: only underscore-boundary forms (``openai_key=``) — NOT
|
||||
# bare ``password=``/``token=``, which appear in prose, URLs, and form bodies.
|
||||
# The lookbehind anchors each attempt to the start of an identifier run; without
|
||||
# it re.sub retries the greedy prefix at every byte of a long opaque payload.
|
||||
# See #77484.
|
||||
_ENV_ASSIGN_LOWER_RE = re.compile(
|
||||
rf"(?<![a-z0-9_])([a-z0-9_]+(?:_|^)(?:key|pass|pw|token|secret|password|passwd|credential|auth)(?=[^a-z0-9_]|$))\s*=\s*(['\"]?)(\S+)\2",
|
||||
re.IGNORECASE,
|
||||
@@ -113,6 +126,14 @@ _ENV_ASSIGN_LOWER_RE = re.compile(
|
||||
# whitespace AND ``&`` (form bodies go pair-by-pair via _redact_form_body);
|
||||
# _CFG_DOTTED_RE needs a NAMESPACED key; _CFG_ANCHORED_RE needs line start
|
||||
# (optionally after ``export``). The ``://`` URL guard lives at the call site.
|
||||
# The uppercase _ENV_ASSIGN_RE above never matched these, so config-file passwords leaked verbatim (issue
|
||||
# #16413). These run only in a config-file context, NOT in prose, code, or URLs — three carve-outs preserved
|
||||
# from the original design (#4367 + the documented web-URL passthrough below): 1. The value is bounded by
|
||||
# ``[^\s&]`` (stops at whitespace AND ``&``) so form-urlencoded bodies are handled pair-by-pair (by
|
||||
# _redact_form_body), not greedily swallowed. 2. _CFG_DOTTED_RE only matches when the key is NAMESPACED
|
||||
# (contains a dot), which is unambiguously a config key — never a prose word. 3. _CFG_ANCHORED_RE matches a
|
||||
# bare secret-word key only at line start (optionally after ``export``), so conversational ``I have
|
||||
# password=foo`` mid-sentence is left alone.
|
||||
_SECRET_CFG_NAMES = r"(?:api[ _.\-]?key|token|secret|passwd|password|credential|auth)"
|
||||
_CFG_VALUE = r"(['\"]?)([^\s&]+?)\2(?=[\s&]|$)"
|
||||
# Linear pre-gate for the _CFG_*_RE subs: no secret keyword => neither can match.
|
||||
@@ -158,6 +179,16 @@ _YAML_ASSIGN_RE = re.compile(
|
||||
# only at a word boundary: key edge, next to a non-letter, or a camelCase
|
||||
# transition (``clientSecret``, ``APIToken``); trailing plural ``s`` is part of
|
||||
# it. Concatenations match via explicit alternatives (``authtoken``, ``apikey``).
|
||||
# The side effect: ordinary prose/document words that merely CONTAIN a keyword also matched — ``Secretary:
|
||||
# J.Smith`` (secret), ``tokenizer: cl100k_base`` (token), ``author=Smith`` (auth) — mangling legitimate
|
||||
# content on the surfaces that run these passes (browser snapshots, log lines, kanban summaries, CLI-echoed
|
||||
# command output). Ported from nearai/ironclaw#6129, where the same substring false positive ("Secretary of
|
||||
# the Treasury" matching the ``secret`` marker) scrubbed legitimate tool results from the replayed
|
||||
# transcript and sent the model into a re-fetch loop. Common concatenated compounds keep matching via
|
||||
# explicit alternatives (``authtoken`` ngrok, ``authkey`` tailscale, ``secretkey`` minio, ``apikey``).
|
||||
# Embedded occurrences inside a larger word (``secretary``, ``tokenizer``, ``authored``, ``credentialing``)
|
||||
# no longer match. ALL-CAPS keys keep the legacy embedded matching (``MYTOKEN=…``) — an all-caps key is
|
||||
# almost never prose, the same rationale as _ENV_ASSIGN_RE.
|
||||
_KEY_KEYWORD_RE = re.compile(
|
||||
r"(?:api|auth|access|refresh|session|secret)[ _.\\-]?(?:key|token)"
|
||||
r"|token|secret|passwd|password|pass|pw|credential|auth|key",
|
||||
@@ -223,6 +254,12 @@ def _should_redact_assignment(key: str, value: str, *, check_keyword: bool) -> b
|
||||
"""Shared gate for the ENV / JSON / YAML assignment passes: skip programmatic env
|
||||
lookups used as values, optionally require a word-bounded keyword in the key,
|
||||
then redact when the key is unambiguously credential-bearing or the value looks opaque."""
|
||||
# Programmatic env lookups reference variable *names*, not secret values — masking them corrupts code
|
||||
# snippets in prose/log contexts (issue #2852): ``KEY=os.getenv('X')``.
|
||||
# Same programmatic-env-lookup exception as _redact_env above (issue #2852): "apiKey": "os.getenv('X')"
|
||||
# is a code snippet, not a leaked secret value.
|
||||
# Same programmatic-env-lookup exception as _redact_env above (issue #2852): api_key: os.getenv('X') is
|
||||
# a code snippet, not a leaked secret value.
|
||||
if _ENV_LOOKUP_VALUE_RE.match(value):
|
||||
return False
|
||||
if check_keyword and not _key_has_secret_keyword(key):
|
||||
@@ -254,6 +291,11 @@ _PRIVATE_KEY_RE = re.compile(r"-----BEGIN[A-Z ]*PRIVATE KEY-----[\s\S]*?-----END
|
||||
# Database connection strings: protocol://user:PASSWORD@host. The userinfo and
|
||||
# password groups forbid whitespace so a match can never span a line break (a
|
||||
# greedy ``[^@]+`` once ran to a decorator's ``@`` on the next code line).
|
||||
# Database connection strings: protocol://user:PASSWORD@host Catches postgres, mysql, mongodb, redis, amqp
|
||||
# URLs and redacts the password. A real DSN password never contains whitespace; without this bound the
|
||||
# greedy [^@]+ would scan past the end of a code line to the next stray "@" (e.g. a Python decorator),
|
||||
# swallowing intervening lines and corrupting tool OUTPUT for any source containing a postgresql:// f-string
|
||||
# template. See issue #33801.
|
||||
_DB_CONNSTR_RE = re.compile(
|
||||
r"((?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp)://[^:\s]+:)([^@\s]+)(@)",
|
||||
re.IGNORECASE,
|
||||
@@ -265,6 +307,11 @@ _DB_CONNSTR_RE = re.compile(
|
||||
# bare userinfo. ``user:pass@`` passes through (class forbids ``:``); DB schemes
|
||||
# belong to _DB_CONNSTR_RE. 8+ char floor skips short usernames; the class
|
||||
# forbids ``/`` so an ``@`` in a path/query (``?q=user@example.com``) never counts.
|
||||
# This is the ``git remote set-url origin https://PASSWORD@github.com/...`` shape from issue #6396 — a
|
||||
# single opaque credential in the userinfo position with NO ``user:pass`` colon. The colon form
|
||||
# ``user:pass@`` is deliberately left to pass through (commit "pass web URLs through unchanged", #34029) and
|
||||
# is NOT matched here — the token class forbids ``:``. DB schemes are handled by _DB_CONNSTR_RE above and
|
||||
# excluded here. Guards against false positives:
|
||||
_URL_BARE_TOKEN_RE = re.compile(
|
||||
r"((?:https?|wss?|git|ssh|ftp|ftps|sftp)://)" # scheme
|
||||
r"([^\s:@/]{8,})" # bare token (no colon/slash/@), 8+ chars
|
||||
@@ -320,6 +367,10 @@ def _mask_control_split_tokens(text: str, mask_fn) -> str:
|
||||
Match on a control-stripped copy, then mask the corresponding span in the
|
||||
ORIGINAL — only when that span holds solely token-body and control chars, so
|
||||
a match can never cross into another line's unrelated text.
|
||||
|
||||
A credential like ``sk-abc\\x1bdef456…`` or ``ghp_abc\\n123def…`` has its token body interrupted, so the
|
||||
contiguous _PREFIX_RE cannot match it and the secret leaks verbatim (issue #77484). ``EXA_API_KEY=*** is
|
||||
rejected).
|
||||
"""
|
||||
stripped = _CONTROL_CHARS_RE.sub("", text)
|
||||
if stripped == text:
|
||||
@@ -428,7 +479,12 @@ def _redact_form_body(text: str) -> str:
|
||||
def _mask_token_nonreusable(token: str) -> str:
|
||||
"""Redact a prefix-matched credential to a NON-REUSABLE sentinel: no head/tail
|
||||
chars (an agent once wrote a truncated-looking mask back into a config file),
|
||||
only the vendor prefix label so the credential KIND stays visible."""
|
||||
only the vendor prefix label so the credential KIND stays visible.
|
||||
|
||||
* cannot be mistaken for a usable-but-truncated key, so an agent that reads it from a config file and
|
||||
writes it back does NOT corrupt the stored credential into a dead 13-char string (issue #35519); and *
|
||||
still does not leak the secret material (no head/tail chars).
|
||||
"""
|
||||
label = next((sub for sub in _PREFIX_SUBSTRINGS if token.startswith(sub)), "") if token else ""
|
||||
return f"«redacted:{label}…»" if label else "«redacted-secret»"
|
||||
|
||||
@@ -451,9 +507,20 @@ def _redact_assignments(text: str) -> str:
|
||||
_redact_env = _assignment_sub(lambda g: f"{g[0]}={g[1]}{_mask_token(g[2])}{g[1]}", check_keyword=True)
|
||||
text = _ENV_ASSIGN_RE.sub(_redact_env, text)
|
||||
if "://" not in text: # lowercase names would match URL params
|
||||
# Skip URLs — the query string may contain ``token=``/``key=`` params that are intentionally
|
||||
# passed through (see note near the bottom of this function; _redact_strict_url_credentials
|
||||
# handles the opt-in case). The uppercase regex above is all-caps-only, so it never matches URL
|
||||
# params; the lowercase one would (issue #77484).
|
||||
text = _ENV_ASSIGN_LOWER_RE.sub(_redact_env, text)
|
||||
# The keyword pre-gate is exact and matters: _CFG_DOTTED_RE backtracks
|
||||
# quadratically on long unbroken [A-Za-z0-9_.\-] runs.
|
||||
# Lowercase/dotted config keys (issue #16413). Skip URLs entirely — web-URL query params are
|
||||
# intentionally passed through (see note near the bottom of this function); _DB_CONNSTR_RE still
|
||||
# guards connection-string passwords. Extra gate: every _CFG_*_RE match requires a secret keyword in
|
||||
# the key, so a text without any secret keyword cannot match — skipping is exact. This matters
|
||||
# because _CFG_DOTTED_RE backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs (e.g.
|
||||
# base64/hex blobs in compaction payloads); the linear keyword scan prevents that pathological path
|
||||
# on secret-free text.
|
||||
if "://" not in text and _CFG_SECRET_WORD_RE.search(text):
|
||||
text = _CFG_DOTTED_RE.sub(_redact_env, text)
|
||||
text = _CFG_ANCHORED_RE.sub(_redact_env, text)
|
||||
@@ -507,6 +574,11 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
|
||||
|
||||
Every regex sits behind a cheap substring gate that its pattern requires,
|
||||
so the gates are never false-negative.
|
||||
|
||||
Set file_read=True for file *content* returned to the agent (read_file / search_files / cat). The old
|
||||
mask looked like a real-but-truncated key, so an agent reading it from config.yaml and writing it back
|
||||
silently corrupted the stored credential into a dead 13-char value → 401 (issue #35519). The sentinel is
|
||||
syntactically invalid as a token, so it can't be mistaken for a usable key or written back as one.
|
||||
"""
|
||||
if text is None:
|
||||
return None
|
||||
@@ -518,6 +590,10 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
|
||||
# Control/zero-width chars can split a token body so _PREFIX_RE alone misses it.
|
||||
if _has_known_prefix_substring(text):
|
||||
_prefix_sub = _mask_token_nonreusable if file_read else _mask_token
|
||||
# Control/zero-width chars (\\n, \\r, ESC, U+200B, …) split a token body so _PREFIX_RE cannot match
|
||||
# across them — a secret smuggled as ``sk-abc\\x1bdef…`` leaks verbatim (issue #77484). Mask such
|
||||
# runs by first matching on a control-stripped copy, then re-masking the corresponding span in the
|
||||
# original (the stripped copy and the original are aligned 1:1 for non-control chars).
|
||||
text = _mask_control_split_tokens(text, _prefix_sub)
|
||||
text = _PREFIX_RE.sub(lambda m: _prefix_sub(m.group(1)), text)
|
||||
|
||||
@@ -534,6 +610,11 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
|
||||
if "BEGIN" in text and "-----" in text:
|
||||
text = _PRIVATE_KEY_RE.sub("[REDACTED PRIVATE KEY]", text)
|
||||
|
||||
# Database connection string passwords. With code_file=True, a password group that is a pure ``{...}``
|
||||
# brace expression is an f-string template reference (e.g. f"postgresql://{user}:{pass}@{host}"), not a
|
||||
# literal credential — preserve it. Literal passwords are still redacted. The regex forbids whitespace
|
||||
# in the password group, so a single-line template's group(2) is exactly the brace expression. See issue
|
||||
# #33801.
|
||||
if "://" in text:
|
||||
text = _redact_url_credentials(text, code_file)
|
||||
|
||||
@@ -541,6 +622,15 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
|
||||
text = _JWT_RE.sub(lambda m: _mask_token(m.group(0)), text)
|
||||
|
||||
if redact_url_credentials: # opt-in; known credential shapes in URLs are caught above
|
||||
# NOTE: Web-URL redaction (query params + userinfo + HTTP access-log request targets) is
|
||||
# intentionally OFF. Many legitimate workflows pass opaque tokens through query strings — magic-link
|
||||
# checkouts, OAuth callbacks the agent is meant to follow, pre-signed share URLs — and
|
||||
# blanket-redacting param values by name breaks those skills mid-flow. DB connection-string
|
||||
# passwords are still caught by _DB_CONNSTR_RE. The ONE userinfo case still redacted is the
|
||||
# colon-less bare-token form ``scheme://TOKEN@host`` (#6396, handled by _URL_BARE_TOKEN_RE in the
|
||||
# ``://`` block above): a bare credential in userinfo is never a round-trip workflow token (those
|
||||
# live in the query string), so masking it can't break a skill. The ``user:pass@`` form is left to
|
||||
# pass through per #34029.
|
||||
text = _redact_strict_url_credentials(text)
|
||||
|
||||
if "&" in text and "=" in text:
|
||||
@@ -555,6 +645,10 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F
|
||||
# Commands whose stdout is an env-var dump: terminal redaction runs the
|
||||
# ENV-assignment pass (code_file=False) for these so opaque tokens with no vendor
|
||||
# prefix are masked; everything else uses code_file=True (``MAX_TOKENS=100``).
|
||||
# Commands whose stdout is an environment-variable dump (KEY=value lines), NOT source code.
|
||||
# ``MY_SERVICE_TOKEN=abc123randomstring``) are still masked. For all other commands, code_file=True is used
|
||||
# to avoid mangling legitimate source/config dumps (``MAX_TOKENS=100``, ``"apiKey": "x"`` fixtures,
|
||||
# ``postgresql://{user}`` f-string templates). See issue #43025.
|
||||
_ENV_DUMP_COMMANDS = frozenset({"env", "printenv", "set", "export", "declare"})
|
||||
|
||||
# Commands that read file contents to stdout. A ``.env`` target is a credential
|
||||
@@ -702,6 +796,8 @@ def _has_known_prefix_substring(text: str) -> bool:
|
||||
# ADDITIVE-ONLY: a plugin can extend what gets masked but cannot weaken a
|
||||
# built-in, so it can only over-redact. Keyed by registration source so plugin
|
||||
# unload has a clean seam to drop ONE plugin's patterns.
|
||||
# There is deliberately no public removal API — additive-only stands; unload is a host-owned lifecycle
|
||||
# concern. See #64229.
|
||||
_PLUGIN_PREFIX_PATTERNS: dict = {}
|
||||
_registry_lock = threading.Lock()
|
||||
|
||||
|
||||
+12
-1
@@ -98,6 +98,10 @@ class _ManagedAttempt:
|
||||
Relay can invoke callbacks while another still owns the captured Context (hence the
|
||||
copy); nested relay calls run unmanaged — see relay_runtime.managed_callback_guard."""
|
||||
def guarded() -> Any:
|
||||
# See #77244.
|
||||
# See #77244.
|
||||
# Hermes-side callbacks run while the native pipeline drives this stream; nested relay calls
|
||||
# they make must bypass managed execution (#77244).
|
||||
with relay_runtime.managed_callback_guard():
|
||||
return callback(*args)
|
||||
|
||||
@@ -223,7 +227,14 @@ def stream_current(
|
||||
With ``completed_response_predicate`` set, a factory that ignores ``stream=True`` and returns a
|
||||
complete response is unwrapped and returned directly (pre-Relay behavior). Detecting that primes
|
||||
the lazy pipeline: a genuine first chunk is buffered, but provider latency and pre-first-yield
|
||||
errors may surface before this returns."""
|
||||
errors may surface before this returns.
|
||||
|
||||
AnthropicAuxiliaryClient and other shims that ignore ``stream=True``), unwrap and return the completed
|
||||
response directly. This mirrors the pre-Relay behavior where ``call_llm(stream=True)`` returned the raw
|
||||
response and the consumer's own ``hasattr(stream, "choices")`` check handled it (#11732, #55933) —
|
||||
without the unwrap the response stays trapped as ``final_response`` on the inner ManagedLlmStream and
|
||||
the outer consumer sees an empty stream.
|
||||
"""
|
||||
session_id = _current_session_id()
|
||||
# Inside a managed callback (on the Relay session's loop) a nested ManagedLlmStream would be
|
||||
# iterated synchronously on that loop, which asyncio forbids; the outer stream tracks this attempt.
|
||||
|
||||
@@ -1139,6 +1139,13 @@ def resolve_execution_context(session_id: str) -> tuple[RelayRuntime | None, Rel
|
||||
# Nested managed execution is impossible (see _MANAGED_CALLBACK_DEPTH); the outer scope
|
||||
# still records the tool-level event.
|
||||
if _MANAGED_CALLBACK_DEPTH.get() > 0 or not relay_instrumentation_enabled():
|
||||
# A managed Relay callback is already executing on this logical call path (e.g. the native
|
||||
# ``tools.execute`` pipeline is mid-dispatch of a Hermes tool). Nested managed execution here is
|
||||
# structurally impossible: the native pipeline binds its Futures to the OUTER call's event loop,
|
||||
# which is blocked inside the synchronous tool callback until the tool returns. A nested managed LLM
|
||||
# call (the vision_analyze auxiliary path) therefore awaits a foreign-loop Future that can never
|
||||
# complete — "attached to a different loop" at best, deadlock at worst, and "Event loop is closed"
|
||||
# during shutdown when the orphaned Future is completed late (#77244).
|
||||
return None, None, None
|
||||
turn = active_turn(session_id)
|
||||
host = turn.lease.live_runtime() if turn is not None else None
|
||||
|
||||
@@ -30,6 +30,7 @@ def execute(
|
||||
# Everything the tool transitively calls (incl. auxiliary LLM calls on worker
|
||||
# threads) must bypass managed Relay: the pipeline's Futures bind to THIS loop,
|
||||
# which is blocked until the tool returns.
|
||||
# See #77244.
|
||||
with relay_runtime.managed_callback_guard():
|
||||
return callback(final_args)
|
||||
|
||||
|
||||
@@ -25,7 +25,11 @@ _DOMINANCE_RATIO = 0.5
|
||||
|
||||
def is_repetition_dominated(text: str) -> bool:
|
||||
"""True when a single 60+ char substring recurs often enough to cover at least half
|
||||
of ``text`` — the signature of a repetition loop. Fail-open for non-string/short input."""
|
||||
of ``text`` — the signature of a repetition loop. Fail-open for non-string/short input.
|
||||
|
||||
That shape is the signature of a model repetition loop (issue #86581), and continuing such a fragment is
|
||||
pointless — the continuation nudge would just stitch more repeated text into the final response.
|
||||
"""
|
||||
if not isinstance(text, str):
|
||||
return False
|
||||
n = len(text)
|
||||
|
||||
+15
-2
@@ -95,7 +95,12 @@ def strip_interrupted_tool_tails(agent_history: List[Dict[str, Any]]) -> List[Di
|
||||
def strip_dangling_tool_call_tail(agent_history: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""Strip a trailing ``assistant(tool_calls)`` with NO answers — a call that killed the gateway itself
|
||||
(``docker restart``) left zero ``tool`` rows, invisible to ``strip_interrupted_tool_tails``. A partially
|
||||
answered block still resumes. Read-only tails are dropped; side-effecting ones get UNKNOWN-effect results."""
|
||||
answered block still resumes. Read-only tails are dropped; side-effecting ones get UNKNOWN-effect results.
|
||||
|
||||
On resume the model sees an unanswered tool call at the tail and naturally re-issues it — which restarts
|
||||
the gateway again, producing the infinite reboot loop in #49201. ``strip_interrupted_tool_tails`` does
|
||||
not catch this because there is no tool result to inspect for an interrupt marker.
|
||||
"""
|
||||
if not agent_history:
|
||||
return agent_history
|
||||
last = agent_history[-1]
|
||||
@@ -124,6 +129,8 @@ def sanitize_replay_history(agent_history: List[Dict[str, Any]]) -> List[Dict[st
|
||||
# --- Stale dangerous-confirmation text expiry ---
|
||||
|
||||
# Short on purpose: a dangerous confirmation must not survive any restart or resume gap.
|
||||
# ────────────────────────────────────────────────────────────────────── Stale dangerous-confirmation text
|
||||
# expiry (#59607) ──────────────────────────────────────────────────────────────────────
|
||||
_DANGEROUS_CONFIRMATION_EXPIRY_SECONDS = 60.0
|
||||
|
||||
# Phrases that unlock destructive host actions; case-insensitive substring match so trailing punctuation /
|
||||
@@ -155,7 +162,13 @@ def strip_stale_dangerous_confirmations(
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Redact IN PLACE dangerous-confirmation text older than ``expiry_seconds`` in user messages: a confirmation
|
||||
surviving a restart reads as a fresh re-confirmation minutes later. Untimestamped messages (legacy
|
||||
transcripts, test scaffolding) are left untouched."""
|
||||
transcripts, test scaffolding) are left untouched.
|
||||
|
||||
See #59607.
|
||||
On the next inbound message — possibly a casual "are you there?" from the user minutes later — the LLM
|
||||
sees the stale confirmation and may interpret the new turn as a fresh re-confirmation, re-executing the
|
||||
destructive action. This is the failure mode reported in #59607.
|
||||
"""
|
||||
if not agent_history:
|
||||
return agent_history
|
||||
cleaned: List[Dict[str, Any]] = []
|
||||
|
||||
@@ -76,6 +76,7 @@ _GLOBAL_ENV_EXACT = frozenset({
|
||||
# API-server LISTENER settings — deployment config (compose/systemd env),
|
||||
# which the scoped runner reload must keep seeing or containers silently
|
||||
# lose the api_server platform. API_SERVER_KEY is a credential: NOT here.
|
||||
# See #64674, #69379.
|
||||
"API_SERVER_ENABLED", "API_SERVER_HOST", "API_SERVER_PORT",
|
||||
"API_SERVER_CORS_ORIGINS",
|
||||
# Relay-connector ROUTING stamps injected by managed deploys. Every reader
|
||||
|
||||
@@ -242,6 +242,8 @@ def _b64e(raw: bytes) -> str:
|
||||
def _derive_encrypted_cache_key(access_token: str, salt: bytes) -> bytes:
|
||||
"""HKDF the local cache key from the bootstrap BWS token. cryptography is imported
|
||||
lazily: eagerly mapping ``_rust.pyd`` on Windows blocks the updater replacing it."""
|
||||
# Keep the native cryptography extension lazy. Most CLI commands import this module while building
|
||||
# argparse, even though only encrypted-cache reads/writes need it. See #73381.
|
||||
from cryptography.hazmat.primitives import hashes
|
||||
from cryptography.hazmat.primitives.kdf.hkdf import HKDF
|
||||
|
||||
|
||||
@@ -174,7 +174,12 @@ def list_sources(*, scope: Optional[str] = None) -> List[SecretSource]:
|
||||
|
||||
def list_plugin_sources() -> List[SecretSource]:
|
||||
"""Sources registered outside the bundled set: global ``"plugin"`` origins
|
||||
plus every scoped registration (bundled sources register with scope=None)."""
|
||||
plus every scoped registration (bundled sources register with scope=None).
|
||||
|
||||
Includes both legacy global plugin registrations (``_SOURCE_ORIGINS == "plugin"``) and the current
|
||||
scope's profile-keyed registrations — every scoped entry is plugin-registered by definition, since
|
||||
bundled sources register with ``scope=None`` (#64229 profile isolation).
|
||||
"""
|
||||
_ensure_builtin_sources()
|
||||
with _REGISTRY_LOCK:
|
||||
merged = {n: s for n, s in _SOURCES.items() if _SOURCE_ORIGINS.get(n) == "plugin"}
|
||||
@@ -374,6 +379,9 @@ def apply_all(secrets_cfg: dict, home_path: Path,
|
||||
Profile aliasing: under a named profile an applied ``FOO_<PROFILE>``
|
||||
(credential-shaped suffixes only) also hydrates canonical ``FOO``, under the
|
||||
same guards; disabled with ``secrets.profile_alias: false``.
|
||||
|
||||
1. 2. 3. 4. See #58073.
|
||||
See #51447.
|
||||
"""
|
||||
env = environ if environ is not None else os.environ
|
||||
report = ApplyReport()
|
||||
|
||||
@@ -22,6 +22,7 @@ class ActivityProvenance(str, Enum):
|
||||
|
||||
UNKNOWN = "unknown"
|
||||
# Compression writers: heartbeat, host timeout, cooldown, turn hold.
|
||||
# See #72424.
|
||||
AGENT_COMPRESSION = "agent.compression"
|
||||
AGENT_COMPRESSION_TIMEOUT = "agent.compression_timeout"
|
||||
AGENT_COMPRESSION_COOLDOWN = "agent.compression_cooldown"
|
||||
|
||||
@@ -309,7 +309,12 @@ class SessionPersistenceMixin:
|
||||
|
||||
def _persist_session(self, messages: List[Dict], conversation_history: List[Dict] = None):
|
||||
"""Save to JSON log and SQLite on any exit path. Trailing empty-response scaffolding is dropped from
|
||||
the live list; the persist override is applied to the DB row only."""
|
||||
the live list; the persist override is applied to the DB row only.
|
||||
|
||||
The persist user-message *override* is NOT applied here — it is resolved inside
|
||||
``_flush_messages_to_session_db`` and written only to the DB row, never mutating the live message
|
||||
list used by the API call (#48677 is thus closed for every persist caller, not just this one).
|
||||
"""
|
||||
from agent.agent_runtime_helpers import note_turn_persisted
|
||||
with _persist_lock(self):
|
||||
self._drop_trailing_empty_response_scaffolding(messages)
|
||||
@@ -356,7 +361,15 @@ class SessionPersistenceMixin:
|
||||
"""Persist un-flushed messages to SQLite. Dedup is the intrinsic ``_DB_PERSISTED_MARKER`` on each written
|
||||
dict — not positional slices (drift after sequence repair) nor an ``id(msg)`` set (address reuse). The
|
||||
persist override touches the written row only. A compression-closed session adopts its live tip and
|
||||
retries exactly once."""
|
||||
retries exactly once.
|
||||
|
||||
Deduplicates via an intrinsic ``_DB_PERSISTED_MARKER`` stamped on each written message dict, so
|
||||
repeated calls (from multiple exit paths) only write truly new messages — preventing the
|
||||
duplicate-write bug (#860) without relying on positional slices that can drift after
|
||||
message-sequence repair, and without a retained ``id(msg)`` set that CPython could alias onto a
|
||||
freed-then-reused address (#50372). The ``_flushed_db_message_ids`` attribute is now only a one-shot
|
||||
seed (translated to markers, then cleared each flush), not a persisted set.
|
||||
"""
|
||||
# Persistence-isolated agents (background review fork) share the parent's session_id for cache warmth;
|
||||
# a write here would land the curator's turn in the user's real history.
|
||||
if getattr(self, "_persist_disabled", False) or not self._session_db:
|
||||
|
||||
+12
-1
@@ -181,7 +181,18 @@ def iter_configured_hooks(cfg: Optional[Dict[str, Any]]) -> List[ShellHookSpec]:
|
||||
|
||||
def re_register_config_hooks() -> None:
|
||||
"""Re-register after a plugin force-reload cleared the manager's hooks; only this home's keys
|
||||
are cleared (profile A's reload never drops B), never re-prompts."""
|
||||
are cleared (profile A's reload never drops B), never re-prompts.
|
||||
|
||||
``PluginManager.discover_and_load(force=True)`` unloads via the ownership ledger and clears the
|
||||
manager's ``_hooks`` dict, which silently drops shell hooks that were registered from ``config.yaml`` at
|
||||
startup (they are config-owned, not plugin-owned, so the ledger cannot restore them). Clear the
|
||||
idempotence set and re-run ``register_from_config()`` so hooks are wired again (#60036 / PR #60267;
|
||||
tracking #64178 — salvaged from PR #64188).
|
||||
Only the idempotence keys for the *current* Hermes home are cleared — ``discover_and_load(force=True)``
|
||||
only unloads the manager scoped to that one home, so clearing every home's keys would make a
|
||||
force-reload in profile A drop profile B's still-live registration from the ledger and duplicate it on
|
||||
B's next registration call (#92682 review).
|
||||
"""
|
||||
_forget_home_registrations(_registered, _registered_lock)
|
||||
from hermes_cli.config import load_config
|
||||
register_from_config(load_config())
|
||||
|
||||
@@ -129,7 +129,15 @@ def build_bundle_invocation_message(
|
||||
loaded_skill_names, missing_skill_names)`` or ``None`` if the bundle wasn't
|
||||
found. Uninstalled members are skipped with a note; disabled ones too, since
|
||||
``_load_skill_payload`` bypasses the scan-time filter (``platform`` scopes
|
||||
that check — gateway passes it, None resolves from env)."""
|
||||
that check — gateway passes it, None resolves from env).
|
||||
|
||||
Disabled skills are also skipped: bundles load members via ``_load_skill_payload`` directly, bypassing
|
||||
the scan-time disabled filter in ``get_skill_commands()``, so the disabled list must be re-applied here.
|
||||
``platform`` scopes the check to a specific platform's ``skills.platform_disabled`` config (gateway
|
||||
dispatch passes it explicitly because the gateway handles multiple platforms in one process); when
|
||||
*None*, the platform resolves from session env vars and the global disabled list still applies. Mirrors
|
||||
the stacked-skill gate in gateway dispatch (#58888).
|
||||
"""
|
||||
info = get_skill_bundles().get(cmd_key)
|
||||
if not info:
|
||||
return None
|
||||
|
||||
+48
-5
@@ -61,7 +61,13 @@ def append_user_instruction(parts: list, instruction: str) -> str:
|
||||
"""Append the instruction line to ``parts``; return the stable prefix, which
|
||||
ends exactly at the instruction marker so (registered with
|
||||
``agent.prompt_cache_boundary``) the cache planner can break on the scaffold.
|
||||
Single construction site guarantees the prefix is a byte-prefix of the message."""
|
||||
Single construction site guarantees the prefix is a byte-prefix of the message.
|
||||
|
||||
Shared by every builder that ends a static skill scaffold with the caller-supplied volatile instruction
|
||||
(single-skill invocations, cron job prompts). Keeping construction in one place guarantees the
|
||||
registered prefix stays a byte-prefix of the built message — the invariant the request-time split
|
||||
depends on. See #81867.
|
||||
"""
|
||||
stable_prefix = "\n".join(parts) + "\n" + _SINGLE_SKILL_INSTRUCTION
|
||||
parts.append(f"{_SINGLE_SKILL_INSTRUCTION}{instruction}")
|
||||
return stable_prefix
|
||||
@@ -116,7 +122,11 @@ def _cut_after(message: str, marker: str, stop_marker: str, find) -> Optional[st
|
||||
def _resolve_skill_commands_platform() -> Optional[str]:
|
||||
"""Current platform scope for disabled-skill filtering, or None (CLI, RL,
|
||||
scripts). A change invalidates the scan cache so each platform sees its
|
||||
own ``skills.platform_disabled`` view."""
|
||||
own ``skills.platform_disabled`` view.
|
||||
|
||||
Used to detect when the active platform has shifted so :func:`get_skill_commands` can drop a stale cache
|
||||
that was populated for a different platform's ``skills.platform_disabled`` view (#14536).
|
||||
"""
|
||||
try:
|
||||
from gateway.session_context import get_session_env
|
||||
resolved_platform = os.getenv("HERMES_PLATFORM") or get_session_env("HERMES_SESSION_PLATFORM")
|
||||
@@ -127,7 +137,14 @@ def _resolve_skill_commands_platform() -> Optional[str]:
|
||||
|
||||
def _resolve_skill_commands_home() -> str:
|
||||
"""Effective Hermes home the scan is scoped to (profiles carry their own
|
||||
``skills.external_dirs``, so a profile switch must invalidate the cache)."""
|
||||
``skills.external_dirs``, so a profile switch must invalidate the cache).
|
||||
|
||||
A gateway session can switch between profiles that each carry their own ``skills.external_dirs`` (via
|
||||
``set_hermes_home_override``), but the module-level scan only tracked
|
||||
``_resolve_skill_commands_platform()``. Switching profiles without a platform change left the previous
|
||||
profile's skill list cached, so ``get_skill_commands()`` reported a cache miss for skills that only
|
||||
exist under the new profile (#88023).
|
||||
"""
|
||||
from hermes_constants import get_hermes_home
|
||||
return str(get_hermes_home())
|
||||
|
||||
@@ -248,6 +265,10 @@ def _build_skill_message(
|
||||
parts.append("")
|
||||
# Everything before the volatile instruction is a stable scaffold; the
|
||||
# registered boundary lets the cache planner break there (see append_user_instruction).
|
||||
# Everything before the caller-supplied instruction is a stable scaffold; declare the exact boundary
|
||||
# so the Anthropic cache planner can put a breakpoint on it instead of caching the whole message as
|
||||
# one atomic block (#81867). The static instruction prose stays on the stable side; the volatile
|
||||
# instruction (webhook payload, ticket IDs, timestamps) and any runtime note ride in the tail.
|
||||
stable_prefix = append_user_instruction(parts, user_instruction)
|
||||
if runtime_note:
|
||||
parts += ["", f"[Runtime note: {runtime_note}]"]
|
||||
@@ -263,6 +284,9 @@ def _render_skill_block(
|
||||
"""Bump Curator usage tracking (never fatal) and build the message block for one loaded skill."""
|
||||
loaded_skill, skill_dir, skill_name = loaded
|
||||
try:
|
||||
# Track active usage for Curator lifecycle management (#17782)
|
||||
# Track active usage for Curator lifecycle management (#17782)
|
||||
# Track active usage for Curator lifecycle management (#17782)
|
||||
from tools.skill_usage import bump_use
|
||||
bump_use(skill_name, task_id=task_id)
|
||||
except Exception:
|
||||
@@ -344,6 +368,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
|
||||
global _skill_commands, _skill_commands_platform, _skill_commands_home
|
||||
platform = _resolve_skill_commands_platform()
|
||||
home = _resolve_skill_commands_home()
|
||||
# Build into a local map and publish once, at the end. Writing straight into the global made a scan's
|
||||
# partial results visible to everything else in the process: a second, overlapping scan deduped against
|
||||
# its own (empty) ``seen_names`` but collided against the first scan's already- published slugs, logging
|
||||
# one bogus "already claimed" warning per skill — each naming the same skill as its own incumbent
|
||||
# (#74574).
|
||||
commands: Dict[str, Dict[str, Any]] = {}
|
||||
try:
|
||||
from tools.skills_tool import _skills_dir, _get_disabled_skill_names
|
||||
@@ -356,6 +385,7 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
|
||||
# Precedence: project (through the quarantine chokepoint) > local > external.
|
||||
# Resolve the local dir at call time: import-time SKILLS_DIR is frozen to
|
||||
# the launch home, but a multiplexed profile scope may have changed it.
|
||||
# See #67277.
|
||||
skills_dir = _skills_dir()
|
||||
iters = [iter_project_skill_files(d) for d in get_project_skills_dirs()]
|
||||
local = [skills_dir] if skills_dir.exists() else []
|
||||
@@ -372,6 +402,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
|
||||
# could accept the new map under a stale platform tag and serve another
|
||||
# platform's disabled-skill view.
|
||||
with _publish_lock:
|
||||
# Bare assignments are not atomic together: a reader landing between them sees the NEW map still
|
||||
# carrying the OLD platform tag, and if that stale tag happens to match its own platform it accepts
|
||||
# the map without rescanning — serving another platform's disabled-skill view, exactly the leak
|
||||
# #14536 closed. Only the publish/lookup pair is locked; the scan above (file I/O, deferred imports)
|
||||
# stays outside it.
|
||||
_skill_commands = commands
|
||||
_skill_commands_platform = platform
|
||||
_skill_commands_home = home
|
||||
@@ -382,7 +417,10 @@ def get_skill_commands() -> Dict[str, Dict[str, Any]]:
|
||||
"""Return the current skill commands mapping (scan first if empty). Rescans
|
||||
when the platform scope (one gateway serving Telegram and Discord) or the
|
||||
active profile's home (Desktop profile switch) changes, so each sees its
|
||||
own ``platform_disabled`` / ``external_dirs`` view."""
|
||||
own ``platform_disabled`` / ``external_dirs`` view.
|
||||
|
||||
See #14536, #88023.
|
||||
"""
|
||||
current_platform = _resolve_skill_commands_platform()
|
||||
current_home = _resolve_skill_commands_home()
|
||||
with _publish_lock:
|
||||
@@ -539,7 +577,12 @@ def build_preloaded_skills_prompt(skill_identifiers: list[str], task_id: str | N
|
||||
"""Load skills for session-wide CLI/TUI preloading; returns (prompt_text,
|
||||
loaded_skill_names, missing_identifiers). Disabled skills count as missing:
|
||||
this path bypasses the scan-time filter, and ``hermes -s <skill>`` must not
|
||||
force-load an operator-disabled skill."""
|
||||
force-load an operator-disabled skill.
|
||||
|
||||
Disabled skills are treated the same as missing ones: this loads via a raw identifier straight into
|
||||
``_load_skill_payload``, bypassing ``get_skill_commands()``'s scan-time disabled filter — mirrors the
|
||||
bundle-invocation gate (#59156).
|
||||
"""
|
||||
loaded_names, missing, _disabled, prompt_parts = _load_skill_blocks(
|
||||
[(raw or "").strip() for raw in skill_identifiers],
|
||||
lambda identifier: _load_skill_payload(identifier, task_id=task_id),
|
||||
|
||||
+23
-2
@@ -289,7 +289,10 @@ def parse_config_string_list(value) -> List[str]:
|
||||
"""Normalize a config value that may hold a JSON-array string into a list.
|
||||
``hermes config set`` stores lists as quoted JSON/Python-literal strings;
|
||||
treating one as a single name would silently filter nothing. A scalar
|
||||
string still means one name."""
|
||||
string still means one name.
|
||||
|
||||
See #13026, #86661.
|
||||
"""
|
||||
if isinstance(value, str):
|
||||
if value.strip().startswith("["):
|
||||
try:
|
||||
@@ -412,7 +415,14 @@ _PROJECT_ROOT_MAX_DEPTH = 64 # walk-up bound for pathological cwds
|
||||
def find_project_root(start: Optional[Path] = None) -> Optional[Path]:
|
||||
"""Nearest ancestor containing ``.git`` (dir or worktree file), or None.
|
||||
Without *start*, the surface's ``TERMINAL_CWD`` wins over process cwd so
|
||||
cron/API surfaces inherit an interactive trust decision by project identity."""
|
||||
cron/API surfaces inherit an interactive trust decision by project identity.
|
||||
|
||||
When *start* is not given, the surface's working directory wins over the process cwd: ``TERMINAL_CWD``
|
||||
is the same per-surface workdir the terminal tool and cron jobs use (a cron job sets it from its per-job
|
||||
``workdir`` without chdir'ing the scheduler process). This is what lets non-interactive surfaces inherit
|
||||
a prior interactive trust decision by project identity — and a surface with no workdir in a trusted repo
|
||||
simply resolves no project and loads nothing (#48975).
|
||||
"""
|
||||
try:
|
||||
if start is None:
|
||||
from agent.runtime_cwd import scope_terminal_cwd
|
||||
@@ -501,6 +511,16 @@ def get_untrusted_project_skills_root() -> Optional[Tuple[Path, int]]:
|
||||
# cached under HERMES_HOME, never inside the repo); "dangerous" excludes the
|
||||
# skill from index, list, view and slash commands ("caution" loads, as on the hub).
|
||||
|
||||
# ── Project skill quarantine (scan-time injection defense) ──────────────── Trust (`hermes skills trust`)
|
||||
# is a REPO-level decision made once; the repo's skill content keeps changing underneath it with every pull.
|
||||
# The hub install path runs skills_guard on install, but project skills are read straight from a checkout —
|
||||
# without this gate a `git pull` could inject a malicious skill into an already-trusted repo with no scan
|
||||
# anywhere (#48974). Every project SKILL.md's parent dir is scanned with the same skills_guard scanner the
|
||||
# hub uses (content-hash cached, so the cost is one scan per skill per content change). A "dangerous"
|
||||
# verdict quarantines the skill: it is excluded from the index, skills_list, skill_view, and slash commands.
|
||||
# "caution" loads (matches hub behavior for prose-level keyword hits) — the quarantine is for
|
||||
# high-confidence findings only. The scan cache lives under HERMES_HOME, never inside the repo (we don't
|
||||
# write artifacts into the user's checkout).
|
||||
_PROJECT_SCAN_SOURCE = "project-local"
|
||||
_PROJECT_QUARANTINE_CACHE: Dict[str, bool] = {} # skill_dir -> quarantined
|
||||
|
||||
@@ -552,6 +572,7 @@ def normalize_skill_lookup_name(identifier: str) -> str:
|
||||
# (which follows the live profile-scoped HERMES_HOME), so normalization
|
||||
# must agree with that exact root. Import deferred (cycle).
|
||||
try:
|
||||
# See #67277.
|
||||
from tools import skills_tool as _skills_tool
|
||||
primary_root = _skills_tool._skills_dir()
|
||||
except Exception:
|
||||
|
||||
@@ -58,6 +58,15 @@ class StreamDeliveryMixin:
|
||||
self._deliver_to_stream_callbacks(tail)
|
||||
self._record_streamed_assistant_text(tail)
|
||||
|
||||
# Flush any benign partial-tag tail held by the think scrubber first (#17924): an innocent '<' at
|
||||
# the end of the stream that turned out not to be a tag prefix should reach the UI. Then flush the
|
||||
# context scrubber. Order matters — the think scrubber's output feeds into the context scrubber's
|
||||
# state.
|
||||
# Suppress reasoning/thinking blocks via the stateful scrubber (#17924). Earlier versions ran
|
||||
# _strip_think_blocks per-delta here, which destroyed downstream state machines when a tag was split
|
||||
# across deltas (e.g. MiniMax-M2.7 sends '<think>' and its content as separate deltas — regex case 2
|
||||
# erased the first delta, so the CLI/gateway state machine never saw the open tag and leaked the
|
||||
# reasoning content as regular response text).
|
||||
if think_scrubber is not None:
|
||||
think_tail = think_scrubber.flush()
|
||||
deliver(ctx_scrubber.feed(think_tail) if think_tail and ctx_scrubber is not None else think_tail)
|
||||
@@ -194,7 +203,10 @@ class StreamDeliveryMixin:
|
||||
self._deliver_interim(visible, already_streamed=already_streamed, record=undelivered_parts or [visible])
|
||||
|
||||
def _ensure_stream_writer_state(self) -> None:
|
||||
"""Lazily create the single-writer guard fields (``AIAgent.__new__``-built instances skip ``agent_init``)."""
|
||||
"""Lazily create the single-writer guard fields (``AIAgent.__new__``-built instances skip ``agent_init``).
|
||||
|
||||
See #65991.
|
||||
"""
|
||||
if getattr(self, "_stream_writer_lock", None) is None:
|
||||
self._stream_writer_lock = threading.Lock()
|
||||
if getattr(self, "_stream_writer_tls", None) is None:
|
||||
@@ -209,6 +221,8 @@ class StreamDeliveryMixin:
|
||||
Every attempt (each provider path, each retry) claims right before consuming; claiming bumps
|
||||
the shared token, so an earlier attempt still alive on another thread is superseded and its
|
||||
late chunks fenced out. Stored per-thread: a thread that never claimed can never be fenced.
|
||||
|
||||
See #65991.
|
||||
"""
|
||||
self._ensure_stream_writer_state()
|
||||
with self._stream_writer_lock:
|
||||
@@ -217,11 +231,18 @@ class StreamDeliveryMixin:
|
||||
return token
|
||||
|
||||
def _stream_writer_is_current(self, token: int) -> bool:
|
||||
"""True when ``token`` is still the active writer, so a stream loop can bail the instant it is superseded."""
|
||||
"""True when ``token`` is still the active writer, so a stream loop can bail the instant it is superseded.
|
||||
|
||||
active writer — i.e. no newer stream attempt has claimed the sink since (#65991).
|
||||
"""
|
||||
return token == getattr(self, "_stream_writer_token", token)
|
||||
|
||||
def _stream_writer_superseded(self) -> bool:
|
||||
"""True when this thread claimed the sink but a newer attempt has since claimed it (never for a non-claimer)."""
|
||||
"""True when this thread claimed the sink but a newer attempt has since claimed it (never for a non-claimer).
|
||||
|
||||
stream attempt has since claimed it — i.e. this thread is a stale writer whose chunks must be
|
||||
dropped (#65991).
|
||||
"""
|
||||
token = getattr(getattr(self, "_stream_writer_tls", None), "token", None)
|
||||
return token is not None and token != getattr(self, "_stream_writer_token", token)
|
||||
|
||||
@@ -261,6 +282,7 @@ class StreamDeliveryMixin:
|
||||
"""Fire all registered stream delta callbacks (display + TTS)."""
|
||||
# A superseded stream must not interleave its tokens alongside the retry that replaced it.
|
||||
if self._stream_writer_superseded():
|
||||
# See #65991.
|
||||
self._note_dropped_stream_writer("_fire_stream_delta")
|
||||
return
|
||||
# One paragraph break before the first text delta after a tool iteration, without
|
||||
@@ -274,6 +296,7 @@ class StreamDeliveryMixin:
|
||||
# tag was split across deltas; memory-context spans split across chunks must not leak to
|
||||
# the UI. Legacy callers lack the scrubber attributes and get the whole-string fallbacks.
|
||||
think_scrubber = getattr(self, "_stream_think_scrubber", None)
|
||||
# See #5719.
|
||||
scrubber = getattr(self, "_stream_context_scrubber", None)
|
||||
text = think_scrubber.feed(text) if think_scrubber is not None else self._strip_think_blocks(text)
|
||||
text = scrubber.feed(text) if scrubber is not None else sanitize_context(text)
|
||||
@@ -291,6 +314,8 @@ class StreamDeliveryMixin:
|
||||
def _fire_reasoning_delta(self, text: str) -> None:
|
||||
"""Fire reasoning callback if registered; superseded writers are fenced like content deltas."""
|
||||
if self._stream_writer_superseded():
|
||||
# Single-writer guard (#65991): fence out a superseded stream's reasoning deltas the same way as
|
||||
# content deltas.
|
||||
self._note_dropped_stream_writer("_fire_reasoning_delta")
|
||||
return
|
||||
self._call_quietly(self.reasoning_callback, text)
|
||||
|
||||
@@ -20,6 +20,14 @@ from agent.transports.types import NormalizedResponse, ToolCall, Usage
|
||||
|
||||
# xAI reserves ``tool_search`` for its server-side tool (HTTP 400 on client
|
||||
# declarations); aliased on the wire, mapped back in normalize_response.
|
||||
# xAI's chat-completions API reserves the function name ``tool_search`` for its own server-side tool and
|
||||
# rejects any request declaring a client function with that name (HTTP 400 "The function name tool_search is
|
||||
# reserved for the tool_search tool", #95003). The Tool Search bridge (tools/tool_search.py) assembles its
|
||||
# client-side discovery tool under the same literal name for every provider, so Grok providers are unusable
|
||||
# whenever the bridge is active. Mirror the web_search treatment in transports/codex.py
|
||||
# (_rename_client_web_search_for_xai): alias the wire declaration and map the alias back in
|
||||
# normalize_response. The alias value matches _CODEX_TOOL_SEARCH_ALIAS from the Codex-side fix for the same
|
||||
# reserved-name class (#83122) so the two transports stay consistent.
|
||||
_XAI_TOOL_SEARCH_ALIAS = "hermes_tool_search"
|
||||
|
||||
# Persistence-only / cross-transport message keys that strict OpenAI-compatible
|
||||
@@ -70,7 +78,19 @@ def _add_prompt_cache_key(
|
||||
survives compression rotation. A caller-supplied key is authoritative but is
|
||||
bounded to OpenAI's 64-char cap in place. Shares the Responses transport's hash
|
||||
so equivalent prefixes hit one bucket across modes.
|
||||
|
||||
``cache_scope_id``, when provided, is the rotation-stable logical scope (compression-lineage root —
|
||||
agent/prompt_cache_scope.py) and takes precedence over the physical ``session_id`` so the key survives
|
||||
context-compression session rotation (#79017).
|
||||
"""
|
||||
# Stable prompt-cache routing for the Codex/Responses aux path, mirroring the main transport
|
||||
# (agent/transports/codex.py::build_kwargs, which sets prompt_cache_key =
|
||||
# _content_cache_key(instructions, tools)). Without this, MoA acting-aggregator and other auxiliary
|
||||
# Responses calls stay cache-cold while the main Responses transport is warm (issue #53735). The key is
|
||||
# content-addressed from the static prefix (instructions + tool schemas) so it stays warm across
|
||||
# turns/fires. Guard the top-level field the same way the main transport does: xAI Responses takes the
|
||||
# key in extra_body (not top-level) and GitHub/Copilot Responses opts out of cache-key routing entirely
|
||||
# — for those hosts, skip it here.
|
||||
from agent.transports.codex import (
|
||||
_bound_prompt_cache_key_field, _cache_scope_from_session_id, _content_cache_key
|
||||
)
|
||||
@@ -90,7 +110,14 @@ def _add_prompt_cache_key(
|
||||
|
||||
|
||||
def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None:
|
||||
"""Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary."""
|
||||
"""Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary.
|
||||
|
||||
Hermes' internal effort set extends the wire vocabulary with ``ultra`` (the /reasoning command documents
|
||||
none..xhigh|max|ultra). OpenAI- compatible wires — OpenRouter chief among them — accept exactly
|
||||
max|xhigh|high|medium|low|minimal|none and reject the extension with HTTP 400 (#89503). Clamp against
|
||||
the declared wire vocabulary via the shared policy in ``agent.reasoning_effort``; provider profiles with
|
||||
narrower sets clamp again downstream.
|
||||
"""
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return reasoning_config
|
||||
effort = str(reasoning_config.get("effort") or "").strip().lower()
|
||||
@@ -104,6 +131,10 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) ->
|
||||
return None
|
||||
normalized_model = (model or "").strip().lower().removeprefix("google/")
|
||||
# Gemini-only; Gemma/PaLM on the same provider 400 on the field even as ``{"includeThoughts": False}``.
|
||||
# ``thinking_config`` is a Gemini-only request parameter. The same ``gemini`` provider also serves Gemma
|
||||
# (and historically PaLM/Bard); those reject the field with HTTP 400 "Unknown name 'thinking_config':
|
||||
# Cannot find field" — including the polite ``{"includeThoughts": False}`` form. Omit the field entirely
|
||||
# on non-Gemini models. (#17426)
|
||||
if not normalized_model.startswith("gemini"):
|
||||
return None
|
||||
effort = str(reasoning_config.get("effort", "medium") or "medium").strip().lower()
|
||||
|
||||
@@ -12,6 +12,8 @@ from typing import Any, Callable, Optional
|
||||
|
||||
from agent.reasoning_effort import (
|
||||
ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
|
||||
# Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort):
|
||||
# per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365).
|
||||
codex_supported_efforts,
|
||||
)
|
||||
from agent.transports.base import ProviderTransport
|
||||
@@ -21,6 +23,7 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
# Cron fires use ``cron_<job_id>_<YYYYMMDD_HHMMSS>``; the per-fire timestamp is
|
||||
# stripped so repeat fires of one job share a cache scope.
|
||||
# See #51395, #52295.
|
||||
_CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$")
|
||||
|
||||
|
||||
@@ -64,6 +67,10 @@ _XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search"
|
||||
# OpenCode /v1/responses rejects client tools using these names (HTTP 400
|
||||
# "custom function name 'X' is reserved"); xAI reserves ``tool_search`` for
|
||||
# Grok's native Tool Search. Aliased as hermes_<name>.
|
||||
# OpenCode's /v1/responses endpoints (Zen and Go, including custom providers pointing at opencode.ai)
|
||||
# reserve certain function names server-side and reject client tools that use them with HTTP 400 ("custom
|
||||
# function name 'X' is reserved"). Same treatment as the xAI web_search collision: rename on the wire
|
||||
# (hermes_<name>), map back in normalize_response so Hermes dispatch is unaffected. See #85589.
|
||||
_OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files")
|
||||
_XAI_RESERVED_TOOL_NAMES = ("tool_search",)
|
||||
_RESERVED_TOOL_ALIAS_PREFIX = "hermes_"
|
||||
@@ -126,6 +133,11 @@ def _xai_prefers_native_web_search() -> bool:
|
||||
"""True when xAI Responses should use Grok's native ``web_search`` built-in.
|
||||
|
||||
Web-search registry first, then the legacy ``_get_search_backend`` probe; fails closed to native (True).
|
||||
|
||||
Delegates to the web-search registry's provider resolution (which reads ``web.search_backend`` /
|
||||
``web.backend`` from config) and checks whether the resolved provider is xAI. On any resolution failure,
|
||||
returns True (fail-closed to native — preserves the #48108 incomplete-hang fix rather than risk
|
||||
reintroducing it).
|
||||
"""
|
||||
try:
|
||||
from agent.web_search_registry import get_active_search_provider
|
||||
@@ -160,9 +172,25 @@ def _alias_wire_tools(response_tools: Any, params: dict[str, Any], is_xai_respon
|
||||
{**t, "name": _XAI_CLIENT_WEB_SEARCH_ALIAS} if is_client_web_search(t) else t for t in response_tools
|
||||
]
|
||||
wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search"
|
||||
# OpenCode Responses backends reserve web_search / search_files as function names (HTTP 400 "custom
|
||||
# function name 'X' is reserved", #85589). Alias them on the wire; normalize_response maps them back.
|
||||
if response_tools and _is_opencode_responses_backend(params):
|
||||
response_tools, _oc_aliases = _alias_reserved_tools(response_tools, _OPENCODE_RESERVED_TOOL_NAMES)
|
||||
wire_aliases.update(_oc_aliases)
|
||||
# xAI server-side web search vs Hermes web providers. grok models on xAI's /v1/responses surface have a
|
||||
# *native*, server-executed web search. A client-side function literally named ``web_search`` collides
|
||||
# with that engine: declared as a plain ``function`` rather than ``{"type": "web_search"}``, the search
|
||||
# dispatches but never reconciles → incomplete turn + 3 retries. Verified live against
|
||||
# grok-composer-2.5-fast (2026-06); see #48108. Two modes, chosen by the user's web-search backend
|
||||
# config: 1. **Native** (active/configured backend is ``xai``, or resolution fails): drop the client
|
||||
# ``web_search`` function and declare xAI's built-in instead. 1:1 swap only when client ``web_search``
|
||||
# was already present — never an additive grant. 2. **Client** (Firecrawl / Tavily / Exa / … configured
|
||||
# or resolved): keep Hermes dispatch so ``web.backend`` / ``web.search_backend`` is honored, but rename
|
||||
# the wire tool to ``hermes_web_search`` so Grok cannot hijack the name. The alias is mapped back to
|
||||
# ``web_search`` in ``normalize_response``. Request-local alias provenance: every wire alias THIS
|
||||
# request emits is recorded here and stashed on the transport, so the reverse rewrite in
|
||||
# ``normalize_response`` applies only to aliases that were actually sent (never to a real tool that
|
||||
# merely shares an alias-shaped name).
|
||||
if is_xai_responses and response_tools:
|
||||
response_tools, _xai_aliases = _alias_reserved_tools(response_tools, _XAI_RESERVED_TOOL_NAMES)
|
||||
wire_aliases.update(_xai_aliases)
|
||||
@@ -183,6 +211,10 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
|
||||
elif reasoning_config.get("effort"):
|
||||
reasoning_effort = reasoning_config["effort"]
|
||||
|
||||
# Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker
|
||||
# supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that
|
||||
# repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a
|
||||
# model's real ceiling (#87279).
|
||||
if params.get("is_xai_responses", False):
|
||||
from agent.model_metadata import is_grok_46_family
|
||||
|
||||
@@ -215,6 +247,11 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op
|
||||
|
||||
hostname = base_url_hostname(str(base_url or "")).lower()
|
||||
# Meta Model API: caching is opt-in via prompt_cache_retention (0% hits without).
|
||||
# Meta Model API (api.meta.ai) only achieves prompt-cache hits on the Responses API with
|
||||
# prompt_cache_retention; chat/completions stays cache-cold (0% vs 93-99% measured). Exact-hostname
|
||||
# match per #32243.
|
||||
# Meta Model API: prompt caching only on Responses API (0% on chat/completions vs 93-99% on /responses
|
||||
# with retention). See #32243.
|
||||
if hostname == "api.meta.ai":
|
||||
return "24h"
|
||||
parts = hostname.split(".")
|
||||
@@ -229,6 +266,12 @@ def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]],
|
||||
"""``pck_<sha256[:24]>`` of (scope_id, instructions, name-sorted tools), or None if nothing static.
|
||||
|
||||
Routing hint only; ``scope_id`` keeps unrelated sessions off one bucket.
|
||||
|
||||
``scope_id`` (pass ``_cache_scope_from_session_id(session_id)``) keeps unrelated sessions — independent
|
||||
conversations, main vs. child/subagent, sibling children — from concentrating onto the same bucket
|
||||
merely because their static prefix matches (see #78941), while still letting recurring cron fires of one
|
||||
job share a stable key across their timestamped session_ids (the original #51395/#52295 fix this built
|
||||
on). Sorting tools by name keeps the hash insertion-order independent.
|
||||
"""
|
||||
if not instructions and not tools:
|
||||
return None
|
||||
@@ -421,6 +464,20 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
cache key / xAI conv header), max_tokens, timeout, request_overrides, provider, base_url,
|
||||
is_github_responses, is_codex_backend, is_xai_responses, github_reasoning_extra,
|
||||
context_management, replay_encrypted_reasoning.
|
||||
|
||||
params: instructions: str — system prompt (extracted from messages[0] if not given)
|
||||
reasoning_config: dict | None — {effort, enabled} session_id: str | None — transcript/session id;
|
||||
drives the Codex ``session_id`` header, and is the cache-scope fallback when no ``cache_scope_id``
|
||||
is given cache_scope_id: str | None — rotation-stable logical scope id (compression-lineage root;
|
||||
see agent/prompt_cache_scope.py). Preferred over session_id when deriving the prompt_cache_key
|
||||
content hash and the xAI x-grok-conv-id header; the Codex x-client-request-id header mirrors the
|
||||
resulting body key. Keeps the cache warm across context-compression session rotation (#79017)
|
||||
max_tokens: int | None — max_output_tokens timeout: float | None — per-request timeout forwarded to
|
||||
the SDK request_overrides: dict | None — extra kwargs merged in provider: str | None — provider name
|
||||
for backend-specific logic base_url: str | None — endpoint URL base_url_hostname: str | None —
|
||||
hostname for backend detection is_github_responses: bool — Copilot/GitHub models backend
|
||||
is_codex_backend: bool — chatgpt.com/backend-api/codex is_xai_responses: bool — xAI/Grok backend
|
||||
github_reasoning_extra: dict | None — Copilot reasoning params
|
||||
"""
|
||||
from run_agent import DEFAULT_AGENT_IDENTITY
|
||||
|
||||
@@ -490,6 +547,8 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
_bound_prompt_cache_key_field(kwargs)
|
||||
|
||||
# Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing.
|
||||
# Grok 4.6 accepts Priority Processing, but continue stripping stale or unsupported tier values on
|
||||
# every other xAI path. See #28490 and #84799.
|
||||
if is_xai_responses:
|
||||
from agent.model_metadata import is_grok_46_family
|
||||
|
||||
@@ -521,6 +580,13 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
_merge_extra_headers(kwargs, **{"x-grok-conv-id": _cache_scope})
|
||||
# xAI reads prompt_cache_key from the body; extra_body survives SDK builds whose
|
||||
# Responses.stream() dropped the typed kwarg. An explicit request_overrides value wins.
|
||||
# Scoped like the body cache key below — otherwise cron's per-fire timestamp in session_id
|
||||
# (cron_<id>_<ts>) pins every fire of the same job to a different xAI backend server (#78941).
|
||||
# xAI Responses cache-routing — body-level field per
|
||||
# https://docs.x.ai/developers/advanced-api-usage/prompt-caching/maximizing-cache-hits. A
|
||||
# caller's request_overrides={"prompt_cache_key": ...} lands on the top-level kwarg set above —
|
||||
# read it back here so an explicit override actually governs the field xAI reads, instead of
|
||||
# being silently outrun by the auto-derived cache_key (#78941).
|
||||
existing_extra_body = kwargs.get("extra_body")
|
||||
kwargs["extra_body"] = dict(existing_extra_body) if isinstance(existing_extra_body, dict) else {}
|
||||
kwargs["extra_body"].setdefault("prompt_cache_key", kwargs.get("prompt_cache_key", cache_key))
|
||||
|
||||
@@ -48,6 +48,14 @@ class CodexAppServerClient:
|
||||
) -> None:
|
||||
self._codex_bin = codex_bin
|
||||
# codex needs LLM provider creds but must not receive Tier-1 Hermes secrets (gateway/GitHub/infra tokens).
|
||||
# codex app-server is a model-driving CLI executor: it runs a model-chosen agentic loop that
|
||||
# executes shell commands, so it legitimately needs LLM provider credentials
|
||||
# (inherit_credentials=True) to authenticate against the model endpoint. But the previous
|
||||
# `os.environ.copy()` also handed it every Tier-1 Hermes secret — gateway bot tokens, GitHub auth,
|
||||
# Modal/Daytona infra tokens, the dashboard session token, AUXILIARY_* side-LLM keys,
|
||||
# GATEWAY_RELAY_* auth — none of which a coding subprocess has any use for. Route through the
|
||||
# centralized helper so Tier-1 + dynamic-internal secrets are always stripped while provider creds
|
||||
# still flow, matching copilot_acp_client (#29157 sibling spawn-site gap).
|
||||
spawn_env = hermes_subprocess_env(inherit_credentials=True)
|
||||
if env:
|
||||
spawn_env.update(env)
|
||||
@@ -71,6 +79,7 @@ class CodexAppServerClient:
|
||||
|
||||
# Hide the console the codex child would otherwise flash on Windows (#56747).
|
||||
# Hide-only — stdio pipes stay intact for the app-server wire.
|
||||
# See #56747.
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
|
||||
self._proc = subprocess.Popen(
|
||||
|
||||
@@ -328,6 +328,11 @@ class CodexAppServerSession:
|
||||
"""Send a user message and block until turn/completed, bridging approvals and projecting items.
|
||||
|
||||
post_tool_quiet_timeout: silence this long after a tool completes fast-fails and retires.
|
||||
|
||||
post_tool_quiet_timeout: if codex emits a tool completion and then goes quiet for this many seconds
|
||||
without emitting another item or `turn/completed`, fast-fail and mark the session for retirement.
|
||||
Mirrors openclaw beta.8's post-tool completion watchdog (#81697) so a wedged codex doesn't burn the
|
||||
full turn deadline.
|
||||
"""
|
||||
result = TurnResult()
|
||||
if self._start_for(result):
|
||||
|
||||
@@ -86,7 +86,10 @@ class TTSProvider(CatalogProviderBase):
|
||||
"""Whether output suits voice-bubble delivery (mirrors
|
||||
``tts.providers.<name>.voice_compatible``): True → the gateway converts
|
||||
to Opus via ffmpeg if needed; False → regular audio attachment. Default
|
||||
False (opt in)."""
|
||||
False (opt in).
|
||||
|
||||
See #17843.
|
||||
"""
|
||||
return False
|
||||
|
||||
|
||||
|
||||
@@ -122,6 +122,12 @@ def build_api_request(
|
||||
api_kwargs = agent._build_api_kwargs(api_messages, tools_for_api=tools_for_api)
|
||||
# Surrogate chokepoint: tool descriptions, extra_body and kwargs strings can carry
|
||||
# invalid code points (HTTP 400). One walk makes the payload json.dumps()-safe.
|
||||
# Outbound-request surrogate chokepoint (#50959): the messages were scrubbed above, but the rest of the
|
||||
# request body — tool/function descriptions (session_search's ±-heavy text is the recorded repro),
|
||||
# extra_body, system strings routed via kwargs — can still carry invalid code points that providers
|
||||
# reject with a non-retryable HTTP 400 ("invalid unicode code point"). One in-place walk here guarantees
|
||||
# the entire payload json.dumps()-safe regardless of which leaf produced the string. Fast no-op when the
|
||||
# payload is clean.
|
||||
_sanitize_structure_surrogates(api_kwargs)
|
||||
if agent._force_ascii_payload:
|
||||
_sanitize_structure_non_ascii(api_kwargs)
|
||||
|
||||
+45
-3
@@ -172,6 +172,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None:
|
||||
main_runtime = {
|
||||
k: getattr(agent, k, None) for k in ("model", "provider", "base_url", "api_key", "api_mode")
|
||||
}
|
||||
# See #19027.
|
||||
maybe_auto_title(
|
||||
session_db,
|
||||
session_id,
|
||||
@@ -197,7 +198,17 @@ def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> in
|
||||
|
||||
Prefers the LAST user message whose content exactly matches this turn's text, else
|
||||
the last user-originated turn; compaction handoffs are never the fallback.
|
||||
Returns -1 when there is no user-originated message."""
|
||||
Returns -1 when there is no user-originated message.
|
||||
|
||||
Compression replaces list entries with fresh copies (and may append a todo-snapshot user message or a
|
||||
restored user turn AFTER the surviving copy of the current turn's message), so a pre-compression index
|
||||
is meaningless. Prefer the LAST user message whose content exactly matches this turn's text — the
|
||||
surviving copy in the common case — so the injection stamp and the #48677 persist override can't land on
|
||||
a todo-snapshot or historical row. Fall back to the last *user-originated* turn when no exact match
|
||||
survives (merge-summary-into-tail rewrites the content but the trackers still need a live anchor).
|
||||
Compaction handoffs must never become the fallback anchor (#80622) — they are reference-only
|
||||
scaffolding, not the active ask.
|
||||
"""
|
||||
from agent.context_compressor import user_originated_turn_view
|
||||
|
||||
fallback = -1
|
||||
@@ -225,11 +236,24 @@ def compression_made_progress(
|
||||
orig_len: int, new_len: int, orig_tokens: int, new_tokens: int
|
||||
) -> bool:
|
||||
"""``True`` if a compression pass materially reduced the request: fewer rows, or a
|
||||
>5% token cut with the same rows (same floor as the overflow-handler retry)."""
|
||||
>5% token cut with the same rows (same floor as the overflow-handler retry).
|
||||
|
||||
Compression can succeed by summarising message contents — reducing the estimated request token count —
|
||||
without reducing the message row count. Treating row count as the sole progress signal false-positives
|
||||
on size-only wins and surfaces a misleading "Cannot compress further" failure even when post-compression
|
||||
tokens are well below the model context window. See issue #39548 for an observed case: 220 → 220
|
||||
messages, ~288k → ~183k tokens on a 1M-context model still triggered auto-reset.
|
||||
The token reduction must be *material* (>5%) to count as progress — the same floor the overflow-handler
|
||||
retry path uses (conversation_loop.py, 39550) — so a sub-5% wobble doesn't keep the multi-pass loop
|
||||
spinning. See #39550.
|
||||
"""
|
||||
return new_len < orig_len or (orig_tokens > 0 and new_tokens < orig_tokens * 0.95)
|
||||
|
||||
|
||||
# Back-compat alias: gateway callers and tests patch ``_compression_made_progress``.
|
||||
# Back-compat alias: this predicate was module-private until the gateway's session-hygiene recovery gate
|
||||
# needed the same semantics (#79624). Keeping the old name bound means existing callers and any test that
|
||||
# patches ``_compression_made_progress`` continue to work unchanged.
|
||||
_compression_made_progress = compression_made_progress
|
||||
|
||||
|
||||
@@ -296,7 +320,12 @@ def _should_idle_compact(
|
||||
produced (``ContextCompressor.last_compression_rough_tokens``, same rough shape as
|
||||
``tokens``) — raises the floor to ``last + floor_tokens`` so the transcript must gain a
|
||||
floor's worth of NEW content first. ``0`` (nothing compacted yet / counter reset) keeps
|
||||
the original semantics exactly."""
|
||||
the original semantics exactly.
|
||||
|
||||
A session that compacted to well above that target therefore stays above it forever, so every later idle
|
||||
resume re-runs a full summarisation over a transcript that has not grown — minutes of silently blocked
|
||||
prompt on a slow route, reclaiming nothing (#97239).
|
||||
"""
|
||||
if not enabled or idle_after_seconds <= 0 or idle_gap_seconds < idle_after_seconds or cooldown_active:
|
||||
return False
|
||||
effective_floor = floor_tokens
|
||||
@@ -350,6 +379,12 @@ def _publish_runtime_main(agent: Any) -> None:
|
||||
from agent.prompt_cache_scope import resolve_prompt_cache_scope_safe
|
||||
# Rotation-stable prompt-cache scope (lineage root), memoized per segment; a new
|
||||
# session uses the physical id until build_api_kwargs re-resolves.
|
||||
# Memoized per segment on the agent, so this is a DB walk at most once per segment — except a
|
||||
# brand-new session whose row lands later in turn setup (_ensure_db_session); that first turn falls
|
||||
# back to the physical id here and the first build_api_kwargs re-resolves. Stays valid through a
|
||||
# mid-turn compression rotation because the lineage root is by definition rotation-invariant
|
||||
# (#79017). Resolved with the never-raising variant OUTSIDE the argument list, so a resolution
|
||||
# failure can only lose the scope — never the whole runtime binding.
|
||||
_cache_scope = resolve_prompt_cache_scope_safe(agent) or ""
|
||||
set_runtime_main(
|
||||
_str_attr(agent, "provider"), _str_attr(agent, "model"),
|
||||
@@ -569,6 +604,8 @@ def _collect_pre_llm_call_context(
|
||||
sender_id=getattr(agent, "_user_id", None) or "",
|
||||
)
|
||||
try:
|
||||
# Spill oversized per-hook context to disk so a runaway plugin can't inflate every subsequent
|
||||
# turn's prompt. Ported from openai/codex PR #21069 ("Spill large hook outputs from context").
|
||||
from tools.hook_output_spill import (
|
||||
get_spill_config as _spill_cfg, spill_if_oversized as _spill_if_oversized
|
||||
)
|
||||
@@ -738,6 +775,11 @@ def build_turn_context(
|
||||
|
||||
# Tag log records on this thread with the session ID for ``hermes logs``; bind the
|
||||
# skill write-origin ContextVar; restore the primary runtime after a fallback turn.
|
||||
# NOTE: the DB session row is created later, AFTER the system prompt is restored/built (see
|
||||
# _ensure_db_session() below the system-prompt block). Creating it here — before _cached_system_prompt
|
||||
# is populated — inserts a row with system_prompt=NULL on a fresh API/gateway agent that carries
|
||||
# client-managed history, which then trips the "stored system prompt is null; rebuilding from scratch"
|
||||
# warning and a needless first-turn prefix cache miss. (Issue #45499.)
|
||||
set_session_context(agent.session_id)
|
||||
set_current_write_origin(getattr(agent, "_memory_write_origin", "assistant_tool"))
|
||||
agent._restore_primary_runtime()
|
||||
|
||||
@@ -42,6 +42,14 @@ class CompactionOutcome:
|
||||
|
||||
def _clear_overflow_warn(agent: Any) -> None:
|
||||
"""Re-arm the context-overflow warning dedup (test doubles may lack the method)."""
|
||||
# Compression is actually running (block cleared / was never blocked) — reset the blocked-overflow
|
||||
# warning dedup so a future blocked-over-threshold turn can warn again. Mirrors the turn-context
|
||||
# preflight reset (silent-overflow fix #62625). getattr guard: test doubles built via object.__new__
|
||||
# lack the method (gateway test-double pitfall) — treat absence as no-op.
|
||||
# Compression is actually running (block cleared / was never blocked) — reset the blocked-overflow
|
||||
# warning dedup so a future blocked-over-threshold turn can warn again (silent-overflow fix #62625).
|
||||
# getattr guard: test doubles built via object.__new__ lack the method (gateway test-double pitfall) —
|
||||
# treat absence as no-op.
|
||||
_clear_warn = getattr(agent, "_clear_context_overflow_warn", None)
|
||||
if callable(_clear_warn):
|
||||
_clear_warn()
|
||||
@@ -94,6 +102,10 @@ def _apply_grown_window(agent: Any, compressor: Any, grown: int) -> None:
|
||||
|
||||
def _refund_api_call(agent: Any, api_call_count: int) -> int:
|
||||
"""A pass that never reached the provider refunds the call count and budget."""
|
||||
# Host progress-aware timeout (#98722, salvaged from #98741): this preflight iteration never reached the
|
||||
# provider. Refund its provisional call/budget exactly like a successful pre-API compaction, then stop
|
||||
# before the unchanged oversized request reaches the provider — its overflow error would only invoke
|
||||
# compression again on the same transcript with the wait budget already spent.
|
||||
api_call_count -= 1
|
||||
agent._api_call_count = api_call_count
|
||||
agent.iteration_budget.refund()
|
||||
@@ -200,6 +212,7 @@ def _codex_native_auto_compaction(agent: Any) -> bool:
|
||||
"""Codex app-server threads are compacted by the codex agent itself; Hermes only
|
||||
initiates compaction in "hermes" mode."""
|
||||
return (
|
||||
# See #36801.
|
||||
getattr(agent, "api_mode", None) == "codex_app_server"
|
||||
and str(
|
||||
getattr(agent, "codex_app_server_auto_compaction", "native") or "native"
|
||||
@@ -367,6 +380,10 @@ def _run_preflight_passes(
|
||||
# Lock-skip: another path holds the lock, so this is a DEFER, not proof of
|
||||
# incompressibility — don't arm the blocker; stop passes this turn.
|
||||
logger.info(
|
||||
# That is a temporary DEFER, not proof the transcript cannot compress — do NOT arm the
|
||||
# insufficient-progress blocker (the loop's error handlers must keep their provider-proven
|
||||
# retry budget) and stop preflight passes for this turn; the lock winner is shrinking the
|
||||
# same session concurrently. See #69870.
|
||||
"Preflight compression deferred: compression lock "
|
||||
"held by another path (session %s)",
|
||||
agent.session_id or "none",
|
||||
|
||||
@@ -31,6 +31,8 @@ class TurnFacadeMixin:
|
||||
"""Forwarder — see ``agent.conversation_loop.run_conversation``."""
|
||||
# A review shares this session_id for cache parity: fence review startup or interrupt
|
||||
# an admitted request and await its exit before opening live-turn instrumentation.
|
||||
# Foreground priority is retained if the review does not acknowledge within the bounded deadline
|
||||
# (#84423).
|
||||
from agent.background_review import cancel_background_review_for_live_turn
|
||||
|
||||
cancel_background_review_for_live_turn(self)
|
||||
@@ -106,6 +108,9 @@ class TurnFacadeMixin:
|
||||
# affinity scope falls back to it; accounting handles route aux usage to the session.
|
||||
token = set_conversation_context(self._conversation_root_id())
|
||||
affinity_token = set_affinity_scope(declared_conversation_scope_safe(self))
|
||||
# Publish the session accounting handles the same way so auxiliary calls record their token
|
||||
# usage into session_model_usage (task dimension) — the fix for aux spend being invisible in
|
||||
# analytics (issue #23270).
|
||||
acct_token = set_accounting_context(
|
||||
getattr(self, "_session_db", None), getattr(self, "session_id", None)
|
||||
)
|
||||
|
||||
@@ -220,6 +220,9 @@ def _durable_session_exists(db, session_id: str) -> bool:
|
||||
# A locked / non-WAL read is not proof the row is absent; treating probe failure as "fresh"
|
||||
# ran fail-open at the exact contention point. Acquire, or fail closed.
|
||||
logger.warning(
|
||||
# Acquire (or fail closed if acquire itself cannot) rather than start load/run/flush
|
||||
# unsynchronized. get_session returns None — it does not raise — when the row is missing. See
|
||||
# #84234.
|
||||
"Could not check durable session before turn lease; "
|
||||
"will acquire rather than run without serialization",
|
||||
exc_info=True,
|
||||
|
||||
@@ -103,6 +103,16 @@ def finish_text_response(
|
||||
agent._emit_pending_fallback_notice()
|
||||
agent._clear_status_buffer()
|
||||
|
||||
# Defensive: repair malformed role-alternation before API call. Catches cases where the history got
|
||||
# wedged into a ``tool → user`` or ``user → user`` tail (e.g. after empty- response scaffolding was
|
||||
# stripped and a new user message landed after an orphan tool result). Most providers return empty
|
||||
# content on malformed sequences, which would otherwise retrigger the empty-retry loop indefinitely.
|
||||
# repair_message_sequence_with_cursor also recomputes the SessionDB flush cursor (_last_flushed_db_idx)
|
||||
# when repair compacts the list, so the turn-end flush doesn't skip the assistant/tool chain (#44837).
|
||||
# One-time repeated-heal escalation notice (#96870): if the sanitizer above just crossed the per-session
|
||||
# heal threshold, deliver the queued notice through the status/warning callback — the normal out-of-band
|
||||
# delivery channel (gateway status message / CLI print). NEVER appended to messages/api_messages:
|
||||
# conversation context and the cached prompt prefix stay byte-identical.
|
||||
from agent.agent_runtime_helpers import (
|
||||
intent_ack_continuation_mode, trailing_continue_intent
|
||||
)
|
||||
|
||||
+42
-1
@@ -46,7 +46,12 @@ def _record_kanban_budget_exhausted(
|
||||
|
||||
Routed via ``_record_task_failure`` (not ``kanban_block``) so it counts toward the
|
||||
consecutive-failure circuit breaker. Idempotent via the ``_end_run`` CAS
|
||||
(``WHERE ended_at IS NULL``), so safe from multiple exit paths."""
|
||||
(``WHERE ended_at IS NULL``), so safe from multiple exit paths.
|
||||
|
||||
This is a bounded fallback (#87096): the CAS invariant in ``_end_run`` (``WHERE ended_at IS NULL``)
|
||||
guarantees idempotence — if another path already closed the run this is a no-op — so it is safe to call
|
||||
from multiple exit paths.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli import kanban_db as _kb
|
||||
_conn = _kb.connect()
|
||||
@@ -149,6 +154,17 @@ def _resolve_budget_fallback(
|
||||
# A kanban worker must record a terminal outcome whether or not a fallback path
|
||||
# was eligible, so the dispatcher learns the worker could not complete.
|
||||
_kanban_task = os.environ.get("HERMES_KANBAN_TASK") if budget_exhausted else None
|
||||
# If running as a kanban worker, signal the dispatcher that the worker could not complete (rather than
|
||||
# treating it as a protocol violation). This applies whether the user-facing fallback came from the
|
||||
# summary call or an explicitly pending continuation; both exhausted the task budget and must advance
|
||||
# the failure circuit. We route through ``_record_task_failure(outcome="timed_out")`` rather than
|
||||
# ``kanban_block`` so this counts toward the dispatcher's consecutive-failure circuit breaker (#29747
|
||||
# gap 2).
|
||||
# Bounded fallback (#87096): budget was exhausted but none of the normal fallback paths were eligible
|
||||
# (interrupted / failed / anomalous exit_reason). If running as a kanban worker we must still record a
|
||||
# terminal outcome so the task does not remain in an ambiguous lifecycle state. The worker's run is
|
||||
# closed via ``_record_task_failure`` (compare-and-swap receipt path) which is a no-op if another path
|
||||
# closed it — the CAS invariant in ``_end_run`` (``WHERE ended_at IS NULL``) guarantees idempotence.
|
||||
if _kanban_task:
|
||||
_record_kanban_budget_exhausted(_kanban_task, api_call_count, agent.max_iterations, logger)
|
||||
return final_response, _turn_exit_reason, preserved_verification_fallback
|
||||
@@ -202,6 +218,16 @@ def _close_transcript_tail(agent, messages, final_response, interrupted, failed)
|
||||
# row; enforce "delivered final_response ⇒ assistant row" here. Compare content,
|
||||
# not role, so a matching verification candidate isn't dup'd.
|
||||
if final_response and not interrupted:
|
||||
# Some recovery/fallback paths return a real final_response without adding a closing assistant
|
||||
# message to the transcript (e.g. the partial-stream and prior-turn-content recovery ``break`` sites
|
||||
# in ``conversation_loop``). If persisted as-is, the durable session can end at a tool/user message
|
||||
# even though the caller — and the gateway platform — already saw a completed assistant response.
|
||||
# The next turn then replays a user-only backlog and the model re-answers every "unanswered"
|
||||
# message. Close the durable turn at the source, at the single chokepoint every recovery ``break``
|
||||
# flows through, so the invariant "delivered final_response ⇒ assistant row in transcript" holds
|
||||
# regardless of which path produced it. (#43849 / #44100) Compare content (not just role) so a
|
||||
# verification candidate that matches the final response is not duplicated at budget exhaustion.
|
||||
# (#65919 §7)
|
||||
_tail = messages[-1] if messages else None
|
||||
if not isinstance(_tail, dict) or _tail.get("role") != "assistant":
|
||||
append_message(messages, {"role": "assistant", "content": final_response})
|
||||
@@ -219,6 +245,8 @@ def _close_transcript_tail(agent, messages, final_response, interrupted, failed)
|
||||
|
||||
# Request is complete, so replace API-local voice/model/skill guidance with the
|
||||
# clean user input before the durable snapshot (earlier flushes still needed them).
|
||||
# Earlier turn-start flushes use the DB-only override because their messages are still needed for the
|
||||
# API request; this finalizer runs after that request is complete (#48677 / #63766).
|
||||
_apply_override = getattr(agent, "_apply_persist_user_message_override", None)
|
||||
if callable(_apply_override):
|
||||
_apply_override(messages)
|
||||
@@ -302,6 +330,13 @@ def _append_file_mutation_footer(agent, final_response, logger):
|
||||
"""Append the verifier advisory when ``write_file`` / ``patch`` calls failed and were
|
||||
never superseded by a successful write to the same path (surfaces over-claiming)."""
|
||||
try:
|
||||
# File-mutation verifier footer. This catches the specific case — reported by Ben Eng
|
||||
# (#15524-adjacent) — where a model issues a batch of parallel patches, half of them fail with
|
||||
# "Could not find old_string", and the model summarises the turn claiming every file was edited. The
|
||||
# user then has to manually run ``git status`` to catch the lie. With this footer the truth is
|
||||
# surfaced on every turn, so over-claiming is structurally impossible past the model. Gate: only
|
||||
# applied when a real text response exists for this turn and the user didn't interrupt.
|
||||
# Empty/interrupted turns already have other surface text that shouldn't be augmented.
|
||||
_failed = getattr(agent, "_turn_failed_file_mutations", None) or {}
|
||||
if _failed and agent._file_mutation_verifier_enabled():
|
||||
footer = agent._format_file_mutation_failure_footer(_failed)
|
||||
@@ -473,6 +508,12 @@ def finalize_turn(
|
||||
|
||||
# Surrogate chokepoint: RAW SDK text with a lone UTF-16 surrogate crashes downstream
|
||||
# consumers (stdout, Telegram ``utf16_len``, JSON); scrub once where it leaves the loop.
|
||||
# Class-level surrogate chokepoint (#80366, #55143, #55309, #19819): ``final_response`` is often the RAW
|
||||
# SDK content (``assistant_message.content``), not the sanitized copy stored in history by
|
||||
# ``build_assistant_message``. Any lone UTF-16 surrogate (U+D800–U+DFFF) in it crashes downstream
|
||||
# consumers — oneshot stdout writes, Telegram's ``utf16_len`` length check, Signal formatting, JSON
|
||||
# envelope encodes — on every provider (Ollama, NVIDIA NIM, …). Scrub once here, where model text leaves
|
||||
# the conversation loop, so every delivery surface receives valid Unicode.
|
||||
if isinstance(final_response, str):
|
||||
final_response = _sanitize_surrogates(final_response)
|
||||
|
||||
|
||||
@@ -273,6 +273,11 @@ def begin_iteration(
|
||||
# Grace call: budget exhausted but the model gets one more call. Consume the
|
||||
# flag so the loop exits after this iteration regardless of outcome.
|
||||
if agent._budget_grace_call:
|
||||
# Iteration budget: the LLM is only notified when it actually exhausts the iteration budget
|
||||
# (api_call_count >= max_iterations). At that point we inject ONE message, allow one final API call,
|
||||
# and if the model doesn't produce a text response, force a user-message asking it to summarise. No
|
||||
# intermediate pressure warnings — they caused models to "give up" prematurely on complex tasks
|
||||
# (#7915).
|
||||
agent._budget_grace_call = False
|
||||
elif not agent.iteration_budget.consume():
|
||||
_turn_exit_reason = "budget_exhausted"
|
||||
@@ -340,6 +345,11 @@ def apply_retry_restarts(
|
||||
retry_count += 1
|
||||
_retry.restart_with_compressed_messages = False
|
||||
if _should_skip_model_call_for_reference_handoff(
|
||||
# Compression rebuilt the list (tail messages are fresh compaction copies), so the
|
||||
# pre-compression index of this turn's user message is stale. Re-anchor both index trackers: the
|
||||
# api_content stamp below, the loop's injection site, and the flush's persist-override row
|
||||
# (#48677) must all target the surviving dict, not a stale position. Exact-content match first
|
||||
# so a todo-snapshot user message appended after the tail can't steal the anchor.
|
||||
messages, user_message
|
||||
):
|
||||
logger.info(
|
||||
|
||||
+12
-1
@@ -127,6 +127,11 @@ class TurnLivenessWatchdog:
|
||||
return None
|
||||
# Observational only: the commit below can still veto the abort if progress
|
||||
# resumed; the definitive settlement is _surface_committed_abort.
|
||||
# Pre-commit surface is OBSERVATIONAL only: it reports the stall and that a recovery attempt is
|
||||
# beginning. It must not claim the abort or the lease withdrawal has committed — the next operation
|
||||
# can still veto the outcome. The definitive aborted/lease-stopped settlement is published by
|
||||
# _surface_committed_abort only after _commit_abort succeeds and the turn is deactivated (#95663
|
||||
# review).
|
||||
self._surface_stall(snapshot)
|
||||
message = f"Turn made no progress for {int(snapshot.idle_seconds)}s; aborting to release the session."
|
||||
if not self._commit_abort(snapshot, message):
|
||||
@@ -178,7 +183,13 @@ class TurnLivenessWatchdog:
|
||||
)
|
||||
|
||||
def _surface_committed_abort(self, snapshot: ActivitySnapshot) -> None:
|
||||
"""Publish the definitive settlement once the abort has authority."""
|
||||
"""Publish the definitive settlement once the abort has authority.
|
||||
|
||||
Runs only once ``_commit_abort`` succeeded (the interrupt was published) and the turn lease was
|
||||
deactivated: the turn IS force-aborted and lease renewal IS stopped, so stating that is now true.
|
||||
Separated from the pre-commit surface so a declined abort never reports a committed outcome (#95663
|
||||
review).
|
||||
"""
|
||||
logger.error(
|
||||
"Turn liveness watchdog aborted turn for session %s: "
|
||||
"no progress for %.1fs; turn interrupted and lease renewal "
|
||||
|
||||
@@ -56,6 +56,17 @@ def handle_outer_loop_error(
|
||||
_outer_error_count += 1
|
||||
|
||||
# Interpreter shutdown makes every executor op raise: break.
|
||||
# Phase-aware error classification. The huge outer try/except spans both the actual API request and all
|
||||
# local post-processing of the returned assistant message. Deterministic local bugs (e.g. passing a
|
||||
# multimodal content list into a regex helper after a vision turn or context compaction) should not be
|
||||
# retried: they will fail identically on every iteration and only burn the iteration budget. We classify
|
||||
# an error as local by inspecting the traceback: if the exception propagated through any of the known
|
||||
# local post-processing helpers and never entered the interruptible API-call helpers, it is almost
|
||||
# certainly a local processing bug. (#66267) Interpreter shutdown: if the process is tearing down, every
|
||||
# executor-backed operation (API call, tool dispatch, memory sync) raises ``RuntimeError: cannot
|
||||
# schedule new futures after interpreter shutdown``. Retrying is pointless — the executor is gone for
|
||||
# good — and each retry just spams another traceback. Break immediately so the turn exits cleanly.
|
||||
# (#93217)
|
||||
if sys.is_finalizing() or _is_interpreter_shutdown_error(e):
|
||||
error_msg = f"Interpreter is shutting down — cannot continue (API call #{api_call_count}): {e}"
|
||||
try:
|
||||
|
||||
@@ -109,6 +109,9 @@ class _Recovery(OverflowVerdict):
|
||||
"failed": True,
|
||||
}
|
||||
if compression_exhausted:
|
||||
# Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The
|
||||
# bloated transcript remains intact while future input can move to a clean session instead of
|
||||
# replaying the summarize-timeout loop.
|
||||
result["compression_exhausted"] = True
|
||||
result.update(extra)
|
||||
return self.done("return", result)
|
||||
@@ -190,6 +193,14 @@ class _Recovery(OverflowVerdict):
|
||||
if deferred is not None:
|
||||
return deferred, False, original_tokens
|
||||
messages = self.messages
|
||||
# Re-measure after compression. Same-message-count compression (tool-result pruning, in-place
|
||||
# summarization) can materially reduce request size without reducing the message array (#39550), and
|
||||
# — the image-dominated case — compaction's historical-media aging (#97160) can free megabytes of
|
||||
# base64 that the token estimate never counted. Bytes are the yardstick for a 413; tokens are kept
|
||||
# only for status display.
|
||||
# Re-estimate tokens after compression. Same-message-count compression (tool-result pruning,
|
||||
# in-place summarization) can materially reduce request size without reducing the message array.
|
||||
# (#39550)
|
||||
new_tokens = estimate_messages_tokens_rough(messages)
|
||||
shrank_tokens = new_tokens > 0 and new_tokens < original_tokens * 0.95
|
||||
if len(messages) < original_len:
|
||||
@@ -223,6 +234,13 @@ def _recover_payload_too_large(st: _Recovery, _retry: TurnRetryState) -> Overflo
|
||||
|
||||
messages = st.messages
|
||||
original_len = len(messages)
|
||||
# A 413 is a BYTE-size error, so this branch scores progress in BYTES of the serialized messages payload
|
||||
# — exact and free — never the token estimate. The estimator prices every image at a flat per-image
|
||||
# token cost (see estimate_messages_tokens_rough) so screenshots don't trigger premature compaction;
|
||||
# that deliberate byte-blindness means compaction can free megabytes of base64 (real case: two vision
|
||||
# results = 96.6% of the request body but ~3.7% of the estimate) while the token delta stays under any
|
||||
# threshold. Token-scored progress here burned all attempts on "no progress" and wedged the session
|
||||
# permanently. (#88960 / #47339)
|
||||
original_bytes = serialized_messages_bytes(messages)
|
||||
deferred = st.compress(st.request_tokens())
|
||||
if deferred is not None:
|
||||
@@ -439,6 +457,9 @@ def recover_from_overflow(
|
||||
# failover or generic retries. The classifier also covers 400/disconnect +
|
||||
# large-session heuristics.
|
||||
st.is_context_length_error = (
|
||||
# Check for context-length errors BEFORE generic 4xx handler. The classifier detects context
|
||||
# overflow from: explicit error messages, generic 400 + large session heuristic (#1630), and server
|
||||
# disconnect + large session pattern (#2153).
|
||||
classified.reason == FailoverReason.context_overflow
|
||||
or wrapped_output_cap_budget is not None
|
||||
)
|
||||
|
||||
@@ -262,9 +262,17 @@ def compress_after_tool_results(
|
||||
)
|
||||
|
||||
_compressor = agent.context_compressor
|
||||
# Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens
|
||||
# is the actual context size the provider reported plus the assistant turn — a tight lower bound for the
|
||||
# next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves
|
||||
# ample headroom; if tool results push past it, the next API call will report the real total and trigger
|
||||
# compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage
|
||||
# data), fall back to rough estimate to avoid missing compression. Without this, a session can grow
|
||||
# unbounded after disconnects because should_compress(0) never fires. (#2153)
|
||||
if _compressor.last_prompt_tokens > 0:
|
||||
# Only prompt_tokens: thinking models inflate completion_tokens with
|
||||
# reasoning that uses no context → premature compression.
|
||||
# Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026)
|
||||
_real_tokens = _compressor.last_prompt_tokens
|
||||
elif _compressor.last_prompt_tokens == -1:
|
||||
# Compression just ran, no API prompt count yet: don't treat a rough
|
||||
@@ -274,6 +282,12 @@ def compress_after_tool_results(
|
||||
# Include tool schemas (20-30K tokens the messages-only estimate misses) and
|
||||
# stay route-aware: on a compacted native-Codex session the generic
|
||||
# durable-history figure would false-trigger.
|
||||
# Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate
|
||||
# misses, which can skip compression past the configured threshold (#14695). Route-aware
|
||||
# (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure
|
||||
# overstates the wire and would false-trigger compression here exactly like the pre-API guard — this
|
||||
# fallback runs precisely when no provider usage is available (post-disconnect / gateway restart),
|
||||
# the unanchored case from #97602's repro.
|
||||
_real_tokens = _midturn_request_pressure_tokens(
|
||||
agent, messages, active_system_prompt or "",
|
||||
estimate_request_tokens_rough(messages, tools=agent.tools or None),
|
||||
@@ -298,6 +312,30 @@ def compress_after_tool_results(
|
||||
if messages is _post_tool_input and compression_skipped_due_to_lock(agent):
|
||||
# Lock-skip no-op is a temporary defer, not evidence about compressibility:
|
||||
# refund so a lock-loser loop doesn't burn the budget toward exhausted.
|
||||
# #69870 lock-skip / #97488 transient-block: this pass no-oped for a TEMPORARY reason (another
|
||||
# path holds the compression lock, or a timed cooldown/backoff guard is active). That is a
|
||||
# temporary DEFER, not evidence about compressibility — refund the attempt (it must not burn the
|
||||
# shared overflow-recovery budget toward compression_exhausted → gateway auto-reset,
|
||||
# #9893/#35809) and leave the insufficient-progress blocker unarmed. Proceed with the current
|
||||
# request: if it truly does not fit, the provider's 413/overflow handler returns the soft
|
||||
# compression_deferred result with that stronger signal.
|
||||
# #69870 lock-skip: the provider proved the request does not fit, but this compression pass
|
||||
# no-oped only because another path holds the session's compression lock. Temporary defer, not
|
||||
# exhaustion — refund the attempt and end the turn softly so the gateway does NOT auto-reset the
|
||||
# session (#9893/#35809).
|
||||
# #97488 transient-block: compression no-oped because a timed guard (host-timeout cooldown /
|
||||
# structural backoff) is active — a temporary defer, not evidence of incompressibility. Never
|
||||
# classify it as compression_exhausted (gateway auto-reset).
|
||||
# bypass_cooldown=True, # #100661 provider-proven overflow
|
||||
# #97488: timed transient guard — defer, never exhaustion (gateway auto-reset).
|
||||
# #69870 lock-skip: the provider proved the request does not fit, but this compression pass
|
||||
# no-oped only because another path holds the session's compression lock. Temporary defer, not
|
||||
# exhaustion — refund the attempt and end the turn softly so the gateway does NOT auto-reset the
|
||||
# session (#9893/#35809).
|
||||
# #97488 transient-block: a timed guard (host-timeout cooldown / structural backoff) no-oped
|
||||
# this pass — defer softly, never compression_exhausted (which would auto-reset the session).
|
||||
# #69870 lock-skip: this pass no-oped because another path holds the session's compression lock
|
||||
# — a temporary defer, not evidence about compressibility.
|
||||
compression_attempts -= 1
|
||||
else:
|
||||
conversation_history = conversation_history_after_compression(
|
||||
|
||||
@@ -145,6 +145,9 @@ def _recover_unicode_encode_error(
|
||||
# Non-ASCII in the API key makes httpx fail encoding the Authorization header — the
|
||||
# usual persistent cause after message/tool sanitization. Entra ID bearer providers
|
||||
# are callables minting ASCII JWTs; skip them (``_strip_non_ascii`` would crash).
|
||||
# Sanitize the API key — non-ASCII characters in credentials (e.g. ʋ instead of v from a bad copy-paste)
|
||||
# cause httpx to fail when encoding the Authorization header as ASCII. This is the most common cause of
|
||||
# persistent UnicodeEncodeError that survives message/tool sanitization (#6843).
|
||||
_credential_sanitized = False
|
||||
_raw_key = getattr(agent, "api_key", None) or ""
|
||||
if _raw_key and isinstance(_raw_key, str):
|
||||
@@ -543,6 +546,8 @@ def recover_after_classification(
|
||||
# Anthropic OAuth subscription rejected the 1M-context beta: disable it for this
|
||||
# session, rebuild the client, retry once. Reactive so capable subscriptions keep 1M.
|
||||
if (
|
||||
# See PR #17680 for the original report (we chose reactive recovery over the proposed unconditional
|
||||
# omit so capable subscriptions don't silently lose the capability).
|
||||
classified.reason == FailoverReason.oauth_long_context_beta_forbidden
|
||||
and agent.api_mode == "anthropic_messages"
|
||||
and agent._is_anthropic_oauth
|
||||
@@ -698,6 +703,8 @@ def nonretryable_client_error_result(
|
||||
logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error)
|
||||
# Skip persistence on likely context-overflow (400 + large session): persisting the
|
||||
# failed message grows the session and repeats the failure.
|
||||
# Persisting the failed user message would make the session even larger, causing the same failure on the
|
||||
# next attempt. (#1630)
|
||||
if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80):
|
||||
_vlines(agent, "⚠️ Skipping session persistence for large failed session to prevent growth loop.")
|
||||
else:
|
||||
@@ -981,6 +988,9 @@ def compute_error_backoff(
|
||||
_ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After")
|
||||
if _ra_raw:
|
||||
try:
|
||||
# Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap
|
||||
# caused us to retry before the actual reset window and re-trip the limit. 600s covers all
|
||||
# realistic provider reset windows while still rejecting pathological values. (#26293)
|
||||
_retry_after = min(float(_ra_raw), 600)
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
@@ -1307,6 +1317,9 @@ def route_classified_error(
|
||||
# Overhead-aware request size so recovery arms on the true request
|
||||
# (msgs + tools + system), not the tool-blind message count.
|
||||
messages, active_system_prompt = agent._compress_context(
|
||||
# Route the overhead-aware _real_tokens (computed above) into compression, not the bare
|
||||
# last_prompt_tokens — which is 0 in the no-usage fallback, hiding the true request size
|
||||
# from the engine's overflow guard (upstream PR #77169 review).
|
||||
messages, system_message,
|
||||
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
|
||||
task_id=effective_task_id,
|
||||
@@ -1331,6 +1344,12 @@ def route_classified_error(
|
||||
is_rate_limited = classified.reason in _RATE_LIMIT_REASONS
|
||||
# Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only the max_tokens
|
||||
# clamp fixes it. Parsed once; gates the eager-fallback exemption and overflow entry.
|
||||
# Relay-wrapped output-cap errors: some gateways wrap an upstream "[400]: max_tokens (...) exceeds
|
||||
# model's maximum output tokens (...)" as HTTP 429, which classifies as rate_limit. The failure is a
|
||||
# deterministic request-shape problem — falling back to another provider (or burning generic retries)
|
||||
# can't fix it, but the output-cap clamp below can, in one retry (#72281). Parse once here; the result
|
||||
# gates both the eager-fallback exemption and the widened is_context_length_error entry, and is reused
|
||||
# as available_out inside the handler.
|
||||
_wrapped_output_cap_budget = (
|
||||
parse_available_output_tokens_from_error(error_msg)
|
||||
if classified.reason == FailoverReason.rate_limit else None
|
||||
@@ -1348,6 +1367,7 @@ def route_classified_error(
|
||||
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
|
||||
# No eager fallback while credential pool rotation may recover. Exception: an
|
||||
# upstream-aggregator 429 — the pool can't help, always fall back.
|
||||
# Fixes #11314.
|
||||
_is_upstream = classified.reason == FailoverReason.upstream_rate_limit
|
||||
pool_may_recover = (
|
||||
False if _is_upstream else _ra()._pool_may_recover_from_rate_limit(agent._credential_pool)
|
||||
|
||||
@@ -234,6 +234,10 @@ def assemble_api_request(
|
||||
approx_tokens = estimate_messages_tokens_rough(api_messages, charge_stale_thinking=False)
|
||||
# Route-aware: native Responses compaction prunes the wire payload, so the raw
|
||||
# history figure overstates it and fires needless local compression.
|
||||
# Route-aware pressure: when the upcoming request is eligible for native Responses compaction the
|
||||
# transport will checkpoint-prune the payload before sending — the generic durable-history figure
|
||||
# overstates the wire by orders of magnitude on a compacted session and fires a 600s local compression
|
||||
# the main request never needed (#96995, mirroring the turn-prologue preflight #96644/#96155).
|
||||
request_pressure_tokens = _midturn_request_pressure_tokens(
|
||||
agent, api_messages, effective_system or "", approx_tokens
|
||||
)
|
||||
|
||||
@@ -205,6 +205,11 @@ def run_tool_round(
|
||||
agent._session_messages = messages
|
||||
# Touch activity so slow post-tool work plus a slow follow-up API call can't exceed
|
||||
# the gateway inactivity timeout (HERMES_AGENT_TIMEOUT).
|
||||
# Touch activity before continuing so the gateway's inactivity monitor never sees a stale timestamp
|
||||
# between tool completion and the start of the next API call. Without this, a tool-call result (which
|
||||
# takes ~0s to process) followed by slow post-tool processing (compression, persist) and a slow
|
||||
# follow-up API call can exceed the gateway inactivity timeout (HERMES_AGENT_TIMEOUT, default 1800s) and
|
||||
# the gateway kills the session before the next activity touch fires (#69559, #69131).
|
||||
agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}")
|
||||
return _verdict("continue")
|
||||
|
||||
|
||||
@@ -436,6 +436,14 @@ def continue_codex_incomplete(
|
||||
agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({n}/3)")
|
||||
# Spinner/heartbeat notice: these retries can take minutes and otherwise look
|
||||
# like infinite thinking.
|
||||
# #70773: same FD-recycle corruption vector as #67142. The shared OpenAI client's connection pool
|
||||
# must NOT be closed from this watchdog/poll thread — worker threads from previous stale-killed
|
||||
# attempts may still be unwinding their SSL BIOs. The request-local client is already closed above
|
||||
# via _close_request_client_once. The shared client will be replaced lazily by
|
||||
# _ensure_primary_openai_client on the next request.
|
||||
# Surface the continuation on the live spinner/status line (CLI/TUI/Desktop) and gateway heartbeat:
|
||||
# each of these retries can spend minutes waiting on the provider, and without a distinct notice the
|
||||
# user only sees a generic thinking spinner ("infinite thinking", #64434).
|
||||
agent._emit_wait_notice(
|
||||
f"↻ model returned reasoning with no final answer — asking it to continue ({n}/3)"
|
||||
)
|
||||
|
||||
@@ -17,6 +17,8 @@ _ONE_MILLION = Decimal("1000000")
|
||||
_NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1"
|
||||
|
||||
# Below $0.01, render at 4 dp so cheap-model costs never display as $0.00.
|
||||
# Sub-cent cost threshold: below $0.01, render at 4 decimal places so the display is non-zero (e.g. $0.0046
|
||||
# instead of $0.00). See #79220.
|
||||
_SUBCENT_THRESHOLD = Decimal("0.01")
|
||||
|
||||
# Attached to every CostResult with status="included" so consumers can
|
||||
@@ -27,13 +29,19 @@ _INCLUDED_NOTE = "subscription-included; no provider invoice for usage"
|
||||
def format_cost_label(amount: Decimal) -> str:
|
||||
"""Cost display label: zero → "$0.00"; sub-cent → "~$0.0046" (4 dp, or
|
||||
"~$<0.0001" when it rounds to 0.0000 so the label never reads as zero);
|
||||
else "~$1.23". Shared by per-response labels and insights cost buckets."""
|
||||
else "~$1.23". Shared by per-response labels and insights cost buckets.
|
||||
|
||||
This fixes #79220 where sub-cent per-turn costs on cheap models (DeepSeek, etc.) rendered as "$0.00"
|
||||
despite amount_usd carrying full Decimal precision.
|
||||
"""
|
||||
if amount == _ZERO:
|
||||
return "$0.00"
|
||||
if amount < _SUBCENT_THRESHOLD:
|
||||
label = f"~${amount:.4f}"
|
||||
# Compare the rendered label: a naive `< 0.00005` threshold misses
|
||||
# the exact boundary under ROUND_HALF_EVEN.
|
||||
# A positive amount that rounds to 0.0000 at 4 dp would render "~$0.0000" — a zero-looking label,
|
||||
# the exact #79220 dishonesty.
|
||||
return label if label != "~$0.0000" else "~$<0.0001"
|
||||
return f"~${amount:.2f}"
|
||||
|
||||
|
||||
@@ -135,6 +135,10 @@ def _transaction() -> Iterator[sqlite3.Connection]:
|
||||
|
||||
``sqlite3.Connection`` as a context manager only commits/rolls back; without
|
||||
the close, each call leaks a connection (and WAL/SHM fds) until GC runs.
|
||||
|
||||
Using ``with _connect()`` alone therefore leaks a connection — and its WAL/SHM file descriptors — on
|
||||
every call, deferring the close to the garbage collector, which over a long-running process can exhaust
|
||||
``RLIMIT_NOFILE`` (the cron-ledger sibling of this bug was #69567 / PR #69594).
|
||||
"""
|
||||
conn = _connect()
|
||||
try:
|
||||
|
||||
@@ -294,7 +294,14 @@ class VisionMessagePrepMixin:
|
||||
|
||||
def _anthropic_preserve_dots(self) -> bool:
|
||||
"""True for anthropic-compatible endpoints that keep dots in model names (DashScope, MiniMax, Xiaomi
|
||||
MiMo, OpenCode Go/Zen, ZAI/Zhipu; Bedrock's dotted inference-profile IDs 400 on the hyphenated form)."""
|
||||
MiMo, OpenCode Go/Zen, ZAI/Zhipu; Bedrock's dotted inference-profile IDs 400 on the hyphenated form).
|
||||
|
||||
Alibaba/DashScope keeps dots (e.g. qwen3.5-plus). OpenCode Go/Zen keeps dots for non-Claude models
|
||||
(e.g. minimax-m2.5-free). ``global.anthropic.claude-opus-4-7``,
|
||||
``us.anthropic.claude-sonnet-4-5-20250929-v1:0``) and rejects the hyphenated form with ``HTTP 400
|
||||
The provided model identifier is invalid``. Regression for #11976; mirrors the opencode-go fix for
|
||||
#5211
|
||||
"""
|
||||
if (getattr(self, "provider", "") or "").lower() in {
|
||||
"alibaba", "minimax", "minimax-cn", "opencode-go", "opencode-zen", "zai", "bedrock", "xiaomi", "vertex",
|
||||
}:
|
||||
|
||||
@@ -24,7 +24,11 @@ from agent.provider_base import ProviderBase
|
||||
def get_provider_env(name: str) -> str:
|
||||
"""Config-aware env lookup (``os.environ`` first, then ``~/.hermes/.env``) so
|
||||
credentials set through the config layer are visible in gateway sessions /
|
||||
delegate children / subprocess runs. Stripped value, or ``""`` when unset."""
|
||||
delegate children / subprocess runs. Stripped value, or ``""`` when unset.
|
||||
|
||||
Falls back to a bare ``os.getenv`` when the config module is unavailable (stripped installs, early
|
||||
import contexts). See #40190.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import get_env_value
|
||||
|
||||
|
||||
@@ -160,6 +160,12 @@ def _disabled_web_plugin_for(configured: Optional[str] = None, *, capability: Op
|
||||
backend because a disabled provider fails the availability gate and silently
|
||||
drops to the default. Bundled web plugins live under ``web/<vendor>`` with
|
||||
the provider name differing only by hyphen/underscore, so both are normalized.
|
||||
|
||||
When a user sets ``web.extract_backend: firecrawl`` (or the search equivalent) but also lists
|
||||
``web-firecrawl`` in ``plugins.disabled``, the provider never registers and the dispatcher would
|
||||
otherwise emit a misleading "No web extract provider configured. Set web.extract_backend to ..." error —
|
||||
even though the backend IS configured correctly. This helper detects that case so the dispatcher can
|
||||
point the user at the actual cause (issue #40190 follow-up: pi314's disabled-plugin symptom).
|
||||
"""
|
||||
def _norm(s: str) -> str:
|
||||
return s.strip().lower().replace("-", "_")
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user