diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index b0b12f7259..0fa0ccc374 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -1,52 +1,16 @@ -"""Context compression — extract the AIAgent methods that drive summarisation. +"""Context compression: feasibility probe, warning replay, compress, image fix. -Three concerns live here: - -* :func:`check_compression_model_feasibility` — startup probe of the - configured auxiliary compression model. Warns when the aux context - window can't fit the main model's compression threshold; auto-lowers - the session threshold when possible; hard-rejects auxes below - ``MINIMUM_CONTEXT_LENGTH``. - -* :func:`replay_compression_warning` — re-emit a stored warning through - the gateway ``status_callback`` once it's wired up (the callback is - set after :class:`AIAgent` construction). - -* :func:`compress_context` — the actual compression call. Runs the - configured compressor, splits the SQLite session, rotates the - session_id, notifies plugin context engines / memory providers, and - returns the compressed message list and active system prompt. - -* :func:`try_shrink_image_parts_in_messages` — image-too-large recovery - helper that re-encodes ``data:image/...;base64,...`` parts at a smaller - size so retries can fit under provider ceilings (Anthropic's 5 MB). - -``run_agent`` keeps thin wrappers for each so existing call sites -(``self._compress_context(...)``) keep working. Tests that exercise -these paths see no behavioural change. - -Thread-safety contract for extension points (#76354 review) ------------------------------------------------------------- - -When the host-level progress-aware timeout is enabled (the default: -``compression.context_timeout_seconds > 0``), the WHOLE compression pass — -including plugin/legacy **context engines** (``compress()`` / -``on_session_start`` / boundary callbacks) and **memory providers** -(``on_pre_compress`` / ``on_session_switch``) — runs on a pooled daemon -thread, not the conversation thread. Extension authors must assume: - -* Calls may arrive on an arbitrary pooled thread; do not rely on - thread-affinity or ``threading.local`` state shared with the caller. -* The input message list is a private deep snapshot owned by the worker; - engines MAY mutate it in place (legacy contract preserved), and that - mutation is invisible to the live conversation unless the pass commits. -* Publication to caller-visible / durable state happens ONLY on an admitted - commit (:class:`CompressionCommitFence`); after a host timeout the still- - running engine's work is discarded. -* Two compression passes never run concurrently for one session (durable - per-session lock), but passes for DIFFERENT sessions may run concurrently - on pool siblings — engine/provider instances shared across sessions must - be thread-safe or internally locked. +Thread-safety contract for extension points +-------------------------------------------- +With ``compression.context_timeout_seconds > 0`` (default) the whole pass, +context engines and memory providers included, runs on a pooled daemon thread. +* Calls may arrive on any pooled thread; never rely on thread-affinity/locals. +* The message list is a private deep snapshot; in-place mutation is allowed + but invisible to the live conversation unless the pass commits. +* State is published ONLY on an admitted :class:`CompressionCommitFence` + commit; work of an engine still running after a host timeout is discarded. +* One pass per session at a time (durable lock), but different sessions may + run concurrently, so shared engine/provider instances must be thread-safe. """ from __future__ import annotations @@ -80,11 +44,9 @@ from agent.session_activity import ActivityProvenance, normalize_activity_proven logger = logging.getLogger(__name__) -# Terminal compression outcomes published by host/hygiene timeout or cooldown -# writers. Detached heartbeat workers must not clobber these back to -# agent.compression after cancel (otherwise timeout is unobservable). Observing -# a terminal stamp (or a cancelled commit fence) also latches the heartbeat -# silent so a later UNKNOWN rewrite cannot re-arm a zombie worker. +# Terminal outcomes from host/hygiene timeout or cooldown writers. Detached +# heartbeat workers must not clobber these (timeout unobservable). Seeing one +# latches the heartbeat silent so a later UNKNOWN rewrite can't re-arm a zombie. _TERMINAL_COMPRESSION_PROVENANCES = frozenset( { ActivityProvenance.AGENT_COMPRESSION_TIMEOUT, @@ -92,19 +54,13 @@ _TERMINAL_COMPRESSION_PROVENANCES = frozenset( } ) -# Cooldown armed when a compression SPLIT fails (session_split_failed / -# rotation rollback, #97948 symptom B). Deliberately the FIRST rung of the -# timeout ladder (60/300/900 in context_compressor.py), not the 600s -# _SUMMARY_FAILURE_COOLDOWN_SECONDS: a split failure is usually a transient -# lease/DB condition, unlike a persistent summary-provider fault. +# Split failures are usually transient lease/DB conditions, so use the FIRST +# timeout-ladder rung (60s), not the 600s summary-provider cooldown. _SPLIT_FAILURE_COOLDOWN_SECONDS = 60 -# Stable marker the gateway matches on to re-tag the auto-compaction lifecycle -# status as ``kind="compacting"`` (tui_gateway/server.py::_status_update), so -# drivers like the desktop app can show an explicit "Summarizing…" indicator -# instead of the transcript appearing to silently reset. Keep the marker phrase -# intact if you reword COMPACTION_STATUS. Idle/preflight/retry lines do not -# contain this marker — ``is_compaction_progress_status`` covers those too. +# Marker tui_gateway/server.py::_status_update matches to tag kind="compacting" +# for drivers' "Summarizing…" UI. Keep the phrase intact when rewording. Idle/ +# preflight/retry lines lack it; is_compaction_progress_status covers those. COMPACTION_STATUS_MARKER = "Compacting context" COMPACTION_STATUS = ( f"🗜️ {COMPACTION_STATUS_MARKER} — summarizing earlier conversation so I can continue..." @@ -114,13 +70,11 @@ COMPACTION_DONE_STATUS = "✓ Context compaction complete — continuing turn... def _strip_marker_for_comparison(msgs: Any) -> Any: - """Copy ``msgs`` with the ``_db_persisted`` persistence marker removed. + """Copy ``msgs`` with the ``_db_persisted`` marker removed for no-op comparison. - Used by the no-op progress check: live and loaded dicts carry the marker - (loaded rows are stamped at materialization time, #92231) while - ``compress()`` output is marker-swept, so a raw ``==`` would misclassify a - semantically-identical no-op copy as progress. Non-list inputs and - non-dict entries pass through unchanged. + Live dicts carry the marker while ``compress()`` output is swept, so a raw + ``==`` would misclassify an identical no-op copy as progress. Non-list inputs + and non-dict entries pass through unchanged. """ from agent.context_compressor import _DB_PERSISTED_MARKER @@ -145,16 +99,9 @@ def _emit_compaction_done(agent: Any) -> None: logger.debug("status_callback error in compaction completion", exc_info=True) -# ── Routine compression status templates ──────────────────────────────────── -# Every ROUTINE (non-failure, non-manual-/compress) compression status line the -# agent emits lives here so the gateway noise filter and its tests can couple -# to the real emitted wording instead of hand-copied literals. These are -# suppressed on human-facing chat platforms by _TELEGRAM_NOISY_STATUS_RE -# (gateway/run.py) — when rewording ANY of them, update that regex and the -# pinned data in tests/gateway/test_telegram_noise_filter.py in the same PR. -# Failure notices (⚠ Compression aborted / empty transcript / codex compaction -# failed) and manual /compress feedback (manual_compression_feedback.py) are -# deliberate carve-outs from silence and must NOT be added here. +# Every ROUTINE compression status line lives here: suppressed on chat platforms +# by _TELEGRAM_NOISY_STATUS_RE (gateway/run.py); update that regex + telegram +# noise test when rewording. Failure notices and /compress feedback: NOT here. PRE_API_COMPRESSION_STATUS_TEMPLATE = ( "📦 Pre-API compression: ~{tokens:,} tokens " "near the context/output limit. Compacting before the next model call." @@ -180,14 +127,9 @@ COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE = ( "🗜️ Context reduced to {new_ctx:,} tokens (was {old_ctx:,}), retrying..." ) -# FAILURE-CLASS notice — a deliberate carve-out from routine-compression -# silence (#16775 class): the context is over the compression threshold but -# compression is blocked (summary-LLM cooldown / anti-thrash breaker), so the -# session will keep growing until the hard provider token limit kills it. -# This MUST stay visible on chat gateways. Do NOT add it to -# ROUTINE_COMPRESSION_STATUS_SAMPLES or the gateway noise regex -# (_TELEGRAM_NOISY_STATUS_RE); it is pinned un-swallowed in -# tests/gateway/test_telegram_noise_filter.py::VISIBLE_COMPRESSION_MESSAGES. +# FAILURE-class notice: compression blocked, so the session grows until the +# provider limit kills it. Must stay visible on gateways: never add it to +# ROUTINE_COMPRESSION_STATUS_SAMPLES or _TELEGRAM_NOISY_STATUS_RE. CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = ( "⚠ Context is over the compression threshold " "(~{tokens:,} tokens >= {threshold:,}) " @@ -196,9 +138,8 @@ CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = ( "session or /compress to retry immediately." ) -# Sample-formatted instances of every routine compression status line, for -# behavioral tests that iterate the ACTUAL emitted wording (formatted from the -# same constants the emission sites use) through the gateway noise filter. +# Formatted from the same constants the emission sites use, so noise-filter +# tests exercise the ACTUAL wording. ROUTINE_COMPRESSION_STATUS_SAMPLES = ( COMPACTION_STATUS, COMPACTION_DONE_STATUS, @@ -217,13 +158,10 @@ ROUTINE_COMPRESSION_STATUS_SAMPLES = ( def is_compaction_progress_status(text: str | None) -> bool: """True for in-progress auto-compaction lifecycle lines (not the done edge). - ``tui_gateway.server._status_update`` re-tags matching ``lifecycle`` - statuses as ``kind="compacting"`` so TUI and desktop can show a summarizing - indicator for the whole pause. Matching only ``COMPACTION_STATUS_MARKER`` - left idle/preflight/retry lines looking like a hung turn (#97239). - - The terminal ``COMPACTION_DONE_STATUS`` is emitted as ``kind="compacted"`` - and must not match here. + The gateway re-tags matches as ``kind="compacting"`` for the whole pause; + matching only the marker left idle/preflight/retry lines looking hung. + ``COMPACTION_DONE_STATUS`` is emitted as ``kind="compacted"`` and must not + match here. """ if not isinstance(text, str): return False @@ -248,52 +186,13 @@ def is_compaction_progress_status(text: str | None) -> bool: ) -def _builtin_memory_prompt_snapshot(agent: Any) -> Optional[Tuple[str, str]]: - """Return the built-in memory text that can affect a system prompt. - - ``MemoryStore`` freezes this text until ``load_from_disk()``. Rendering - the frozen blocks after that reload lets compression retain the exact - cached system prompt when it already embeds the current memory (see - :func:`_cached_prompt_reflects_builtin_memory`). An unreadable snapshot - returns ``None`` so callers take the conservative rebuild path. - """ - store = getattr(agent, "_memory_store", None) - if store is None: - return "", "" - try: - memory = ( - store.format_for_system_prompt("memory") or "" - if getattr(agent, "_memory_enabled", False) - else "" - ) - user = ( - store.format_for_system_prompt("user") or "" - if getattr(agent, "_user_profile_enabled", False) - else "" - ) - except Exception: - return None - return memory, user - - def _refresh_agent_tool_definitions(agent) -> bool: """Rebuild agent.tools at the compaction commit boundary. - Forever-sessions (Bot Mode desktop chats, gateway channels) never - restart, so compaction is the only moment a config change can reach the - tool snapshot: dynamic schemas (image_generate capability gating, - delegate_task limits, execute_code interpreter/stub text) are built - once at agent init and then frozen for cache stability. The prompt - cache is already invalidated at this boundary, so refreshing here is - free; between compactions the snapshot stays byte-stable as before. - - Delegates to tools.mcp_tool.refresh_agent_mcp_tools — the single shared - snapshot rebuild (atomic publish, generation guard, memory/LCM tool - re-injection) — in content_aware mode, which also swaps when schema - CONTENT changed under a stable name set (the dynamic-schema case; the - name-only diff is sufficient for its MCP-reload callers). Returns True - when new tools were added. Never raises: tool staleness must not fail - a compaction. + Forever-sessions never restart, so this is the only moment config changes + reach the frozen dynamic tool schemas; the prompt cache is already invalid. + Delegates to refresh_agent_mcp_tools in content_aware mode (swaps on schema + CONTENT change). Returns True when tools were added. Never raises. """ from tools.mcp_tool import refresh_agent_mcp_tools @@ -305,43 +204,6 @@ def _refresh_agent_tool_definitions(agent) -> bool: return bool(added) -def _cached_prompt_reflects_builtin_memory(agent: Any, cached_prompt: str) -> bool: - """Whether the cached system prompt already embeds current built-in memory. - - The retention fast path must NOT compare the memory snapshot before vs - after the disk reload: on fresh-agent surfaces (gateway, TUI) the cached - prompt is restored from the session DB and can predate mid-session memory - writes that the fresh ``MemoryStore`` already picked up at init — the - snapshot is then identical on both sides of the reload while the prompt - itself is stale, and retaining it would latch old memory for the life of - the session (and re-persist it via ``update_system_prompt``). - - Instead, verify the CURRENT (post-reload) rendered blocks appear verbatim - in the cached prompt, and that no leftover block header remains for a - target whose entries have since been emptied or disabled. - """ - snapshot = _builtin_memory_prompt_snapshot(agent) - if snapshot is None: - return False - try: - from tools.memory_tool import MEMORY_BLOCK_HEADERS - except Exception: - return False - for target, block in zip(("memory", "user"), snapshot): - block = block.strip() - if block: - # build_system_prompt_parts embeds the stripped block verbatim; - # the rendered text includes the usage header, so any entry - # change (or char-count change) breaks containment → rebuild. - if block not in cached_prompt: - return False - elif MEMORY_BLOCK_HEADERS[target] in cached_prompt: - # The prompt still carries a block for a target that is now - # empty/disabled — stale; rebuild. - return False - return True - - _COMPRESSOR_ATTEMPT_STATE_FIELDS = ( "_previous_summary", "_summary_has_user_turn", @@ -381,11 +243,10 @@ _COMPRESSOR_COOLDOWN_STATE_FIELDS = ( def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]: - """Copy only mutable bookkeeping owned by one compression attempt. + """Copy only the mutable bookkeeping owned by one compression attempt. - The explicit allow-list avoids copying provider clients, SessionDB handles, - locks, and plugin resources. Missing fields are intentionally ignored so - legacy and third-party compressors keep their existing contract. + The allow-list avoids copying clients, DB handles, locks and plugin resources; + missing fields are ignored so legacy/third-party compressors keep working. """ try: values = vars(compressor) @@ -401,48 +262,25 @@ def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]: return copy.deepcopy(selected) -# --------------------------------------------------------------------------- -# Attempt ownership (#96634 follow-up). -# -# The stall-fallback path deliberately DETACHES a timed-out primary worker -# (fence cancel wins; the future stays on the shared pool) and immediately -# starts a fallback attempt against the SAME ContextCompressor. Two races -# follow from that overlap: -# -# 1. The late primary's unwind still calls _restore_compressor_attempt_state -# with the PRIMARY's pre-attempt snapshot. Landing after the fallback's -# commit, it rolls _previous_summary / cooldown / provenance / telemetry -# back to pre-primary values — silently discarding fallback-owned state. -# 2. _compression_cancelled_check is one shared attribute: the late -# primary's ``finally`` clears the callback the fallback just installed, -# so the fallback's F4 cancellation consult reads None. -# -# Both are fixed with a monotonic per-compressor attempt generation, claimed -# under one module lock. Restores and callback set/clear are keyed to the -# claiming generation and no-op when a newer attempt owns the compressor. -# The commit fence still owns COMMIT admission; the generation owns -# compressor-ATTRIBUTE writes — two different boundaries. -# --------------------------------------------------------------------------- +# Attempt ownership: stall-fallback detaches a timed-out worker and reuses the +# compressor, so its late unwind could restore a stale snapshot or clear the +# fallback's cancel check. Generation guards ATTRIBUTE writes; fence, COMMITs. _COMPRESSOR_ATTEMPT_LOCK = threading.Lock() def _claim_compressor_attempt(compressor: Any) -> int: - """Claim the compressor for a new attempt; returns its generation id. + """Claim the compressor for a new attempt; return its monotonic generation id. - Monotonic per compressor instance. Any restore or cancelled-check - mutation stamped with an OLDER generation becomes a no-op, so a - detached, late-unwinding attempt cannot clobber its successor's state. + Restores or cancelled-check mutations stamped with an OLDER generation no-op, + so a detached late attempt cannot clobber its successor's state. """ with _COMPRESSOR_ATTEMPT_LOCK: generation = int(getattr(compressor, "_compression_attempt_generation", 0) or 0) + 1 try: compressor._compression_attempt_generation = generation except Exception: - # Slotted/frozen third-party compressor: ownership tracking is - # unavailable; generation 0 disables the guard (legacy behavior). - # Per-compressor all-or-nothing, NOT per-attempt: a compressor - # that rejects the setattr rejects it for EVERY claim, so gen-0 + # Slotted/frozen compressor: gen 0 disables the guard. Per-compressor, so gen-0 # and gen>0 attempts can never coexist on one instance. return 0 return generation @@ -476,8 +314,8 @@ def _clear_compression_cancelled_check_if_owner( ) -> bool: """Clear the cancellation consult only when *generation* installed it. - A detached late primary's ``finally`` must not tear down the callback a - newer fallback attempt just installed. Returns True when cleared. + Prevents a detached late primary from tearing down a newer fallback's + callback. Returns True when cleared. """ with _COMPRESSOR_ATTEMPT_LOCK: owner = getattr(compressor, "_compression_cancelled_check_owner", None) @@ -499,14 +337,10 @@ def _restore_compressor_attempt_state( durable_cooldown_state: Optional[dict[str, Any]] = None, attempt_generation: Optional[int] = None, ) -> None: - """Restore the safe per-attempt snapshot after a pre-commit hard cancel. + """Restore the per-attempt snapshot after a pre-commit hard cancel. - ``attempt_generation`` (when provided) is the claim the calling attempt - took via :func:`_claim_compressor_attempt`. A restore stamped with a - stale generation no-ops: the stall-fallback path detaches a timed-out - primary and hands the compressor to a fallback attempt, and the late - primary's unwind must not roll fallback-owned state back to the - primary's pre-attempt snapshot (#96634 post-merge review, claim 1). + A restore stamped with a stale ``attempt_generation`` no-ops so a timed-out + primary's late unwind cannot roll back state owned by the fallback attempt. """ if attempt_generation is not None and not _compressor_attempt_is_current( compressor, attempt_generation @@ -519,11 +353,9 @@ def _restore_compressor_attempt_state( getattr(compressor, "_compression_attempt_generation", None), ) return - # A successful summary clears the durable cooldown before the outer commit - # boundary. Recreate (or clear) that row before restoring exact in-memory - # values, otherwise the next refresh would overwrite this rollback. Unknown - # durable state and intentionally unpersisted local cooldowns are never - # converted into destructive DB writes during cancellation. + # Success clears the durable cooldown pre-commit; recreate/clear that row BEFORE + # restoring in-memory values or the next refresh overwrites the rollback. Never + # turn unknown durable state / unpersisted local cooldowns into DB writes. if ( "_summary_failure_cooldown_until" in snapshot and durable_cooldown_authoritative is not False @@ -589,14 +421,9 @@ def _restore_compressor_attempt_state( exc_info=True, ) restored = copy.deepcopy(snapshot) - # Close the check-then-act window: the entry check above runs before the - # (potentially slow) durable-cooldown rollback, so a fallback attempt can - # claim the compressor in between. Re-validate and write the in-memory - # fields under the SAME lock claims are taken under — a stale attempt can - # never interleave its setattr loop with a newer attempt's writes. The - # durable rollback above is safe either way: the dangerous direction - # (restore landing AFTER the fallback's writes) requires the fallback to - # have claimed first, which the entry check already rejects. + # Re-validate under the claim lock: the slow durable rollback above leaves a + # window where a fallback may have claimed; stale writes must not interleave. + # The rollback itself is safe: landing after a fallback needs a prior claim. with _COMPRESSOR_ATTEMPT_LOCK: if attempt_generation is not None and attempt_generation and ( int(getattr(compressor, "_compression_attempt_generation", 0) or 0) @@ -618,11 +445,9 @@ def _capture_authoritative_cooldown_under_lease( ) -> tuple[Optional[bool], Optional[dict[str, Any]]]: """Refresh and snapshot built-in durable cooldown state under the lease. - Third-party compressors are deliberately not invoked here: arbitrary plugin - callbacks must not run while the session lease is held. A durable read - failure returns ``False`` so rollback cannot mistake unknown durable state - for an authoritative empty row and clear it; an unavailable legacy API - returns ``None`` and preserves the compatibility path. + Third-party compressors are not invoked: plugin code must not run under the + lease. Returns ``False`` on durable read failure (rollback must not mistake + unknown state for an empty row) and ``None`` when the legacy API is absent. """ try: from agent.context_compressor import ContextCompressor @@ -644,9 +469,8 @@ def _capture_authoritative_cooldown_under_lease( return None, None if not callable(raw_reader): return False, None - # Capture the exact persisted representation first. The active getter - # intentionally filters expired rows and therefore cannot serve as a - # lossless rollback snapshot. + # Read the raw persisted row: the active getter filters expired rows and is not + # a lossless rollback snapshot. durable_state = raw_reader(session_db, session_id) if not isinstance(durable_state, dict): raise TypeError("raw compression cooldown snapshot must be a mapping") @@ -673,61 +497,34 @@ def _capture_authoritative_cooldown_under_lease( class CompressionCommitFence: """Fence timeout cancellation against post-summary session mutation. - Compression itself is synchronous and may be running in an executor thread. - A caller can stop waiting for the summary, but it cannot kill that thread. - This fence makes the commit boundary deterministic: cancellation either wins - before session mutation starts, or waits until an already-started commit is - fully complete before the caller proceeds. + The sync worker thread cannot be killed; the fence makes the commit boundary + deterministic: cancellation wins before mutation starts, or waits for an + already-started commit to finish completely. """ def __init__(self, total_ceiling_seconds: float | None = None) -> None: self._lock = threading.Lock() self._cancelled = False self._commit_started = False - # Lock-free commit-phase marker (#76354 review F1). ``begin_commit`` - # RETAINS ``self._lock`` until ``finish_commit``, so any host-side - # observation that needs the lock (``try_cancel_before_commit``) - # blocks/space-outs for the whole commit. This Event is set inside - # ``begin_commit`` while the lock is held but is READABLE WITHOUT the - # lock, so a host can observe "a commit was admitted and may be in - # flight" even while the commit itself is hung — which is exactly when - # the overrun warning must be able to fire. + # begin_commit holds self._lock until finish_commit, so this Event is readable + # WITHOUT the lock: hosts can see a hung commit and fire the overrun warning. self._commit_phase = threading.Event() - # Lock-free admission revocation (#76354 review F2). Set by - # :meth:`revoke_commit_admission` on ANY host unwind (KeyboardInterrupt, - # cancellation, unexpected exception) without touching the fence lock, - # so a host that cannot afford to block behind an in-flight commit can - # still guarantee no FUTURE commit is admitted. Plain bool store — - # atomic in CPython. + # Set on ANY host unwind without the fence lock, so a host that cannot block + # behind an in-flight commit still blocks FUTURE commits. bool store is atomic. self._admission_revoked = False - # Holder-qualified durable-lock release hook (#76354 review F4; - # transplanted from PR #71569 by @ciabata-git). The worker publishes an - # idempotent, holder-scoped release callable once it owns the durable - # compression lock; a timed-out host invokes it to free the lease - # without racing a NEW holder (DB release is holder-qualified, so a - # stale release can never delete a replacement's row — no ABA). + # Worker publishes a holder-scoped release once it owns the durable lock; a + # timed-out host frees the lease without racing a NEW holder (no ABA). self._lock_release_guard = threading.Lock() self._cancelled_lock_release: Optional[Callable[[], None]] = None self._cancelled_lock_release_requested = False - # Forward-progress telemetry: the compression worker touches this - # whenever the streamed summary call produces a token (see - # ContextCompressor._call_summary_llm). Waiters use it to distinguish - # a SLOW-but-alive summary model from a HUNG one, so slow models are - # not killed by a fixed wall-clock deadline while tokens are moving. + # Touched per streamed summary token; waiters distinguish SLOW-but-alive from + # HUNG so slow models are not killed by a fixed wall-clock deadline. self._last_progress = time.monotonic() self._progress_observed = False self._deadline: float | None = None self._retain_cancelled_lock_until_worker_done = False - # #97963: set by the worker (mark_commit_watermark_fenced) once its - # commit path is watermark-fenced — i.e. it captured the session's - # active-row watermark at compression start, so any row appended - # AFTER that point survives a late commit verbatim as concurrent - # tail (archive_and_compact / publish_compression_child clone rows - # above the watermark instead of archiving them). Hosts read this - # at the turn-hold boundary to decide whether a detached worker may - # KEEP its commit admission (safe: newer turns cannot be clobbered) - # or must be cancelled as before (unfenced commit; discard is the - # only safe outcome). Plain bool store — atomic in CPython. + # Set once the commit path captured the active-row watermark: later rows survive + # as concurrent tail, so hosts may KEEP a detached worker's commit admission. self._commit_watermark_fenced = False if total_ceiling_seconds is not None: self.set_total_ceiling_seconds(total_ceiling_seconds) @@ -742,9 +539,8 @@ class CompressionCommitFence: def touch_progress(self) -> None: """Record forward progress (e.g. a streamed summary token arriving). - Called from the compression worker thread; read by async waiters via - :meth:`seconds_since_progress`. A bare float store is atomic in - CPython, so no lock is needed. + Called from the worker thread, read by waiters via ``seconds_since_progress``; + a bare float store is atomic in CPython so no lock is needed. """ self._last_progress = time.monotonic() self._progress_observed = True @@ -763,13 +559,8 @@ class CompressionCommitFence: def deadline_monotonic(self) -> float | None: """The armed deadline as an absolute ``time.monotonic()`` instant. - :meth:`set_total_ceiling_seconds` documents this deadline as "shared by - the host and worker", but until #99692 only the host could read it — - ``deadline_exceeded`` answers "is it past?" for a caller that is already - polling, which is useless to a worker blocked inside a provider stream. - Publishing the instant itself lets the worker's stream consumer stop at - exactly the moment the host stops waiting (see - ``auxiliary_client.aux_stream_deadline``). + Published so the worker's stream consumer can stop exactly when the host + stops waiting (see ``auxiliary_client.aux_stream_deadline``). """ return self._deadline @@ -780,9 +571,8 @@ class CompressionCommitFence: def cancel_before_commit(self, cancel_event: Any = None) -> bool: """Cancel a pending commit, or wait for an active commit to finish. - Returns ``True`` when cancellation won before the commit boundary. - Returns ``False`` when the worker had already entered the boundary; in - that case acquiring this lock waits until all session mutation finishes. + Returns ``True`` when cancellation won before the commit boundary; ``False`` + after blocking until an already-started commit fully completed. """ with self._lock: if self._commit_started: @@ -797,8 +587,8 @@ class CompressionCommitFence: def try_cancel_before_commit(self) -> Optional[bool]: """Non-blocking form of :meth:`cancel_before_commit`. - Returns ``None`` while an active commit owns the fence, allowing an - async caller to yield instead of blocking its event loop. + Returns ``None`` while an active commit owns the fence so an async caller can + yield instead of blocking its event loop. """ if not self._lock.acquire(blocking=False): return None @@ -821,10 +611,8 @@ class CompressionCommitFence: self._cancelled = True self._lock.release() if self._admission_revoked: - # Round-2 #1: a revoke that lost the fence-lock race to this - # very begin_commit deferred its lease release; the commit was - # refused, so the release is safe (and idempotent with the - # worker's own holder-qualified cleanup) right now. + # A revoke that lost the fence-lock race deferred its lease release; commit was + # refused, so releasing now is safe (idempotent with holder-qualified cleanup). self.release_cancelled_compression_lock() return False self._commit_started = True @@ -838,24 +626,16 @@ class CompressionCommitFence: self._commit_phase.clear() self._lock.release() if self._admission_revoked: - # Round-2 #1: a revoke that arrived while THIS commit was in - # flight deferred its durable-lease release rather than freeing - # the lock out from under an active SessionDB mutation. The - # commit is now fully complete, so perform the deferred release - # here — promptly, without relying on the (possibly parked) - # worker thread's outer cleanup. Idempotent with that cleanup: - # the DB release is holder-qualified. + # A revoke during THIS commit deferred its lease release (no freeing under an + # active SessionDB mutation); commit is done, release now. Holder-qualified. self.release_cancelled_compression_lock() @property def commit_in_flight(self) -> bool: """Lock-free read: an admitted commit has begun and not yet finished. - Safe to call from the host while the worker holds the fence lock for - the whole commit (a hung SessionDB write). Hosts use this to reach - their overrun-warning loop WHILE the commit is blocked instead of - spinning on ``try_cancel_before_commit`` (which needs the lock the - worker retains until ``finish_commit``). + Safe while the worker holds the fence lock for a hung commit; lets hosts reach + the overrun-warning loop instead of spinning on ``try_cancel_before_commit``. """ return self._commit_phase.is_set() @@ -871,13 +651,8 @@ class CompressionCommitFence: def mark_commit_watermark_fenced(self) -> None: """Record that this attempt's commit is bounded by a start watermark. - Called by the compression worker right after it captures - ``get_active_message_watermark()`` under the durable compression - lock (#75316/#87484). A watermark-fenced commit archives ONLY rows - at or below the watermark; rows appended later — e.g. the user turn - the host released at the turn-hold boundary (#97963) — are cloned - as live concurrent tail. That is exactly the property a host needs - before letting a detached worker keep its commit admission. + A watermark-fenced commit archives only rows at or below the watermark and + clones later rows as live tail, so a detached worker may keep its admission. """ self._commit_watermark_fenced = True @@ -889,42 +664,18 @@ class CompressionCommitFence: def allow_cancelled_lock_release(self) -> None: """Undo :meth:`retain_compression_lock_until_worker_done`. - Called by the host after a bounded-grace join confirmed the timed-out - worker actually exited: the overlap hazard is gone, so the durable - lease may be released normally and a fallback/retry attempt can - proceed against a genuinely quiescent session. + Called after a bounded join confirmed the timed-out worker exited, so the + durable lease may be released and a fallback attempt can proceed. """ self._retain_cancelled_lock_until_worker_done = False def revoke_commit_admission(self) -> None: """Revoke FUTURE commit admission without blocking on the fence lock. - #76354 review F2: every host unwind path (KeyboardInterrupt, task - cancellation, unexpected exception while waiting) must guarantee a - detached worker cannot later enter the commit boundary and mutate - durable/session state. The flag store is lock-free: a commit that is - ALREADY in flight cannot be safely abandoned (the invariant "commit - never abandoned mid-mutation" holds), but no NEW commit will be - admitted after this call — ``begin_commit`` re-checks the flag under - the fence lock. - - Round-2 #1 (durable-lease timing): the worker's holder-qualified - lease release (F4) must NOT run while an admitted commit is still - mutating SessionDB — a second compressor could otherwise acquire the - durable lock mid-commit and interleave with the first commit's - writes. The release decision is therefore made under the fence lock: - - - non-blocking acquire succeeds → no commit is in flight (an - admitted commit RETAINS the lock until ``finish_commit``), so the - lease is released immediately, while still holding the lock so a - concurrent ``begin_commit`` cannot slip in between the check and - the release (it would be refused anyway — the flag is already set). - - acquire fails → the lock holder is either an in-flight commit or a - transient boundary (lock-setup / cancel admission). Defer: the - release then runs in ``finish_commit`` (after the mutation fully - completes) or on the ``begin_commit``-refusal path, whichever the - worker reaches first. Both are idempotent with the worker's own - outer cleanup because the DB release is holder-qualified. + An in-flight commit is never abandoned, but ``begin_commit`` re-checks the + flag under the lock so no new commit is admitted. The lease release must not + run mid-commit (a second compressor could interleave): released now if the + lock is free, else deferred to ``finish_commit``/refusal (holder-qualified). """ self._admission_revoked = True if self._lock.acquire(blocking=False): @@ -932,25 +683,17 @@ class CompressionCommitFence: self.release_cancelled_compression_lock() finally: self._lock.release() - # else: deferred — finish_commit()/begin_commit() re-check - # _admission_revoked and perform the release once no commit can be - # mid-mutation. + # else: deferred — finish_commit()/begin_commit() re-check _admission_revoked + # and release once no commit can be mid-mutation. - # ── Holder-qualified durable-lease cancellation (#76354 F4) ────────── - # Transplanted from PR #71569 (@ciabata-git): the worker publishes an - # idempotent, holder-scoped release hook once it owns the durable - # compression lock, and the host invokes it after winning cancellation. - # ABA safety comes from SessionDB.release_compression_lock being - # holder-qualified (DELETE ... WHERE holder = ?), so a stale release can - # never free a NEW holder's lease. + # ── Holder-qualified durable-lease cancellation: release is DELETE WHERE + # holder = ?, so a stale release can never free a NEW holder's lease (no ABA). def begin_lock_setup(self) -> bool: """Fence durable-lock acquisition and release-hook publication. - The caller keeps the fence until it has either published the exact - holder-qualified release hook or established that no lock was - acquired. A timeout cannot therefore win in the gap between acquiring - the durable lock and making its cancellation cleanup callable. + The caller holds the fence until the holder-qualified release hook is + published (or no lock was taken), so a timeout cannot win in that gap. """ self._lock.acquire() if self.is_cancelled or self._admission_revoked: @@ -967,8 +710,8 @@ class CompressionCommitFence: ) -> bool: """Publish the timed-out worker's holder-qualified lock release. - Returns whether cancellation cleanup was requested before publication. - In that race, the release runs synchronously before this method returns. + Returns whether cleanup was already requested; in that race the release runs + synchronously before returning. """ with self._lock_release_guard: self._cancelled_lock_release = release @@ -986,10 +729,8 @@ class CompressionCommitFence: def release_cancelled_compression_lock(self) -> None: """Release the cancelled worker's lock without finalizing its clients. - Callers invoke this only after cancellation won (fence cancelled or - admission revoked). A request that races ahead of lock-hook - publication is retained and fulfilled synchronously when the worker - publishes the hook. + Only valid after cancellation won. A request racing ahead of hook publication + is retained and fulfilled when the worker publishes the hook. """ if self._retain_cancelled_lock_until_worker_done: return @@ -1005,45 +746,30 @@ class CompressionCommitFence: DEFAULT_CONTEXT_TIMEOUT_SECONDS = 120.0 DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0 -# Distinct from ``explicit_interrupt``: a /stop that arrived after the summary -# stream had already crossed the no-progress stall window (#96775). Ordinary -# early /stop stays cooldown-neutral; this class arms the durable backoff so -# the next automatic turn does not re-enter the same stalled strategy. +# Unlike explicit_interrupt: a /stop after the stall window arms the durable +# backoff so the next automatic turn does not re-enter the stalled strategy. STALL_INTERRUPTED_FAILURE_CLASS = "stall_interrupted" -# Shared daemon pool for sync compress_context timeout wraps — analogous to -# asyncio's default executor used by gateway session hygiene's -# ``loop.run_in_executor(None, ...)``, but daemon so a fence-cancelled hung -# worker cannot block interpreter exit via concurrent.futures' atexit join. -# Created lazily; never shut down per call (a timed-out worker may still be -# winding down after fence cancel). +# Daemon pool so a fence-cancelled hung worker cannot block interpreter exit +# via the atexit join. Never shut down per call (workers may still be winding). _compress_timeout_executor = None _compress_timeout_executor_lock = threading.Lock() -# Commit-phase overrun wait slice: once an in-flight SessionDB commit runs -# past the total ceiling, keep waiting in bounded increments of this size so -# every overrun window produces a fresh (escalating) log line instead of one -# silent unbounded future.result(). Clamped down to the ceiling for tiny test -# ceilings so overrun reporting stays observable at test timescales. +# Overrun waits proceed in bounded slices so each window logs (escalating) +# instead of one silent future.result(). Clamped to ceiling for tiny test values _COMMIT_OVERRUN_WAIT_SLICE_SECONDS = 30.0 -# Bounded grace given to a fence-cancelled compression worker to actually -# exit before the host moves on (#97488). A worker that exits inside the -# grace window proves no provider call is still in flight, so the durable -# lease can be released safely even on the total-ceiling path. A worker that -# does NOT exit is orphaned behind the poison fence (its late result cannot -# commit) and, on the total-ceiling path, keeps the holder-qualified lease -# retained so a new attempt cannot overlap the unchanged session. +# A worker exiting within the grace proves no provider call is in flight, so the +# lease can be released even on the total-ceiling path. One that doesn't exit is +# orphaned behind the poison fence and keeps its lease so no attempt overlaps. _CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS = 5.0 def _join_cancelled_worker(future: Any, grace_seconds: float) -> bool: """Best-effort bounded join of a fence-cancelled compression worker. - Returns True when the worker future settled (result, exception, or - pre-start cancellation) within ``grace_seconds`` — i.e. the worker thread - provably exited and cannot be holding a provider call open. Returns - False for a worker that is still running; the caller must treat it as an + Returns True when the future settled within ``grace_seconds`` (thread provably + exited); False for a still-running worker, which the caller must treat as an orphan behind the poison fence. """ try: @@ -1059,39 +785,23 @@ def _join_cancelled_worker(future: Any, grace_seconds: float) -> bool: # Never started; nothing can be in flight. return True except Exception: - # The worker raised — it exited. The exception is intentionally - # swallowed here: the host already chose the fallback result, and the - # fence prevents the failed attempt from touching session state. + # Exception swallowed: the host already chose the fallback result and the fence + # keeps the failed attempt from touching session state. logger.debug( "cancelled compression worker exited with an exception", exc_info=True, ) return True -# Bounded admission for the shared compress-timeout pool (#76354 review F6). -# The stdlib executor queue is unbounded: with all four workers wedged in hung -# summaries, a fifth compression would queue silently, wait out its whole -# timeout without ever starting, and remain eligible to run as a stale job -# whenever a worker recovered. Admission is therefore capped at the worker -# count — when every worker slot is occupied (running OR admitted-not-started) -# submission FAILS FAST and the caller continues without compression. -# -# Recovery contract when all workers are wedged: new compressions fail fast -# (no queue growth, conversation continues uncompressed, a warning is logged -# each attempt); wedged workers are fence-cancelled so they cannot publish -# anything when they eventually return, and each recovery frees its admission -# slot via the future done-callback, restoring normal service. If a worker -# NEVER returns, its slot is lost for the process lifetime — bounded, -# observable degradation instead of an unbounded stale-job queue. + +# Executor queue is unbounded: a queued job would wait out its timeout unstarted +# and run stale later. Cap admission at worker count; fail fast (warn, continue +# uncompressed). Slots free via done-callback; a never-returning worker loses 1. _COMPRESS_EXECUTOR_MAX_WORKERS = 4 _compress_admission_lock = threading.Lock() _compress_admitted_count = 0 -class CompressionExecutorSaturatedError(RuntimeError): - """All compression pool slots are occupied; submission was refused.""" - - def _try_admit_compression_job() -> bool: """Reserve one bounded compression-pool admission slot (F6).""" global _compress_admitted_count @@ -1120,9 +830,8 @@ def _get_compress_timeout_executor(): with _compress_timeout_executor_lock: if _compress_timeout_executor is None: - # Small pool: compress is rare and heavy. Sized for a few - # overlapping calls (live compress + fence-cancelled workers - # still winding down), not asyncio's min(32, cpu+4) fan-out. + # Small pool: compress is rare/heavy; sized for live compress + cancelled + # workers still winding down, not asyncio's min(32, cpu+4). _compress_timeout_executor = DaemonThreadPoolExecutor( max_workers=_COMPRESS_EXECUTOR_MAX_WORKERS, thread_name_prefix="compress-ctx-timeout", @@ -1135,9 +844,8 @@ def resolve_context_compression_timeouts( ) -> Tuple[float, float]: """Return ``(idle_timeout_seconds, total_ceiling_seconds)``. - ``idle_timeout_seconds <= 0`` disables the owned progress-aware wrapper. - The ceiling is clamped to at least one idle window when the idle budget - is positive, matching gateway hygiene semantics. + ``idle_timeout_seconds <= 0`` disables the progress-aware wrapper. The ceiling + is clamped to at least one idle window when the idle budget is positive. """ idle = DEFAULT_CONTEXT_TIMEOUT_SECONDS ceiling = DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS @@ -1181,11 +889,8 @@ def compression_attempt_stalled( ) -> bool: """Return whether a pre-commit cancel landed after the stall window. - An ordinary early ``/stop`` must stay cooldown-neutral. When the fence - (or, without a fence, the attempt clock) has already sat idle for the - configured compression inactivity budget, the interrupt is a stalled - attempt — the same condition the host timeout uses — and the next - automatic turn must not blindly retry that strategy (#96775). + An early ``/stop`` stays cooldown-neutral; an interrupt after the inactivity + budget counts as a stall so the next automatic turn does not blindly retry. """ idle = idle_timeout_seconds if idle is None: @@ -1237,8 +942,8 @@ def _record_stall_interrupted_backoff( ) -> bool: """Persist a stall-interrupted cooldown after snapshot restore. - Must run *after* ``_restore_compressor_attempt_state`` so rollback cannot - wipe the new row. Returns True when the stall backoff was recorded. + Must run *after* ``_restore_compressor_attempt_state`` so rollback cannot wipe + the new row. Returns True when the backoff was recorded. """ if not compression_attempt_stalled( commit_fence=commit_fence, started_at=started_at @@ -1271,18 +976,10 @@ def _record_stall_interrupted_backoff( def resolve_compression_fallback_route() -> Optional[dict]: """Return the first usable ``auxiliary.compression.fallback_chain`` entry. - The chain is the user's declared answer to "the configured compression - route is unhealthy". The auxiliary client applies it from its exception - handler, so a route that *stalls* — connection open, no tokens, aborted by - the host's progress-aware timeout — never reaches it (#78981). This - resolves the same config into explicit ``call_llm`` route arguments the - stall path can pin onto one retry. - - Only the first structurally complete entry is returned: one extra bounded - attempt is the budget for a stall, and if that entry cannot build a client - (or errors) the auxiliary client's own exception path still walks the rest - of the chain from there. Returns ``None`` when no entry is usable, which - keeps the historical continue-without-compression behaviour. + The aux client applies the chain only from its exception handler, so a silent + stall never reaches it; this pins the route onto one bounded retry instead. + Only the first complete entry: if it errors, the aux client's own exception + path walks the rest. ``None`` when none is usable (skip compression). """ try: from agent.auxiliary_client import ( @@ -1346,23 +1043,9 @@ def _retry_compression_on_fallback_chain( ) -> Optional[Tuple[list, str]]: """Re-run an aborted compression once with the summary route pinned. - Returns the fallback attempt's ``(messages, system_prompt)`` when it - actually compressed, or ``None`` when there was nothing to fall back to, - the attempt raised, or it produced no compression either — the caller then - degrades exactly as it did before. - - The retry is bounded the same way the primary was: silence for one idle - window ends it, while a fallback that is streaming keeps its ceiling. The - entry's own ``timeout`` (when declared) sets that idle window, so a - fallback tuned for a slower-but-healthy backend is not held to a deadline - the stalled primary defined (#62452 semantics, applied to the stall path). - - Known limitation (accepted, #96634 review): the retry re-runs the COMPLETE - worker, which repeats memory/plugin pre-compression callbacks. Built-in - callbacks are idempotent (re-reads and overwrites of attempt-scoped - state); third-party plugin callbacks are advised to be. Splitting the - worker to resume mid-pipeline would couple this path to every host's - callback ordering — deliberately out of scope. + Returns ``(messages, system_prompt)`` on real compression, else ``None`` and + the caller degrades as before. The entry's ``timeout`` sets the idle window. + Re-runs the whole worker, so pre-compression callbacks must be idempotent. """ # An explicit stop is not a stalled route. The retry worker would abort on # the same event anyway, but starting one at all makes /stop look ignored. @@ -1374,10 +1057,8 @@ def _retry_compression_on_fallback_chain( if route is None: return None - # The aborted fence refuses every future commit, so the retry needs a - # fresh one. Mint it through the host's factory when it has one: hosts - # publish the active fence for hard-interrupt admission, and a /stop - # during the retry must serialize against THIS attempt's commit boundary. + # The aborted fence refuses all commits; mint a fresh one via the host factory + # so a /stop during the retry serializes against THIS attempt's commit boundary. retry_fence = None if new_fence is not None: try: @@ -1460,51 +1141,12 @@ def run_compress_context_with_progress_timeout( stall_fallback: bool = True, new_fence: Optional[Callable[[], CompressionCommitFence]] = None, ) -> Tuple[list, str]: - """Run ``worker(fence)`` under a sync progress-aware timeout. + """Run ``worker(fence)`` under a sync progress-aware (idle + ceiling) timeout. - The idle budget is inactivity-based (same idea as gateway session hygiene): - streamed summary progress via :meth:`CompressionCommitFence.touch_progress` - extends the wait. A hard ceiling still bounds a degenerate trickle stream. - - When cancellation wins before the commit boundary, returns - ``(messages, system_prompt_fallback)`` immediately and leaves the worker - thread detached — the fence prevents a late commit from mutating session - state. When the worker already entered the commit boundary, waits for that - commit to finish and returns its result. - - Timeout budgets (``idle_timeout_seconds`` / ``total_ceiling_seconds``) cover - the **pre-commit** wait only — the summary / stream phase before - :meth:`CompressionCommitFence.begin_commit`. Once the worker holds the - commit fence, SessionDB mutation is already in flight and cannot be safely - abandoned without risking transcript divergence; the commit is therefore - always allowed to complete. The commit-phase wait is still *bounded in - increments* against the remaining total ceiling: if the commit runs past - ``total_ceiling_seconds``, the overrun is logged loudly (escalating from - WARNING to ERROR on repeat) and surfaced once via ``on_commit_overrun``, - while the host keeps waiting in bounded slices until the commit finishes. - The documented guarantee is: **summary phase bounded by the ceiling; - commit phase logged + surfaced if it exceeds it** (never silently hung, - never abandoned mid-commit). - - ``system_prompt_fallback`` may be a string or a zero-arg callable resolved - only on the timeout path, so successful compression never pays for (or - fails on) an eager prompt rebuild. ``on_timeout_cause`` receives whether - the total ceiling expired and whether provider progress was observed before - ``on_timeout`` runs, allowing hosts to report the timeout accurately while - preserving the existing three-argument timeout callback contract. - - ``stall_fallback`` (default on) makes an aborted stall attempt the - configured ``auxiliary.compression.fallback_chain`` once — pinned onto a - fresh fence — before degrading to "continue without compression". Nothing - raises out of a silent stall, so the auxiliary client's own fallback - handling (exception-path only) never sees it (#78981). ``on_timeout`` - therefore fires only after that attempt has also failed, which keeps its - cooldown bookkeeping from suppressing the retry it precedes. - - ``new_fence`` mints that retry's fence. Hosts that publish the active - fence for hard-interrupt admission pass a factory that publishes the new - one too, so a ``/stop`` during the retry serializes against the retry's - commit boundary rather than the aborted attempt's. + Budgets bound the PRE-commit phase only: an admitted commit always completes + (overrun logged, surfaced once via ``on_commit_overrun``). A pre-commit cancel + returns ``(messages, system_prompt_fallback)`` (lazy callable), detaching the + worker; a stall first retries the chain once on ``new_fence``, then on_timeout """ if idle_timeout_seconds <= 0: raise ValueError( @@ -1521,18 +1163,13 @@ def run_compress_context_with_progress_timeout( idle = float(idle_timeout_seconds) fence = fence if fence is not None else CompressionCommitFence() fence.set_total_ceiling_seconds(ceiling) - # Sync mirror of gateway session-hygiene's run_in_executor(None, ...) + - # wait_for loop (gateway/run.py): offload compress_context onto the shared - # daemon pool, poll with an inactivity budget + total ceiling, then - # fence-cancel on timeout so a late commit cannot land. Daemon workers - # match tool_executor: a cancelled hung summary must not block process exit. + # Sync mirror of gateway hygiene's run_in_executor + wait_for loop: offload, + # poll idle budget + ceiling, fence-cancel on timeout so no late commit lands. from tools.thread_context import propagate_context_to_thread executor = _get_compress_timeout_executor() - # Bounded admission (#76354 F6): refuse rather than queue when every pool - # slot is occupied. A queued job would silently wait out its whole budget - # without starting and stay eligible to run as a stale cancelled job when - # a worker recovers. Fail fast: continue without compression this cycle. + # Refuse rather than queue when the pool is full: a queued job would wait out + # its budget unstarted and run stale later. Skip compression this cycle. if not _try_admit_compression_job(): logger.warning( "Context compression pool saturated (%d workers busy) — " @@ -1542,9 +1179,8 @@ def run_compress_context_with_progress_timeout( "provider health.", _COMPRESS_EXECUTOR_MAX_WORKERS, ) - # Round-2 #6: saturation refusals must be visible in the same - # telemetry stream as every other failed attempt, or a wedged pool - # looks like compression simply stopped being attempted. + # Saturation refusals must hit the same telemetry stream as other failures, or + # a wedged pool looks like compression simply stopped being attempted. if telemetry_agent is not None: _emit_compression_attempt_telemetry( telemetry_agent, @@ -1556,10 +1192,8 @@ def run_compress_context_with_progress_timeout( return messages, _resolve_fallback_prompt() def _fence_gated_worker(worker_fence: CompressionCommitFence): - # F6: an admitted job can still start after the host stopped waiting - # (worker slot freed late). Check the fence BEFORE any expensive - # summary work so a stale job never burns an LLM call; its return - # value is discarded by the already-departed host. + # An admitted job may start after the host stopped waiting; check the fence + # BEFORE summary work so a stale job never burns an LLM call. if worker_fence.deadline_exceeded: raise concurrent.futures.TimeoutError( "compression deadline expired before worker start" @@ -1582,12 +1216,8 @@ def run_compress_context_with_progress_timeout( raise future.add_done_callback(_release_compression_admission) wait_started = time.monotonic() - # F2: EVERY host unwind (KeyboardInterrupt, task cancellation, unexpected - # exception while waiting) must revoke future commit admission before the - # host resumes, or a detached worker could later commit and mutate durable - # state behind the caller's back. ``handled_exit`` marks the paths that - # settle admission themselves (worker result returned, or fence cancel - # won); everything else revokes in the ``finally``. + # EVERY host unwind must revoke commit admission or a detached worker could + # later mutate durable state; handled_exit marks paths that settle it themselves handled_exit = False try: while True: @@ -1595,10 +1225,8 @@ def run_compress_context_with_progress_timeout( remaining_ceiling = ceiling - waited if remaining_ceiling <= 0: break - # #76354 S3 analogue for this wait: charge the idle budget from - # the LAST PROGRESS event, not from the start of this wait slice. - # Waiting a full ``idle`` after progress that landed early in the - # previous slice would allow silence to approach 2x the budget. + # Charge idle budget from LAST PROGRESS, not slice start, or silence could + # approach 2x the budget. since_progress = fence.seconds_since_progress() wait_slice = min( max(idle - since_progress, 0.005), remaining_ceiling @@ -1634,9 +1262,8 @@ def run_compress_context_with_progress_timeout( time.monotonic() - wait_started >= ceiling or fence.deadline_exceeded ) if total_exhausted: - # A total-ceiling candidate can still be unwinding a healthy - # provider call. Keep its session lease until that worker exits so - # another automatic attempt cannot overlap the unchanged source. + # A total-ceiling candidate may be unwinding a healthy provider call; keep its + # lease until it exits so no other attempt overlaps the unchanged source. fence.retain_compression_lock_until_worker_done() if on_timeout_cause is not None: @@ -1650,43 +1277,28 @@ def run_compress_context_with_progress_timeout( cancelled: Optional[bool] = None while cancelled is None: - # F1: ``begin_commit`` retains the fence lock until - # ``finish_commit``, so a hung commit makes - # ``try_cancel_before_commit`` return None forever. The lock-free - # phase marker breaks the spin so the overrun-warning loop below - # is reachable WHILE the commit is still blocked. + # begin_commit holds the fence lock until finish_commit, so try_cancel spins + # forever on a hung commit; lock-free marker makes the overrun loop reachable. if fence.commit_in_flight: cancelled = False break cancelled = fence.try_cancel_before_commit() if cancelled is None: - # Round-2 #5: the fence is only held transiently here (lock - # setup / cancel admission — an in-flight commit is caught by - # the commit_in_flight check above), but that window rides - # SessionDB write patience and can last seconds. 25ms keeps - # sub-tick latency without a 1kHz spin. + # Fence is held only transiently here, but that window rides SessionDB write + # patience (seconds). 25ms keeps sub-tick latency without a 1kHz spin. time.sleep(0.025) if not cancelled: - # Pre-commit ceiling already elapsed, but begin_commit() won the - # race. Waiting is intentional: SessionDB mutation cannot be - # fence-cancelled. The wait is bounded in increments against the - # remaining ceiling: a commit that overruns total_ceiling_seconds - # is logged loudly and surfaced once (on_commit_overrun), then - # waited on in bounded slices with escalating log level until it - # completes. Guarantee: summary phase bounded by ceiling; commit - # phase logged + surfaced if it exceeds it — never silently hung, - # never abandoned mid-commit. F1: this loop is reachable WHILE - # the commit is blocked (commit_in_flight is lock-free), so the - # warning + on_commit_overrun fire during the hang, not after it. + # begin_commit won the race: SessionDB mutation cannot be fence-cancelled, so + # wait in bounded slices, logging (escalating) + surfacing once via + # on_commit_overrun WHILE the commit hangs. Never silently hung or abandoned. overrun_surfaced = False overrun_reports = 0 while True: waited = time.monotonic() - wait_started remaining = ceiling - waited if remaining <= 0: - # Ceiling breached while the commit is in flight. Wait in - # bounded increments so each overrun window is visible in - # logs rather than one silent unbounded block. + # Bounded increments so each overrun window is visible in logs rather than one + # silent unbounded block. remaining = min( _COMMIT_OVERRUN_WAIT_SLICE_SECONDS, max(ceiling, 0.05), @@ -1720,37 +1332,24 @@ def run_compress_context_with_progress_timeout( handled_exit = True return result except concurrent.futures.TimeoutError: - # Fence progress (commit-phase touch_progress) is - # informative only — the commit must complete regardless; - # loop and re-report with the updated overrun window. + # Commit-phase progress is informative only — the commit must complete; loop + # and re-report with the updated overrun window. continue - # Idle-timeout path: cancellation won before the commit boundary. - # The fence already blocks any future commit; F4 additionally frees - # the timed-out worker's durable lease via the holder-qualified hook - # so a NEW compressor can acquire the lock immediately (no ABA: the - # DB release is holder-scoped). + # Idle-timeout: cancel won pre-commit. Also free the worker's durable lease via + # the holder-qualified hook so a NEW compressor can acquire at once (no ABA). handled_exit = True - # #97488 teardown (total-ceiling path only): give the cancelled - # worker a bounded grace to actually exit before this host moves on. - # The worker checks the poison fence between provider phases, so a - # cooperative worker exits quickly; an uninterruptible provider call - # is orphaned behind the fence after the grace elapses (its late - # result is discarded and cannot touch session state). The - # idle-stall path intentionally skips the join: its worker is by - # definition silent/hung, the stall-fallback retry below needs a - # prompt host return (pinned by the #76354 S3 latency contract), and - # the fence poison + attempt-generation supersession already protect - # state against its late unwind. + # Total-ceiling only: bounded grace for the worker to exit (it checks the fence + # between provider phases; an uninterruptible call is orphaned). Idle-stall + # skips the join: worker is hung, fallback needs a prompt return, fence guards. if total_exhausted: worker_exited = _join_cancelled_worker( future, min(_CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS, ceiling), ) if worker_exited: - # The worker provably exited: no in-flight provider call can - # outlive this attempt, so the total-ceiling lease retention - # is no longer needed and a retry cannot overlap anything. + # Worker provably exited: no provider call can outlive this attempt, so lease + # retention is unneeded and a retry cannot overlap. fence.allow_cancelled_lock_release() else: logger.warning( @@ -1764,10 +1363,8 @@ def run_compress_context_with_progress_timeout( fence.release_cancelled_compression_lock() waited = time.monotonic() - wait_started since_progress = fence.seconds_since_progress() - # The durable lease is free again (above), so a fallback attempt can - # acquire it immediately. Run it BEFORE on_timeout: that callback - # records the summary-failure cooldown, which would make the retry's - # own summary call a no-op. + # Lease is free, so run the fallback BEFORE on_timeout: that callback records + # the summary-failure cooldown, which would no-op the retry's summary call. if stall_fallback: recovered = _retry_compression_on_fallback_chain( worker=worker, @@ -1804,12 +1401,11 @@ def run_compress_context_with_progress_timeout( return messages, _resolve_fallback_prompt() finally: if not handled_exit: - # F2: KeyboardInterrupt / cancellation / any unexpected exception - # while waiting — revoke commit admission (and release the - # worker's durable lease via the holder-qualified hook) before - # the host unwinds, so the detached worker can never publish. + # Any unwind while waiting: revoke commit admission and release the worker's + # lease before the host unwinds, so the detached worker can never publish. fence.revoke_commit_admission() + class CompressionCheckpointUnavailable(RuntimeError): """Raised when required durable pre-compress checkpointing is unavailable.""" @@ -1824,11 +1420,8 @@ def _checkpoint_blocked(reason: str) -> CompressionCheckpointUnavailable: def _lock_api_is_absent_on_session_db(lock_db: Any) -> bool: """Whether the live in-memory SessionDB class structurally predates locks. - In the supported hot-reload skew, this module is new while the already - imported ``hermes_state.SessionDB`` class (and its live instances) is old. - Only that exact class identity may fail open. Proxies, nominal lookalikes, - non-callables, and descriptor failures must fail closed. Static lookup - avoids invoking a present-but-broken descriptor. + Only the exact old ``hermes_state.SessionDB`` class (hot-reload skew) may fail + open; proxies, lookalikes, non-callables and descriptor failures fail closed. """ try: from hermes_state import SessionDB @@ -1924,19 +1517,49 @@ def _emit_compression_attempt_telemetry( logger.debug("failed to emit compression attempt telemetry: %s", exc) +def _existing_system_prompt(agent: Any, system_message: str) -> str: + """Cached system prompt, or a fresh build when nothing is cached (abort paths).""" + existing = getattr(agent, "_cached_system_prompt", None) + if not existing: + existing = agent._build_system_prompt(system_message) + return existing + + +def _emit_aborted_attempt_telemetry( + agent: Any, started_at: float, failure_class: str | None +) -> None: + _emit_compression_attempt_telemetry( + agent, + started_at=started_at, + commit_status="aborted", + split_status="aborted", + failure_class=failure_class, + ) + + +def _restore_messages_snapshot(messages: list, snapshot: Optional[list]) -> None: + """Put the pre-compression deep snapshot back into the live list if it drifted.""" + if snapshot is not None and messages != snapshot: + messages[:] = copy.deepcopy(snapshot) + + +def _restore_prune_rearm_tokens(compressor: Any, snapshot: dict) -> None: + """Restore ONLY the prune runway from the attempt snapshot. + + compress() zeroes it in memory while the durable copy only clears on a + successful commit; a kept transcript keeps its cached prefix, and 0 would let + the next prune break that cache. + """ + if "_proactive_prune_rearm_tokens" in snapshot: + compressor._proactive_prune_rearm_tokens = snapshot["_proactive_prune_rearm_tokens"] + + def compression_skipped_due_to_lock(agent: Any) -> bool: - """Type-pinned read of the #69870 lock-skip signal. + """Type-pinned read of the per-session lock-skip signal. - ``agent._compression_skipped_due_to_lock`` is set by ``compress_context`` - when a compression pass no-ops because another path holds the per-session - compression lock (holder string when the holder was confirmed, ``True`` - otherwise) and cleared to ``None`` at the entry of every call. - - The read MUST be type-pinned (``is True or isinstance(x, str)``), never - bare truthiness: MagicMock test-double agents auto-create truthy - attributes, and a bare ``if getattr(agent, ...)`` would hijack every - mocked agent in sibling suites into the lock-skip branch (the - #69870 × #69840 type-ahead incident). + ``agent._compression_skipped_due_to_lock`` is a holder string or ``True`` when + a pass no-oped because the lock was held, ``None`` otherwise. Pinning avoids + MagicMock auto-attributes hijacking mocked agents into the lock-skip branch. """ _sig = getattr(agent, "_compression_skipped_due_to_lock", None) return _sig is True or isinstance(_sig, str) @@ -1968,10 +1591,8 @@ def _get_context_compression_timeout_state( def reset_context_compression_timeout_outcome(agent: Any) -> None: """Clear the current thread's owned-compression timeout outcome. - The compatibility mirror ``agent._last_compression_timed_out`` is the - simple per-attempt flag landed by #98424 (turn-start preflight fail-closed - boundary); it stays authoritative for minimal/older agent doubles that do - not support ``vars()``. + The ``agent._last_compression_timed_out`` mirror stays authoritative for + minimal agent doubles that do not support ``vars()``. """ locked_state = _get_context_compression_timeout_state(agent, create=True) if locked_state is None or locked_state[1] is None: @@ -1998,10 +1619,8 @@ def mark_context_compression_timed_out(agent: Any) -> None: def context_compression_timed_out(agent: Any) -> bool: """Return whether this thread's owned compression hit its host timeout. - A single agent may receive overlapping automatic/manual entrypoints. The - thread-local outcome prevents one entrypoint's reset from hiding another's - timeout. The attribute fallback supports older/minimal agent doubles. - Every read is type-pinned to avoid MagicMock auto-attributes. + Thread-local so overlapping automatic/manual entrypoints cannot hide each + other's timeout; attribute fallback for minimal doubles; reads type-pinned. """ locked_state = _get_context_compression_timeout_state(agent, create=False) if locked_state is not None: @@ -2017,9 +1636,8 @@ def _automatic_gate_blocked( ) -> bool: """Evaluate the automatic breaker gate, optionally ignoring the cooldown. - Provider-proven overflow recovery (#100661) passes ``bypass_cooldown``; - engines whose gate predates the kwarg (plugins, test doubles) are called - with the legacy no-argument shape. + Engines whose gate predates ``bypass_cooldown`` are called with the legacy + no-argument shape. """ if bypass_cooldown: try: @@ -2032,24 +1650,12 @@ def _automatic_gate_blocked( def compression_blocked_transiently(agent: Any) -> bool: - """Type-pinned read of the transient-block signal (#97488). + """Type-pinned read of the transient-block signal. - ``agent._compression_blocked_transient`` is set by ``compress_context`` - when an automatic pass no-ops because a TRANSIENT compressor guard is - active — a summary-failure cooldown (e.g. one just recorded by the host - ceiling timeout) or a structural no-op backoff — and cleared to ``None`` - at the entry of every call. - - Consumers (the overflow-recovery loops in ``conversation_loop``) must - treat such a no-op as a temporary defer, NOT as evidence the session is - incompressible: counting it toward ``compression_exhausted`` lets a real - upstream ``context_length_exceeded`` auto-reset (wipe) a session whose - compression was merely cooling down (#97488). The permanent - ``ineffective`` breaker intentionally does NOT set this signal — a - genuinely incompressible session must still be able to exhaust. - - Type-pinned for the same reason as :func:`compression_skipped_due_to_lock` - (MagicMock auto-attribute hijack). + Set when an automatic pass no-ops on a TRANSIENT guard (summary-failure + cooldown or structural backoff). Consumers must defer, not count it toward + ``compression_exhausted``, or an overflow auto-reset wipes a session that was + merely cooling down. The permanent ``ineffective`` breaker never sets it. """ _sig = getattr(agent, "_compression_blocked_transient", None) return isinstance(_sig, str) and bool(_sig) @@ -2058,11 +1664,8 @@ def compression_blocked_transiently(agent: Any) -> bool: def _mark_compression_blocked_transient(agent: Any, compressor: Any) -> None: """Publish the transient-block signal when the active guard is transient. - Reads the compressor's own block reason so the transient/permanent - classification lives in one place (``_compression_block_reason``): - ``cooldown:*`` and ``structural_backoff:*`` are timed guards that lapse - on their own; ``ineffective`` is the permanent breaker and stays - unmarked so exhaustion semantics are preserved. + Classification comes from ``_compression_block_reason``: ``cooldown:*`` and + ``structural_backoff:*`` are transient; ``ineffective`` stays unmarked. """ reason_fn = getattr(compressor, "_compression_block_reason", None) reason = None @@ -2095,17 +1698,9 @@ def _adopt_live_compression_child( ) -> Optional[List[Dict[str, Any]]]: """Move a stale compression contender onto the live continuation tip. - Resolve and load first, then mutate the live agent. This ordering keeps the - stale contender fail-closed when lineage is ambiguous or the compacted - handoff cannot be read. - - Resolution uses the canonical transitive walk ``get_compression_tip`` so a - lineage with >=2 compression hops (root -> mid -> tip) recovers to the live - tip — the depth-1 ``find_live_compression_child`` lookup this used to call - finds no live *direct* child in that shape and skipped recovery (#82001). - The tip walk returns the input id when no continuation exists, and a - resolved tip is adopted only while its row is still live — both cases fail - closed exactly as before. + Resolve and load first, then mutate the agent, so ambiguous lineage or an + unreadable handoff fails closed. Uses the transitive ``get_compression_tip`` + walk; a tip is adopted only while its row is still live. """ resolver = getattr(type(session_db), "get_compression_tip", None) row_getter = getattr(type(session_db), "get_session", None) @@ -2196,9 +1791,8 @@ def recover_rotated_compression_session( try: if not _session_was_rotated_by_compression(session_db, session_id): return None - # Rotation publication holds the parent compression lease until the - # child handoff is durable. A concurrent turn waits briefly rather than - # observing the intentional parent-ended/child-empty intermediate state. + # Rotation holds the parent lease until the child handoff is durable; wait + # briefly rather than observe the parent-ended/child-empty intermediate state. holder_getter = getattr(session_db, "get_compression_lock_holder", None) for attempt in range(21): recovered = _adopt_live_compression_child(agent, session_db, session_id) @@ -2241,15 +1835,10 @@ def recover_rotated_compression_session( def _compression_lock_holder(agent: Any) -> str: - """Build a unique holder id for the lock: pid:tid:agent-instance:uuid. + """Build a unique lock holder id: ``pid:tid:agent-instance:uuid``. - The pid+tid prefix lets ops tell crashed/abandoned holders apart from - live ones (expiry-based recovery uses the timestamp, but ``holder`` - is what shows up in diagnostics + log lines). The agent instance id - and a per-acquire uuid disambiguate two co-resident agents on the - same thread (background_review forks run on a worker thread, but - on machines where compression itself dispatches to a thread pool - we want each acquire to be unique). + pid+tid tell crashed holders apart in diagnostics; instance id and per-acquire + uuid disambiguate co-resident agents on one thread or pooled compressions. """ import threading return ( @@ -2271,10 +1860,8 @@ def _supported_compression_kwargs( ) -> dict: """Return only compression kwargs accepted by an engine callable. - Context-engine plugins can outlive additions to the optional host contract. - Inspecting the callable before invoking it keeps those older signatures - compatible without catching an internal ``TypeError`` and executing a - stateful compressor twice. + Inspecting first keeps older plugin signatures compatible without catching + ``TypeError`` and running a stateful compressor twice. """ candidates = { "current_tokens": current_tokens, @@ -2288,9 +1875,8 @@ def _supported_compression_kwargs( try: parameters = inspect.signature(compress_fn).parameters except (TypeError, ValueError): - # ``current_tokens`` has been part of the ContextEngine ABC since its - # introduction. Keep the oldest documented call shape when a C-backed - # or otherwise opaque callable has no inspectable signature. + # current_tokens has always been in the ContextEngine ABC; use the oldest call + # shape when the callable has no inspectable signature. return {"current_tokens": current_tokens} accepts_kwargs = any( @@ -2348,10 +1934,8 @@ class _CompressionActivityHeartbeat: # late stop must not republish agent.compression / "completed". if self._should_suppress(): return - # Terminal completed/failed must reach SessionDB even inside the - # ordinary 60s activity persist window — otherwise durable labels - # stay on "context compression in progress" after /compress (which - # never hits run_conversation's turn-end clear). + # Force persist: /compress never hits run_conversation's turn-end clear, so + # durable labels would stay "in progress" for the 60s persist window. self._touch(desc, force_persist=True) def _fence_cancelled(self) -> bool: @@ -2403,19 +1987,15 @@ class _CompressionActivityHeartbeat: return self._touch("context compression in progress") + def _direct_messages_for_pre_compress_memory(messages: Any) -> list[dict[str, Any]]: """Return direct user/assistant evidence safe for memory checkpointing. - Compression summaries are derivative context, not new source evidence. - Tool rows and system messages are likewise omitted so memory providers - receive one normalized host contract instead of having to infer Hermes - transcript internals independently. Assistant messages that carry both - prose and ``tool_calls`` keep their prose (the ``tool_calls`` payload is - stripped); pure tool-call wrappers without prose are dropped. + Summaries, tool rows and system messages are omitted; assistant prose is kept + with ``tool_calls`` stripped, and pure tool-call wrappers are dropped. """ - # Deferred import: context_compressor imports turn_context, which imports - # this module — a module-level import here would close that cycle - # (upstream introduced the turn_context edge with the api_content work). + # Deferred import: context_compressor → turn_context → this module would form + # an import cycle. from agent.context_compressor import COMPRESSED_SUMMARY_METADATA_KEY direct_messages: list[dict[str, Any]] = [] @@ -2455,11 +2035,8 @@ class _CompressionLockLeaseRefresher: if refresh_interval_seconds is None: refresh_interval_seconds = max(1.0, min(60.0, ttl_seconds / 2.0)) self._refresh_interval_seconds = max(0.1, float(refresh_interval_seconds)) - # Tolerate transient refresh failures for at most one lease's worth of - # time, so the give-up window is genuinely bounded by the TTL the - # acquirer set (a single blip recovers on the next tick; a persistent - # failure stops before the lease could outlive its TTL). Floor of 1 so a - # degenerate interval >= ttl still tolerates one blip. + # Tolerate transient refresh failures for at most one TTL so the lease cannot + # outlive its TTL; floor 1 so interval >= ttl still tolerates one blip. self._max_consecutive_failures = max( 1, int(self._ttl_seconds / self._refresh_interval_seconds) ) @@ -2476,30 +2053,17 @@ class _CompressionLockLeaseRefresher: def stop(self) -> None: self._stop.set() - # join() may time out while the refresher is mid-UPDATE; that's safe — - # it's a daemon thread, and a late refresh on an already-released lock - # matches rowcount 0 (a no-op). stop() returning does not guarantee the - # thread has fully quiesced, only that we've signalled it and waited - # briefly. + # join() timing out mid-UPDATE is safe: daemon thread, and a late refresh on a + # released lock is a rowcount-0 no-op. stop() does not guarantee quiescence. if self._thread.is_alive() and threading.current_thread() is not self._thread: self._thread.join(timeout=1.0) def _run(self) -> None: - # A single falsy refresh must NOT permanently kill the lease: a - # transient DB blip (write contention escaping _execute_write's retry - # budget, a momentary "database is locked") returns False just like a - # genuine lost-ownership, but only the latter should stop the loop. - # Tolerate consecutive failures for at most one lease's worth of time - # (_max_consecutive_failures = ttl / interval), so a one-off blip - # recovers on the next tick while the total give-up window stays bounded - # by the TTL the acquirer set — the lock can never be held past its TTL - # by a stuck refresher. + # A single falsy refresh (transient DB blip) must not kill the lease; only + # ttl/interval consecutive failures do, so a stuck refresher never outlives TTL. consecutive_failures = 0 - # First refresh happens immediately, not one interval late. Everything - # between try_acquire() and start() (the rotation-ownership lookup, the - # durable-breaker re-read, thread startup) is charged against the very - # first lease, so on a short TTL under load the lock could already be - # expired — and reclaimable by a competing path — before tick #1. + # Refresh immediately: work between try_acquire() and start() is charged to the + # first lease, so on a short TTL it could expire before tick #1. first = True while first or not self._stop.wait(self._refresh_interval_seconds): if first: @@ -2529,18 +2093,10 @@ class _CompressionLockLeaseRefresher: def check_compression_model_feasibility(agent: Any) -> None: - """Warn at session start if the auxiliary compression model's context - window is smaller than the main model's compression threshold. + """Warn at session start if the aux compression context is below the threshold. - When the auxiliary model cannot fit the content that needs summarising, - compression will either fail outright (the LLM call errors) or produce - a severely truncated summary. - - Called during ``AIAgent.__init__`` so CLI users see the warning - immediately (via ``_vprint``). The gateway sets ``status_callback`` - *after* construction, so :func:`replay_compression_warning` re-sends - the stored warning through the callback on the first - ``run_conversation()`` call. + Called from ``AIAgent.__init__`` (CLI sees it via ``_vprint``); the gateway + wires ``status_callback`` later, so ``replay_compression_warning`` resends it. """ if not agent.compression_enabled: return @@ -2555,10 +2111,8 @@ def check_compression_model_feasibility(agent: Any) -> None: get_model_context_length, ) - # Best-effort aux provider label for the warning message. The - # configured provider may be "auto", in which case we fall back - # to the client's base_url hostname so the user can still tell - # where the compression model is actually being called. + # Provider may be "auto"; fall back to the client's base_url hostname so the + # user can tell where the compression model is actually called. try: _aux_cfg_provider, _, _, _, _ = _resolve_task_provider_model("compression") except Exception: @@ -2600,13 +2154,8 @@ def check_compression_model_feasibility(agent: Any) -> None: return aux_base_url = str(getattr(client, "base_url", "")) - # ``client.api_key`` may be a callable (Azure Foundry Entra ID - # bearer provider). The context-length resolver chain expects a - # string, but it only needs a key for live catalogue probes - # (provider model lists). For Entra clients the model-metadata - # chain still resolves via models.dev + hardcoded family - # fallbacks, which don't require auth — pass empty string rather - # than minting a bearer JWT just to look up a context length. + # client.api_key may be a callable (Entra bearer); the resolver only needs a key + # for live catalogue probes, so pass "" rather than mint a JWT for a lookup. _raw_aux_key = getattr(client, "api_key", "") aux_api_key = "" if (callable(_raw_aux_key) and not isinstance(_raw_aux_key, str)) else str(_raw_aux_key or "") @@ -2615,19 +2164,14 @@ def check_compression_model_feasibility(agent: Any) -> None: base_url=aux_base_url, api_key=aux_api_key, config_context_length=getattr(agent, "_aux_compression_context_length_config", None), - # Each model must be resolved with its own provider so that - # provider-specific paths (e.g. Bedrock static table, OpenRouter API) - # are invoked for the correct client, not inherited from the main model. + # Resolve each model with its own provider so provider-specific paths (Bedrock + # table, OpenRouter API) hit the correct client, not the main model's. provider=(_aux_cfg_provider if _aux_cfg_provider and _aux_cfg_provider != "auto" else getattr(agent, "provider", "")), custom_providers=agent._custom_providers, ) - # Hard floor: the auxiliary compression model must have at least - # MINIMUM_CONTEXT_LENGTH (64K) tokens of context. The main model - # is already required to meet this floor (checked earlier in - # __init__), so the compression model must too — otherwise it - # cannot summarise a full threshold-sized window of main-model - # content. Mirrors the main-model rejection pattern. + # Aux model must meet MINIMUM_CONTEXT_LENGTH like the main model, else it cannot + # summarise a full threshold-sized window. if aux_context and aux_context < MINIMUM_CONTEXT_LENGTH: raise ValueError( f"Auxiliary compression model {aux_model} has a context " @@ -2642,24 +2186,13 @@ def check_compression_model_feasibility(agent: Any) -> None: threshold = agent.context_compressor.threshold_tokens if aux_context < threshold: - # Auto-correct: lower the live session threshold so - # compression actually works this session. The hard floor - # above guarantees aux_context >= MINIMUM_CONTEXT_LENGTH, - # so the new threshold is always >= 64K. - # - # The compression summariser sends a single user-role - # prompt (no system prompt, no tools) to the aux model, so - # new_threshold == aux_context is safe: the request is - # the raw messages plus a small summarisation instruction. + # Lower the live threshold so compression works this session. The summariser + # sends one user prompt (no system/tools), so threshold == aux_context is safe. old_threshold = threshold new_threshold = aux_context agent.context_compressor.threshold_tokens = new_threshold - # ``tail_token_budget`` is derived from the trigger threshold, not - # directly from the model window. Keep it in lockstep with this - # just-in-time correction exactly as ContextCompressor.update_model() - # does. Leaving the old budget behind can make the tail's 1.5x soft - # ceiling wider than the lowered trigger, so compression preserves - # nearly the entire request and repeatedly re-fires. + # tail_token_budget derives from the threshold; keep it in lockstep (as + # update_model does) or the 1.5x tail ceiling exceeds the trigger and re-fires. summary_target_ratio = getattr( agent.context_compressor, "summary_target_ratio", None ) @@ -2667,26 +2200,17 @@ def check_compression_model_feasibility(agent: Any) -> None: agent.context_compressor.tail_token_budget = int( new_threshold * summary_target_ratio ) - # Keep threshold_percent in sync so future main-model - # context_length changes (update_model) re-derive from a - # sensible number rather than the original too-high value. + # Keep threshold_percent in sync so update_model re-derives from a sensible + # value rather than the original too-high one. main_ctx = agent.context_compressor.context_length if main_ctx: agent.context_compressor.threshold_percent = ( new_threshold / main_ctx ) safe_pct = int((aux_context / main_ctx) * 100) if main_ctx else 50 - # The "lower the threshold" suggestion must survive the built-in - # trigger recomputation (#67422): _effective_threshold_percent() - # raises sub-75% values back up for main windows under 512K, and - # _compute_threshold_tokens() further applies the output-token - # reservation, the 64K floor, and the degenerate-window guard. - # Recommending a value those would override is silently ignored - # and this warning would reappear every session — so mirror the - # compressor's own math and only offer the option when the - # recomputed trigger actually fits the auxiliary model's context. - # External engines own compaction policy (#44439); the built-in - # floor doesn't apply to them, so keep the plain suggestion. + # Mirror the compressor's threshold math (percent floor, output reservation, + # 64K floor): a suggestion it would override is silently ignored and this + # warning reappears every session. External engines own policy: keep it plain. from agent.context_compressor import ContextCompressor as _CC recomputed_threshold = None @@ -2699,11 +2223,8 @@ def check_compression_model_feasibility(agent: Any) -> None: threshold_suggestion_viable = ( recomputed_threshold is None or recomputed_threshold <= aux_context ) - # Build human-readable "model (provider)" labels for both - # the main model and the compression model so users can - # tell at a glance which provider each side is actually - # using. When the configured provider is empty or "auto", - # fall back to the client's base_url hostname. + # "model (provider)" labels for both sides; empty/"auto" provider falls back to + # the client's base_url hostname. _main_model = getattr(agent, "model", "") or "?" _main_provider = getattr(agent, "provider", "") or "" _aux_provider_label = ( @@ -2781,14 +2302,10 @@ def check_compression_model_feasibility(agent: Any) -> None: def replay_compression_warning(agent: Any) -> None: - """Re-send the compression warning through ``status_callback``. + """Re-send the stored compression warning through ``status_callback``. - During ``__init__`` the gateway's ``status_callback`` is not yet - wired, so ``_emit_status`` only reaches ``_vprint`` (CLI). This - method is called once at the start of the first - ``run_conversation()`` — by then the gateway has set the callback, - so every platform (Telegram, Discord, Slack, etc.) receives the - warning. + Called once at the start of the first ``run_conversation()``, when the gateway + callback (absent during ``__init__``) is finally wired. """ msg = getattr(agent, "_compression_warning", None) if msg and agent.status_callback: @@ -2805,25 +2322,10 @@ def conversation_history_after_compression( ) -> Optional[list]: """Return the correct flush baseline after a compression boundary. - Legacy compression rotates to a fresh child session. That child has not - seen the compacted transcript through the normal same-turn flush path yet, - so callers must clear ``conversation_history`` to ``None`` and let the next - persistence call write the whole compacted list. - - In-place compaction is different: ``archive_and_compact()`` has already - soft-archived the previous active rows and inserted ``messages`` as the new - active live transcript under the same session id. If the same agent turn - continues with ``conversation_history=None``, the identity-based flush path - treats those already-persisted compacted dicts as new and appends them a - second time, doubling the active context and retriggering compression. - - A shallow copy is intentional: it captures the current compacted dict - identities as history while allowing later same-turn appends to remain new. - - An aborted or no-op attempt after an earlier in-place compaction must retain - the pre-attempt baseline. Treating all current messages as persisted would - drop any later, unflushed turns on restart; clearing the baseline would - append the already-persisted compacted rows a second time. + Session rotation returns ``None`` so the child gets the full compacted list. + In-place compaction returns a shallow copy of the already-persisted rows (else + the identity flush re-appends them). Aborted/no-op attempts keep the baseline: + marking all persisted drops unflushed turns; clearing re-appends rows. """ if bool(getattr(agent, "_last_compression_attempt_recorded", False)): attempt_in_place = getattr(agent, "_last_compression_attempt_in_place", None) @@ -2871,11 +2373,8 @@ _SYNTHETIC_USER_FLAGS = ( def _is_real_user_message(message: Any) -> bool: """Distinguish human intent from user-role runtime scaffolding. - A compaction summary pinned to ``role="user"`` (the compressor flips the - summary role to preserve alternation when the tail starts with an - assistant message) is scaffolding too: treating it as human intent would - short-circuit anchor restoration with a message the model is explicitly - told NOT to act on. + A compaction summary flipped to ``role="user"`` for alternation is scaffolding + and must not short-circuit anchor restoration. """ if not isinstance(message, dict) or message.get("role") != "user": return False @@ -2894,12 +2393,8 @@ def _is_real_user_message(message: Any) -> bool: def _message_contains_busy_steer(message: Any) -> bool: """Return whether *message* carries a busy-steer marker. - With ``display.busy_input_mode: steer`` the follow-up is embedded as an - out-of-band marker inside a ``role=tool`` result (see - ``agent_runtime_helpers.apply_pending_steer_to_tool_results``). That marker - carries real user intent but lives outside ``role=user``, so the - ``_is_real_user_message`` / ``_transcript_has_real_user_turn`` checks - alone would miss it. + Steer follow-ups live as markers inside ``role=tool`` results, so they carry + user intent that ``_is_real_user_message`` alone would miss. """ text = _message_text(message) if not text: @@ -2950,11 +2445,10 @@ def _extract_steer_text_from_message(message: Any) -> Optional[str]: def _compressed_has_busy_steer(messages: list) -> bool: - """Whether *messages* already carries a steer marker (intent present). + """Whether *messages* already carries a steer marker in a ``role=tool`` row. - Only ``role=tool`` rows count: that is the sole place the runtime ever - delivers a steer, so a compaction summary that merely quotes the marker - text must not be mistaken for live intent. + Only tool rows count, so a summary merely quoting the marker text is not + mistaken for live intent. """ for msg in messages: if not isinstance(msg, dict) or msg.get("role") != "tool": @@ -2967,11 +2461,8 @@ def _compressed_has_busy_steer(messages: list) -> bool: def _strip_stale_todo_snapshot(content: Any) -> Any: """Remove a previously merged todo-snapshot block from message content. - Snapshot merges (see the injection site in ``compress_context``) always - append the block at the end of the trailing user turn, so a surviving - header marks stale todo state from an earlier compaction boundary. - Stripping before re-injection keeps repeated boundaries from - accumulating outdated snapshots (#26981). + Snapshots are appended to the trailing user turn, so a surviving header is + stale; stripping before re-injection prevents accumulation across boundaries. """ from tools.todo_tool import TODO_INJECTION_HEADER @@ -3006,10 +2497,9 @@ def _strip_stale_todo_snapshot(content: Any) -> Any: def _todo_snapshot_is_only_content(content: Any, stripped: Any) -> bool: """Return whether stripping the snapshot leaves no structured content. - Text snapshots are appended at the end of a string. Structured snapshots - occupy their own text part, so only an empty remainder proves that the row - was synthetic scaffolding alone. Text extraction is deliberately not used: - image, audio, and future non-text parts are content that must survive. + Text snapshots trail a string; structured ones occupy their own text part, so + only an empty remainder proves the row was scaffolding alone. Text extraction + is deliberately not used: image, audio and other non-text parts must survive. """ if isinstance(content, str) and isinstance(stripped, str): return not stripped.strip() @@ -3026,28 +2516,19 @@ def _replace_message_content(message: dict, content: Any) -> None: drop_stale_api_content(message) -# Retention-parity notice (#84718): compaction re-injects the todo list -# verbatim while skill instructions are pruned to [SKILL_PRUNED: ...] markers, -# so the imperative crosses the boundary without the policy that governed it. -# When BOTH happen at the same boundary, couple them: the re-injected snapshot -# carries an explicit instruction to reload the pruned skills BEFORE acting on -# any preserved task. Deterministic (derived only from the compressed -# transcript), bounded (marker cap shared with the summary re-injection), and -# stripped together with the snapshot at the next boundary because it lives -# after TODO_INJECTION_HEADER inside the same block. +# Compaction re-injects the todo list verbatim but prunes skills to markers, so +# couple them: tell the model to reload pruned skills BEFORE acting on tasks. +# Lives after TODO_INJECTION_HEADER so it strips with the snapshot next time. _PRUNED_SKILL_RELOAD_NOTICE_HEADER = ( "[Skills pruned during compression — reload before acting on these tasks]" ) def _pruned_skill_reload_notice(compressed: list) -> str: - """Reload instruction for skills whose bodies were pruned, or ``""``. + """Reload notice for skills whose bodies were pruned, or ``""``. - Scans the post-compression transcript for the canonical - ``[SKILL_PRUNED: ...]`` markers (summary ``## Pruned Skills`` section, - pruned tool rows surviving in the protected tail) and renders one bounded - notice naming each skill with its exact ``skill_view`` reload call. - First-seen order, deduplicated, capped at ``_MAX_PRUNED_SKILL_MARKERS``. + Scans ``[SKILL_PRUNED: ...]`` markers in the post-compression transcript; + first-seen order, deduplicated, capped at ``_MAX_PRUNED_SKILL_MARKERS``. """ from agent.context_compressor import ( _MAX_PRUNED_SKILL_MARKERS, @@ -3079,10 +2560,8 @@ def _pruned_skill_reload_notice(compressed: list) -> str: def _merge_anchor_into_user_message(target: dict, anchor: dict) -> None: """Fold the human anchor into an existing user-role scaffolding turn. - Used only when every insertion slot would create two consecutive - user-role messages. The anchor text leads (it is the active task), the - scaffolding content is preserved after it, and the synthetic flags are - cleared because the merged turn now carries real human intent. + Used only when any insertion would create consecutive user turns. Anchor text + leads, scaffolding follows, and synthetic flags are cleared. """ anchor_content = anchor.get("content") target_content = target.get("content") @@ -3120,9 +2599,8 @@ def _insert_real_user_anchor(messages: list, anchor: dict) -> CompressedUserTurn def _role(msg: Any) -> Optional[str]: return msg.get("role") if isinstance(msg, dict) else None - # Preferred: the summary boundary — before the first assistant message - # not already preceded by a user turn. The left neighbour is then - # non-user by construction and the right neighbour is an assistant. + # Preferred anchor: the summary boundary — first assistant message not preceded + # by a user turn. Left neighbour is then non-user, right is an assistant. for index, message in enumerate(messages): if _role(message) != "assistant": continue @@ -3144,12 +2622,8 @@ def _insert_real_user_anchor(messages: list, anchor: dict) -> CompressedUserTurn if ContextCompressor._is_context_summary_content( _message_text(messages[-1]) ): - # Never merge into a compaction summary: the summary prefix must - # stay at the start of its message for downstream summary detection. - # Appending after it makes the anchor "the latest user message after - # the summary" — exactly what the handoff prefix instructs — and the - # adjacent user turns are merged summary-first by - # repair_message_sequence before the next API call. + # Never merge into a summary: its prefix must stay at message start for summary + # detection; repair_message_sequence merges adjacent user turns summary-first. anchor[_DB_PERSISTED_MARKER] = True messages.append(anchor) return "inserted" @@ -3173,13 +2647,8 @@ def _ensure_compressed_has_user_turn( _fresh_compaction_message_copy, ) - # One reversed positional scan: the anchor is whichever intent-bearing - # row is LAST in the original transcript — a real ``role=user`` turn or - # a steer marker riding inside a ``role=tool`` result. Scanning the two - # kinds separately (steer first, then user) would let an older, already - # consumed steer outrank a newer real user request and replay it - # (#100053 follow-up: ``[user A, tool(steer B), ..., user C]`` must - # anchor C, not B). + # One reversed scan over BOTH kinds: scanning steer then user would let an older + # consumed steer outrank a newer real user request and replay it. for message in reversed(original_messages): if _is_real_user_message(message): return _insert_real_user_anchor( @@ -3233,10 +2702,8 @@ def _notify_context_engine_compression_complete( old_session_id: str, ) -> bool: """Notify the active context engine after a durable compression commit.""" - # Relay session-span segmentation (opt-in, gateway.telemetry. - # session_segments.on_compaction): flag the session so its telemetry - # scope rotates at the next turn boundary. Observer semantics — a - # failure here must never undo or delay the committed compression. + # Opt-in relay session-span segmentation. Observer semantics — failure must + # never undo or delay the committed compression. try: from agent import relay_runtime @@ -3302,6 +2769,1350 @@ def finalize_context_engine_compression_notification( return bool(pending()) +class _CompactionLifecycle: + """Owns the one-shot terminal edge of the compaction status lifecycle. + + ``commit_status`` is rebound to "committed" only on success and read at + ``complete()`` time, so abort paths keep the terminal edge suppressed. + """ + + def __init__(self, agent: Any, status_emitted: bool) -> None: + self._agent = agent + self._status_emitted = status_emitted + self._done_emitted = False + self.commit_status = "aborted" + + def complete(self, *, force_terminal: bool = False) -> None: + if self._done_emitted: + return + self._done_emitted = True + # Suppressed start → no terminal edge. Non-compacting aborts (lock contender, + # cancelled fence) opt in via force_terminal so clients can retire their phase. + # Failure warnings go through _emit_warning and are never suppressed here. + if self._status_emitted and ( + self.commit_status == "committed" or force_terminal + ): + _emit_compaction_done(self._agent) + + +class _CompressionLease: + """The per-attempt durable compression lock plus its lifecycle plumbing. + + ``holder`` is None when no durable lock is owned (legacy DB, no session db); + ``watermark`` is MAX(id) of active rows at lease start (None = archive + everything, no concurrent-tail preservation this cycle). + """ + + def __init__( + self, + agent: Any, + *, + db: Any, + sid: str, + ttl: float, + refresh_interval: Any, + commit_fence: Optional[CompressionCommitFence], + lifecycle: _CompactionLifecycle, + ) -> None: + self._agent = agent + self.db = db + self.sid = sid + self.ttl = ttl + self._refresh_interval = refresh_interval + self._commit_fence = commit_fence + self._lifecycle = lifecycle + self.holder: Optional[str] = None + self.watermark: Optional[int] = None + self._refresher: Optional[_CompressionLockLeaseRefresher] = None + self._released = False + self._release_guard = threading.Lock() + # Fence lock acquisition + release-hook publication together so a host timeout + # cannot win between acquiring the lock and having a way to release it. + self._lock_setup_entered = False + + def begin_lock_setup(self) -> bool: + if self._commit_fence is None: + return True + self._lock_setup_entered = self._commit_fence.begin_lock_setup() + return self._lock_setup_entered + + def finish_lock_setup(self) -> None: + if not self._lock_setup_entered or self._commit_fence is None: + return + self._lock_setup_entered = False + self._commit_fence.finish_lock_setup() + + def start_refresher(self) -> None: + if self.holder is None: + return + candidate = _CompressionLockLeaseRefresher( + self.db, self.sid, self.holder, self.ttl, self._refresh_interval + ) + # Cancellation may release the holder between hook publication and this + # start; serialize with the release path so no refresher starts on a freed lock. + with self._release_guard: + if not self._released: + self._refresher = candidate + self._refresher.start() + + def release_holder_only(self) -> None: + """Stop this holder's refresher and release only its durable lock. + + Holder-qualified and idempotent: safe for the host after a timeout because a + newer holder's lease can never be deleted by this stale release. + """ + with self._release_guard: + if self._released: + return + self._released = True + if getattr(self._agent, "_active_compression_lock_holder", None) == self.holder: + self._agent._active_compression_lock_holder = None + if self._refresher is not None: + try: + self._refresher.stop() + except Exception as _stop_err: + logger.debug("compression lock refresher stop failed: %s", _stop_err) + if self.db is not None and self.sid and self.holder: + try: + self.db.release_compression_lock(self.sid, self.holder) + except Exception as _rel_err: + logger.debug("compression lock release failed: %s", _rel_err) + + def release(self) -> None: + """Finish lifecycle cleanup and release the OLD session lock once.""" + try: + self._lifecycle.complete() + finally: + try: + self.release_holder_only() + finally: + try: + if self._commit_fence is not None: + self._commit_fence.clear_cancelled_lock_release( + self.release_holder_only + ) + finally: + self.finish_lock_setup() + + +def _acquire_compression_lease( + agent: Any, + *, + commit_fence: Optional[CompressionCommitFence], + lifecycle: _CompactionLifecycle, + system_message: str, + approx_tokens: Optional[int], + attempt_started_at: float, +) -> Tuple[Optional[_CompressionLease], Optional[str]]: + """Take the per-session compression lock; ``(None, prompt)`` means sit out. + + Two AIAgents sharing a session_id (e.g. background review fork) would both + rotate and orphan a child. Keyed on the OLD id (what rivals read from + SessionEntry). Loser sits out: messages unchanged, caller sees no-op. + Only structural absence of the lock API (version skew) fails open; once + resolved, any exception fails closed since unlocked runs can fork lineage. + """ + _lock_db = getattr(agent, "_session_db", None) + _lock_sid = agent.session_id or "" + _try_acquire_lock = None + _lock_lookup_error: Optional[Exception] = None + _legacy_session_db_without_lock_api = False + # Clear stale lock-skip so this call's outcome alone is visible; else a manual + # /compress after an auto lock-skip falsely reports "already in progress". + agent._compression_skipped_due_to_lock = None + if _lock_db is not None: + try: + _legacy_session_db_without_lock_api = _lock_api_is_absent_on_session_db( + _lock_db + ) + except Exception as exc: + _lock_lookup_error = exc + if _lock_lookup_error is None and not _legacy_session_db_without_lock_api: + try: + _try_acquire_lock = _lock_db.try_acquire_compression_lock + if not callable(_try_acquire_lock): + _lock_lookup_error = TypeError( + "compression lock API is present but not callable" + ) + except Exception as exc: + _lock_lookup_error = exc + try: + _lock_ttl = float(getattr(agent, "_compression_lock_ttl_seconds", 300.0) or 300.0) + except (TypeError, ValueError): + _lock_ttl = 300.0 + lease = _CompressionLease( + agent, + db=_lock_db, + sid=_lock_sid, + ttl=_lock_ttl, + refresh_interval=getattr(agent, "_compression_lock_refresh_interval", None), + commit_fence=commit_fence, + lifecycle=lifecycle, + ) + + if _lock_db is not None and _lock_sid: + lease.holder = _compression_lock_holder(agent) + if _lock_lookup_error is not None: + # Attribute lookup itself failed for a reason other than a missing + # lock API. It is unsafe to proceed without a lock in that case. + lease.holder = None + logger.warning( + "compression lock lookup raised unexpectedly for session=%s " + "(%s: %s) — skipping compression this cycle", + _lock_sid, type(_lock_lookup_error).__name__, _lock_lookup_error, + ) + _lock_acquired = False + elif _try_acquire_lock is None: + # Lock API absent on this instance: log once, proceed unlocked so version skew + # cannot stall the outer auto-compression loop forever. + lease.holder = None + if getattr(agent, "_last_compression_lock_error_sid", None) != _lock_sid: + agent._last_compression_lock_error_sid = _lock_sid + logger.warning( + "compression lock subsystem unavailable for session=%s " + "— proceeding without lock. This usually means a stale " + "in-memory module after an update; restart the process " + "(or `hermes update`) to resync.", + _lock_sid, + ) + _lock_acquired = True # acquired-but-unlocked compatibility path + else: + if not lease.begin_lock_setup(): + logger.info( + "Compression commit cancelled before lock acquisition " + "(session=%s).", + agent.session_id or "none", + ) + agent._last_compaction_in_place = False + _existing_sp = _existing_system_prompt(agent, system_message) + _emit_aborted_attempt_telemetry(agent, attempt_started_at, "commit_fence_cancelled") + lifecycle.complete(force_terminal=True) + return None, _existing_sp + try: + _lock_acquired = _try_acquire_lock( + _lock_sid, lease.holder, ttl_seconds=_lock_ttl + ) + if _lock_acquired: + # Watermark = MAX(id) of active rows at START. Appends aren't blocked during + # summary; later rows are concurrent tail that archive_and_compact re-sequences. + try: + lease.watermark = _lock_db.get_active_message_watermark( + _lock_sid + ) + # A captured watermark makes the commit safe against later rows on BOTH commit + # paths; tell the fence so a host may keep this attempt's admission. + if commit_fence is not None: + try: + commit_fence.mark_commit_watermark_fenced() + except AttributeError: + pass # test doubles without the method + except Exception as _wm_err: + # Watermark capture is safety-additive (fallback archives everything), so + # failure here must not abort compression. + logger.warning( + "compression watermark capture failed for " + "session=%s (%s) — concurrent appends this cycle " + "will be archived with the snapshot", + _lock_sid, _wm_err, + ) + lease.watermark = None + except Exception as _lock_err: + # Method entered but failed: not version skew, fail closed. Acquire may have + # committed, so release holder-qualified best-effort (safe if never acquired). + try: + _lock_db.release_compression_lock(_lock_sid, lease.holder) + except Exception as _release_err: + logger.debug( + "compression lock cleanup after failed acquire failed: %s", + _release_err, + ) + lease.holder = None + logger.warning( + "compression lock acquisition raised unexpectedly for " + "session=%s (%s: %s) — skipping compression this cycle", + _lock_sid, type(_lock_err).__name__, _lock_err, + ) + _lock_acquired = False + if not _lock_acquired: + lease.finish_lock_setup() + try: + existing = _lock_db.get_compression_lock_holder(_lock_sid) + except Exception: + existing = None + logger.warning( + "compression skipped: another path is compressing session=%s " + "(holder=%s) — returning messages unchanged to avoid session fork", + _lock_sid, existing, + ) + lease.holder = None # don't release a lock we don't own + # Distinguish lock-contention no-op from "nothing to compress" so manual + # /compress can show a clear status instead of "No changes". + agent._compression_skipped_due_to_lock = existing or True + # Surface to the user once — quiet for downstream auto-compress loops + if getattr(agent, "_last_compression_lock_warning_sid", None) != _lock_sid: + agent._last_compression_lock_warning_sid = _lock_sid + try: + agent._emit_warning( + "⚠ Skipping concurrent compression — another path " + "is already compressing this session. Will retry " + "after it finishes." + ) + except Exception: + pass + _existing_sp = _existing_system_prompt(agent, system_message) + try: + if hasattr(agent.context_compressor, "_begin_compression_telemetry"): + agent.context_compressor._begin_compression_telemetry(current_tokens=approx_tokens) + except Exception: + pass + _emit_aborted_attempt_telemetry(agent, attempt_started_at, "lock_contended") + lifecycle.complete(force_terminal=True) + return None, _existing_sp + + if lease.holder is not None: + agent._active_compression_lock_holder = lease.holder + if ( + commit_fence is not None + and commit_fence.register_cancelled_lock_release( + lease.release_holder_only + ) + ): + # Cancellation won during lock setup (hook ran synchronously, lease gone): + # abort before any summary work. + logger.info( + "Compression commit cancelled before summary dispatch " + "(session=%s).", + agent.session_id or "none", + ) + agent._last_compaction_in_place = False + _existing_sp = _existing_system_prompt(agent, system_message) + _emit_aborted_attempt_telemetry(agent, attempt_started_at, "commit_fence_cancelled") + lease.release() + return None, _existing_sp + return lease, None + + +def _adopt_if_parent_rotated( + agent: Any, lease: _CompressionLease, messages: list, system_message: str +) -> Optional[Tuple[list, str]]: + """Sit out (or adopt the live child) when the parent was already rotated. + + A late contender can take the parent lock after the winner released it and + rotated; holding the lock does not prove this agent still owns a live parent. + Returns the ``compress_context`` result to hand back, or None to proceed. + """ + if lease.db is None or not lease.sid: + return None + try: + _parent_already_rotated = _session_was_rotated_by_compression( + lease.db, lease.sid + ) + except Exception as _session_err: + logger.warning( + "compression session ownership lookup failed for session=%s " + "(%s: %s) - skipping compression this cycle", + lease.sid, + type(_session_err).__name__, + _session_err, + ) + lease.release() + return messages, _existing_system_prompt(agent, system_message) + if not _parent_already_rotated: + return None + recovered_messages = _adopt_live_compression_child(agent, lease.db, lease.sid) + lease.release() + _existing_sp = _existing_system_prompt(agent, system_message) + if recovered_messages is not None: + logger.warning( + "compression recovery: stale session=%s adopted live child=%s", + lease.sid, + agent.session_id, + ) + return recovered_messages, _existing_sp + logger.warning( + "compression skipped: session=%s was already rotated by " + "another compression path, but no unique live child could be adopted", + lease.sid, + ) + return messages, _existing_sp + + +def _adopt_grown_durable_parent( + agent: Any, lease: _CompressionLease, messages: list +) -> Optional[list]: + """Return the durable parent transcript when it outgrew the in-memory snapshot. + + Rotation only (in-place never loses rows). The snapshot predates the lease: if + durable grew, a writer committed a turn — ADOPT it (aborting wedged busy + sessions forever). Length check only: in-memory edits of past turns are legal. + """ + if lease.db is None or not lease.sid: + return None + durable_loader = getattr(type(lease.db), "get_messages_as_conversation", None) + if not callable(durable_loader): + return None + durable_parent = durable_loader(lease.db, lease.sid) + if not (isinstance(durable_parent, list) and len(durable_parent) > len(messages)): + return None + # In-memory carries this turn's un-persisted user tail; flush it via the normal + # rotation-boundary path before adopting, else skip adoption (would drop input). + _preflush_idx = getattr(agent, "_persist_user_message_idx", None) + _preflush_ok = False + if isinstance(_preflush_idx, int) and 0 <= _preflush_idx < len(messages): + try: + _preflush_ok = agent._flush_messages_to_session_db( + messages, + conversation_history=messages[:_preflush_idx], + ) + except Exception: + _preflush_ok = False + else: + # No un-persisted tail: transcript is fully durable, so adopting the longer + # parent cannot drop live input — adopt directly. + _preflush_ok = True + if not _preflush_ok: + logger.warning( + "compression: session=%s grew before lease " + "(%d → %d msgs) but the pre-adoption flush of the " + "live tail failed; skipping durable-snapshot " + "adoption so un-persisted user input is kept", + lease.sid, + len(messages), + len(durable_parent), + ) + return None + # Re-read after the flush so the adopted snapshot carries the just-persisted tail. + durable_parent = durable_loader(lease.db, lease.sid) + if not (isinstance(durable_parent, list) and len(durable_parent) > len(messages)): + return None + logger.info( + "compression: session=%s grew before lease " + "(%d → %d msgs); adopting durable snapshot", + lease.sid, + len(messages), + len(durable_parent), + ) + return durable_parent + + +def _pre_compress_memory_context( + agent: Any, messages: list, checkpoint_required: bool +) -> str: + """Provider ``on_pre_compress()`` insights to surface in the summary ("" if none). + + Raw messages stay the API v1 provider contract; normalized evidence goes only + to API v2+ checkpoint providers inside MemoryManager.on_pre_compress(). + Raises :class:`CompressionCheckpointUnavailable` when a required checkpoint + cannot be taken. + """ + memory_context = "" + memory_manager = getattr(agent, "_memory_manager", None) + evidence_messages = _direct_messages_for_pre_compress_memory(messages) + if checkpoint_required: + supports_checkpoint = getattr( + memory_manager, "supports_pre_compress_checkpoint", None + ) + if memory_manager is None or not callable(supports_checkpoint): + raise _checkpoint_blocked( + f"no active provider implements checkpoint API " + f"v{PRE_COMPRESS_CHECKPOINT_API_VERSION}" + ) + try: + compatible = bool( + supports_checkpoint(PRE_COMPRESS_CHECKPOINT_API_VERSION) + ) + except Exception as exc: + raise _checkpoint_blocked("provider capability probe failed") from exc + if not compatible: + raise _checkpoint_blocked( + f"active provider does not implement checkpoint API " + f"v{PRE_COMPRESS_CHECKPOINT_API_VERSION}" + ) + try: + _maybe_ctx = memory_manager.on_pre_compress( + messages, + evidence_messages=evidence_messages, + require_checkpoint=True, + checkpoint_api_version=PRE_COMPRESS_CHECKPOINT_API_VERSION, + ) + except Exception as exc: + logger.warning( + "Required pre-compress checkpoint failed (%s)", + type(exc).__name__, + ) + raise _checkpoint_blocked( + f"provider checkpoint API v{PRE_COMPRESS_CHECKPOINT_API_VERSION} failed" + ) from exc + if isinstance(_maybe_ctx, str): + memory_context = sanitize_memory_context(_maybe_ctx) + elif memory_manager: + try: + _maybe_ctx = memory_manager.on_pre_compress( + messages, evidence_messages=evidence_messages + ) + if isinstance(_maybe_ctx, str): + memory_context = sanitize_memory_context(_maybe_ctx) + except Exception: + pass + return memory_context + + +def _resolve_compress_call( + agent: Any, + *, + approx_tokens: Optional[int], + focus_topic: Optional[str], + force: bool, + memory_context: str, + bypass_cooldown: bool, +) -> Tuple[Callable[..., Any], dict[str, Any]]: + """Bind ``compress()`` and only the kwargs its signature accepts.""" + compress_fn = agent.context_compressor.compress + compress_kwargs = _supported_compression_kwargs( + compress_fn, + current_tokens=approx_tokens, + focus_topic=focus_topic, + force=force, + memory_context=memory_context, + bypass_cooldown=bypass_cooldown, + ) + if memory_context.strip() and "memory_context" not in compress_kwargs: + engine_name = getattr( + agent.context_compressor, + "name", + type(agent.context_compressor).__name__, + ) + if ( + getattr(agent, "_last_memory_context_unsupported_engine", None) + != engine_name + ): + agent._last_memory_context_unsupported_engine = engine_name + logger.warning( + "context engine %s does not accept memory_context; continuing " + "without provider-supplied summary context", + engine_name, + ) + return compress_fn, compress_kwargs + + +def _run_summary_dispatch( + agent: Any, + messages: list, + compress_fn: Callable[..., Any], + compress_kwargs: dict[str, Any], + *, + commit_fence: Optional[CompressionCommitFence], + attempt_generation: Any, + hard_cancel_event: Any, +) -> list: + """Run the compressor under the fence's progress hook, deadline and interrupt guard.""" + # Publish progress to the commit fence so hosts extend deadlines while tokens + # flow. Any active hook (even no-op) selects the streamed path: the timeout is + # inactivity-based and a byte-trickling provider hits the stream total ceiling. + from agent.auxiliary_client import ( + aux_interrupt_protection, + aux_progress_hook, + aux_stream_deadline, + ) + _progress_hook = ( + commit_fence.touch_progress if commit_fence is not None + else (lambda: None) + ) + # Return leg: cancel frees the owner but the provider daemon streams on to its + # own larger ceiling; share the host deadline so orphan streams stop with it. + _host_stream_deadline = ( + commit_fence.deadline_monotonic if commit_fence is not None else None + ) + # A LATE successful summary must not undo the host's timeout cooldown: the + # compressor checks cancellation before clearing; removed in finally (no leak). + if commit_fence is not None: + _install_compression_cancelled_check( + agent.context_compressor, + lambda: commit_fence.is_cancelled, + attempt_generation, + ) + + def _compression_cancel_requested() -> bool: + return bool( + ( + hard_cancel_event is not None + and hard_cancel_event.is_set() + ) + or ( + commit_fence is not None + and commit_fence.is_cancelled + ) + ) + + try: + # F6: never start expensive summary work for an already-cancelled + # fence (a stale queued job admitted after host departure). + if commit_fence is not None and commit_fence.is_cancelled: + logger.info( + "Compression cancelled before summary dispatch " + "(session=%s) — skipping summary work.", + agent.session_id or "none", + ) + compressed = messages + else: + with aux_progress_hook(_progress_hook), aux_stream_deadline( + _host_stream_deadline + ), aux_interrupt_protection( + cancel_check=_compression_cancel_requested + ): + compressed = compress_fn(messages, **compress_kwargs) + # Freeze a hard stop that arrived after the last provider attempt but before + # session state rotates. + if ( + hard_cancel_event is not None + and hard_cancel_event.is_set() + ): + raise AuxiliaryExplicitCancellation() + finally: + if commit_fence is not None: + _clear_compression_cancelled_check_if_owner( + agent.context_compressor, attempt_generation + ) + return compressed + + +def _fold_todo_snapshot(agent: Any, compressed: list) -> None: + """Strip stale todo snapshots from ``compressed`` and fold the live one in (in place).""" + todo_snapshot = agent._todo_store.format_for_injection() + # Non-empty store (even all done) is authoritative: drop the old snapshot. A + # truly empty store may be un-rehydrated post-compaction: keep the snapshot. + _todo_has_items = getattr(agent._todo_store, "has_items", None) + try: + _todo_store_is_authoritative = bool( + _todo_has_items() + ) if callable(_todo_has_items) else False + except Exception: + # Store may implement only format_for_injection(); unknown authority must + # preserve the pending snapshot rather than risk deleting it. + _todo_store_is_authoritative = False + if _todo_store_is_authoritative: + for _todo_idx in range(len(compressed) - 1, -1, -1): + _todo_message = compressed[_todo_idx] + if not isinstance(_todo_message, dict) or _todo_message.get("role") != "user": + continue + _todo_content = _todo_message.get("content") + _todo_stripped = _strip_stale_todo_snapshot(_todo_content) + if _todo_stripped == _todo_content: + continue + if ( + _todo_message.get("_todo_snapshot_synthetic") + and _todo_snapshot_is_only_content( + _todo_content, _todo_stripped + ) + ): + compressed.pop(_todo_idx) + if _todo_idx < len(compressed): + # A standalone snapshot can drift from the tail; deleting it may expose two + # assistant rows, so use the normal replay repair to keep metadata consistent. + agent._repair_message_sequence(compressed) + else: + _replace_message_content(_todo_message, _todo_stripped) + # No longer todo-only scaffolding; other synthetic flags stay authoritative and + # _is_real_user_message() recomputes provenance from content + flags. + _todo_message.pop("_todo_snapshot_synthetic", None) + break + if todo_snapshot: + # If this boundary pruned skill bodies, the policy behind the todos is gone: + # add a reload notice after TODO_INJECTION_HEADER so both strip together. + _reload_notice = _pruned_skill_reload_notice(compressed) + if _reload_notice: + todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}" + # Fold the snapshot into a trailing REAL user msg (no synthetic user/user pair); + # strip old snapshots first. Scaffolding tails must not absorb it (provenance). + from agent.context_compressor import _append_text_to_content + + merged = False + _tail = ( + compressed[-1] + if compressed and isinstance(compressed[-1], dict) + else None + ) + if _tail is not None and _tail.get("role") == "user": + _stripped = _strip_stale_todo_snapshot(_tail.get("content")) + _probe = { + key: value for key, value in _tail.items() if key != "content" + } + _probe["content"] = _stripped + if _is_real_user_message(_probe): + _snapshot_text = ( + f"\n\n{todo_snapshot}" + if isinstance(_stripped, str) and _stripped + else todo_snapshot + ) + _replace_message_content( + _tail, + _append_text_to_content(_stripped, _snapshot_text), + ) + merged = True + elif _stripped != _tail.get("content") and not _message_text( + {"role": "user", "content": _stripped} + ).strip(): + # The tail was nothing but an earlier snapshot row — + # refresh it in place instead of stacking a duplicate. + _replace_message_content(_tail, todo_snapshot) + _tail["_todo_snapshot_synthetic"] = True + merged = True + if not merged: + compressed.append({ + "role": "user", + "content": todo_snapshot, + "_todo_snapshot_synthetic": True, + }) + + +def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str: + """Refresh tool schemas and rebuild the system prompt at the commit boundary.""" + cached_system_prompt = agent._cached_system_prompt + agent._invalidate_system_prompt() + + # Refresh tool schemas at the commit boundary: forever-sessions never restart, + # so config reaches agent.tools here. Keep list identity if byte-equal (cache). + try: + _refresh_agent_tool_definitions(agent) + except Exception: # noqa: BLE001 + logger.warning( + "compaction tool-definition refresh failed; keeping the " + "session's existing tool snapshot", + exc_info=True, + ) + + # ALWAYS rebuild the prompt here: keeping old bytes meant prompt-builder changes + # never reached long sessions. Equal bytes keep KV; preserve object identity. + rebuilt_system_prompt = agent._build_system_prompt(system_message) + if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt: + new_system_prompt = cached_system_prompt + agent._cached_system_prompt = cached_system_prompt + from agent.system_prompt import reconstruct_static_prefix + + reconstruct_static_prefix( + agent, + system_message=system_message, + log_label="compression keep-prompt", + ) + else: + new_system_prompt = rebuilt_system_prompt + agent._cached_system_prompt = new_system_prompt + if cached_system_prompt is not None: + logger.info( + "Compaction rebuilt a drifted system prompt " + "(session=%s, %d -> %d chars): builder output changed " + "since the stored snapshot (update, config change, or " + "memory/skills growth)", + agent.session_id or "none", + len(cached_system_prompt), + len(new_system_prompt), + ) + return new_system_prompt + + +def _salvage_or_refuse_grown_transcript( + agent: Any, + messages: list, + compressed: list, + *, + system_message: str, + attempt_started_at: float, + attempt_snapshot: dict, +) -> Tuple[Optional[list], Optional[str]]: + """Anti-growth guard at the COMMIT SITE (in-place commits before the gateway can inspect). + + Compares like-for-like rough estimates; on growth tries one mechanical salvage + pass, else treats the attempt as a refused no-op. Returns ``(compressed, None)`` + to proceed or ``(None, prompt)`` when refused (caller releases the lease). + """ + # Anti-growth guard at the COMMIT SITE: in-place commits here before the gateway + # can inspect. Compare like-for-like rough estimates; on growth treat as no-op. + _rough_in = estimate_messages_tokens_rough(messages) + _rough_out = estimate_messages_tokens_rough(compressed) + if _rough_out > _rough_in: + # Todo refresh and user-turn anchoring run after the compressor's own size check + # and can tip a break-even candidate; give it one mechanical salvage pass. + from agent.context_compressor import salvage_grown_transcript + + _salvaged = salvage_grown_transcript( + messages, compressed, budget=_rough_in + ) + if _salvaged is not None: + _salv_est = estimate_messages_tokens_rough(_salvaged) + if _salv_est < _rough_in: + logger.info( + "Compression salvage recovered a shrinking " + "transcript (session=%s, ~%s -> ~%s tokens)", + agent.session_id or "none", + f"{_rough_in:,}", + f"{_salv_est:,}", + ) + compressed = _salvaged + _rough_out = _salv_est + if _rough_out > _rough_in: + logger.warning( + "Compression refused: compressed transcript would be " + "larger than the original (session=%s, ~%s -> ~%s " + "tokens); keeping the original transcript unchanged", + agent.session_id or "none", + f"{_rough_in:,}", + f"{_rough_out:,}", + ) + # Flag the refusal on compressor state so /compress feedback reports it instead + # of comparing list lengths (adoption can change the count), claiming success. + try: + agent.context_compressor._last_compress_refused_would_grow = True + except Exception: + pass + try: + agent._emit_warning( + "⚠️ Compression refused: the generated summary " + "would have GROWN the conversation instead of " + "shrinking it. No messages were dropped — " + "conversation continues unchanged." + ) + except Exception: + pass + _existing_sp = _existing_system_prompt(agent, system_message) + _emit_aborted_attempt_telemetry(agent, attempt_started_at, "would_grow") + # Count the refusal as an ineffective-compaction strike so the anti-thrash + # breaker latches; otherwise auto-compress retries the same summary every turn. + try: + agent.context_compressor.record_rejected_compaction() + except Exception: + logger.debug( + "could not record rejected-compaction strike", + exc_info=True, + ) + _restore_prune_rearm_tokens(agent.context_compressor, attempt_snapshot) + return None, _existing_sp + return compressed, None + + +def _publish_rotated_compaction( + agent: Any, + messages: list, + compressed: list, + *, + new_system_prompt: str, + lease: _CompressionLease, + old_session_id: str, + compressed_user_turn_outcome: str, +) -> None: + """Rotate the session: flush the parent, publish the child, re-point the agent. + + Flushes current-turn msgs to the OLD session, passing the durable prefix + (messages[:persist idx]) so preflight, which runs before rows are + marker-stamped, can't re-append them. + """ + current_idx = getattr(agent, "_persist_user_message_idx", None) + persisted_history = ( + messages[:current_idx] + if isinstance(current_idx, int) + and 0 <= current_idx <= len(messages) + else None + ) + # The flush is durable and NOT rolled back on abort: a deliberately-ended parent + # fails publish forever, so check that before writing. Automatic end stamps are + # healed by publish (don't abort); the lease is re-acquirable (don't check it). + _parent_row_reader = getattr(agent._session_db, "get_session", None) + _parent_already_ended = False + if callable(_parent_row_reader): + try: + from hermes_state_common import is_automatic_end_reason + + _parent_row = _parent_row_reader(old_session_id) or {} + _parent_already_ended = ( + _parent_row.get("ended_at") is not None + and not is_automatic_end_reason( + _parent_row.get("end_reason") + ) + ) + except Exception: + # Fail OPEN: an unreadable row must not turn a cheap + # guard into a new way to lose compression. + _parent_already_ended = False + if _parent_already_ended: + raise RuntimeError( + f"Compression parent already ended: {old_session_id}" + ) + # Foreign-tail ceiling: the flush below writes OUR rows (already in handoff); + # rows above the start watermark up to this MAX(id) are foreign appends. + try: + _foreign_tail_ceiling = ( + agent._session_db.get_active_message_watermark( + agent.session_id + ) + ) + except Exception: + # No trustworthy ceiling: the clone could duplicate the handoff, so skip tail + # preservation this rotation. + _foreign_tail_ceiling = None + try: + agent._flush_messages_to_session_db( + messages, + conversation_history=persisted_history, + ) + except Exception: + pass # best-effort — don't block compression on a flush error + # Publish closure + child + handoff in one transaction so no reader sees an + # empty child. Child stays on the parent's profile ("default" persists as NULL); + # publish also COALESCEs from the parent row for threads lacking HERMES_HOME. + try: + from hermes_cli.profiles import get_active_profile_name + + _profile_for_child = get_active_profile_name() + if _profile_for_child == "default": + _profile_for_child = None + except Exception: + _profile_for_child = None + old_title = agent._session_db.get_session_title(agent.session_id) + new_session_id = ( + f"{datetime.now().strftime('%Y%m%d_%H%M%S')}_" + f"{uuid.uuid4().hex[:6]}" + ) + from agent.context_compressor import _DB_PERSISTED_MARKER + agent._session_db.publish_compression_child( + parent_session_id=old_session_id, + child_session_id=new_session_id, + source=agent.platform + or os.environ.get("HERMES_SESSION_SOURCE", "cli"), + model=agent.model, + model_config=agent._session_init_model_config, + system_prompt=new_system_prompt, + messages=compressed, + cwd=getattr(agent, "working_directory", None), + profile_name=_profile_for_child, + compression_lock_holder=lease.holder, + require_compression_lease=lease.holder is not None, + require_lease_refresh=lease.holder is not None, + lease_ttl_seconds=lease.ttl, + watermark=( + lease.watermark + if _foreign_tail_ceiling is not None + else None + ), + watermark_ceiling=_foreign_tail_ceiling, + ) + # `already_present` stamping is done by run_agent's _sync_persisted_markers; + # this branch covers inserted/merged only; direct callers must use that wrapper. + if compressed_user_turn_outcome in {"inserted", "merged"}: + # Stamp the anchor source row itself, not the (drifted, possibly out-of-range) + # persist index; don't match the HANDOFF row — for `merged` it is a superset. + _compressed_anchor_source = None + for _reversed_message in reversed(messages): + if _is_real_user_message(_reversed_message): + _compressed_anchor_source = _reversed_message + break + if isinstance(_compressed_anchor_source, dict): + _compressed_anchor_source[_DB_PERSISTED_MARKER] = True + _session_messages = getattr( + agent, "_session_messages", None + ) + if ( + isinstance(_session_messages, list) + and _session_messages is not messages + ): + # Adoption may leave _session_messages on the pre-adoption list with an out-of- + # range idx; stamp every scoped twin against the ANCHOR SOURCE, as the wrapper. + _anchor_timestamp = _compressed_anchor_source.get( + "timestamp" + ) + _found_exact_timestamp_candidate = False + if _anchor_timestamp is not None: + for _twin_message in _session_messages: + if ( + isinstance(_twin_message, dict) + and _twin_message.get("timestamp") + == _anchor_timestamp + and _messages_match_scoped_identity( + _twin_message, + _compressed_anchor_source, + ) + ): + # Count an exact scoped twin REGARDLESS of marker: an already-stamped twin must + # still suppress the broad fallback or a content-equal old dup gets stamped. + _found_exact_timestamp_candidate = True + if not _twin_message.get( + _DB_PERSISTED_MARKER + ): + _twin_message[ + _DB_PERSISTED_MARKER + ] = True + if not _found_exact_timestamp_candidate: + # No exact twin anywhere (or timestamp-less anchor): stamp every scoped match. + # An already-stamped exact hit never opens this branch. + for _twin_message in _session_messages: + if ( + isinstance(_twin_message, dict) + and not _twin_message.get( + _DB_PERSISTED_MARKER + ) + and _messages_match_scoped_identity( + _twin_message, + _compressed_anchor_source, + ) + ): + _twin_message[ + _DB_PERSISTED_MARKER + ] = True + for _handoff_message in compressed: + if isinstance(_handoff_message, dict): + _handoff_message[_DB_PERSISTED_MARKER] = True + agent.session_id = new_session_id + agent._db_flush_scan_prefix = None + try: + from gateway.session_context import set_current_session_id + + set_current_session_id(agent.session_id) + except Exception: + os.environ["HERMES_SESSION_ID"] = agent.session_id + try: + from hermes_logging import set_session_context + + set_session_context(agent.session_id) + except Exception: + pass + agent._session_db_created = True + # Carry /goal to the child: load_goal is a flat per-session lookup with no + # parent walk, so the goal would silently die at the boundary. + try: + from hermes_cli.goals import migrate_goal_to_session + migrate_goal_to_session(old_session_id, agent.session_id, reason="compression") + except Exception as _goal_err: + logger.debug("Could not migrate goal on compression: %s", _goal_err) + # Same boundary hazard for /heartbeat state — carry it too. + try: + from hermes_cli.heartbeat import migrate_heartbeat_to_session + migrate_heartbeat_to_session(old_session_id, agent.session_id) + except Exception as _hb_err: + logger.debug("Could not migrate heartbeat on compression: %s", _hb_err) + # Same hazard for a persistent /loop: carry it so recurring wakeups survive. + try: + from hermes_cli.loops import migrate_loop_to_session + migrate_loop_to_session(old_session_id, agent.session_id, reason="compression") + except Exception as _loop_err: + logger.debug("Could not migrate loop on compression: %s", _loop_err) + # Carry the title unchanged: renumbering per rotation made one session look + # like many. Uniqueness holds: _set_session_title transfers off the ancestor. + if old_title: + # Read provenance BEFORE the write: the transfer clears the ancestor's row, so + # a later read is None and the child would be frozen as "user". + _src = None + try: + _src = agent._session_db.get_session_title_source( + old_session_id + ) + except Exception as _src_err: + logger.debug( + "Could not read title provenance: %s", _src_err + ) + try: + agent._session_db.set_session_title( + agent.session_id, old_title + ) + except (ValueError, Exception) as e: + logger.debug("Could not propagate title on compression: %s", e) + else: + # set_session_title() records "user"; restore the original authority so an + # inherited auto-title stays upgradeable and a manual one stays pinned. + if _src is not None: + try: + agent._session_db.set_session_title_source( + agent.session_id, _src + ) + except Exception as _src_err: + logger.debug( + "Could not propagate title provenance: %s", + _src_err, + ) + + +def _warn_summary_or_aux_fallback(agent: Any) -> None: + """Surface a failed summary, or a recovered-but-broken aux compression model, once.""" + summary_error = getattr(agent.context_compressor, "_last_summary_error", None) + if summary_error: + if getattr(agent, "_last_compression_summary_warning", None) != summary_error: + agent._last_compression_summary_warning = summary_error + agent._emit_warning( + f"⚠ Compression summary failed: {summary_error}. " + "Inserted a fallback context marker." + ) + else: + # Aux model may have errored and been recovered on main; tell the user their + # auxiliary.compression.model is broken even though compression succeeded. + _aux_fail_model = getattr(agent.context_compressor, "_last_aux_model_failure_model", None) + _aux_fail_err = getattr(agent.context_compressor, "_last_aux_model_failure_error", None) + if _aux_fail_model: + # Dedup on (model, error) so we don't spam on every compaction + _aux_key = (_aux_fail_model, _aux_fail_err) + if getattr(agent, "_last_aux_fallback_warning_key", None) != _aux_key: + agent._last_aux_fallback_warning_key = _aux_key + agent._emit_warning( + f"ℹ Configured compression model '{_aux_fail_model}' failed " + f"({_aux_fail_err or 'unknown error'}). Recovered using main model — " + "check auxiliary.compression.model in config.yaml." + ) + + +def _finish_compaction_boundary( + agent: Any, + compressed: list, + *, + new_system_prompt: str, + old_session_id: Optional[str], + in_place: bool, + compacted_in_place: bool, + session_commit_succeeded: bool, + defer_context_engine_notification: bool, + compression_made_progress: bool, + compression_used_fallback: bool, + compression_feasibility_skip: bool, + task_id: str, +) -> int: + """Post-commit bookkeeping: notify engines/providers/hooks, re-arm usage tracking. + + Returns the rough post-compression token estimate (diagnostics only). + """ + # old_session_id is bound only on rotation; _boundary_parent is the id the + # boundary notifications attribute prior state to (old id, or same id in-place). + _old_sid = old_session_id + _is_boundary = bool(_old_sid) or in_place + _context_engine_boundary_committed = session_commit_succeeded and ( + bool(_old_sid) or compacted_in_place + ) + _boundary_parent = _old_sid or agent.session_id or "" + + # The heartbeat's terminal stamp landed on the PARENT before the id re-pointed; + # clear labels (keep last_activity_at) so the archived row isn't falsely fresh. + if _old_sid and session_commit_succeeded: + try: + _labels_db = getattr(agent, "_session_db", None) + _clear_labels = getattr( + type(_labels_db) if _labels_db is not None else None, + "clear_session_activity_labels", + None, + ) + if callable(_clear_labels): + _clear_labels(_labels_db, _old_sid) + except Exception: + logger.debug( + "failed to clear archived compression parent's activity " + "labels (ignored)", + exc_info=True, + ) + + # Plugin engines use boundary_reason="compression" to keep lineage/checkpoint + # state. Fires in BOTH modes: in-place passes the same id, the boundary is real. + if _context_engine_boundary_committed: + if defer_context_engine_notification: + _queue_context_engine_compression_notification( + agent, + new_session_id=agent.session_id or "", + old_session_id=_boundary_parent, + ) + else: + _notify_context_engine_compression_complete( + agent, + new_session_id=agent.session_id or "", + old_session_id=_boundary_parent, + ) + + # Providers refresh cached per-session state; reset=False, conversation goes on. + # Fires in BOTH modes so buffers don't double-count dropped turns in-place. + try: + if _is_boundary and agent._memory_manager: + agent._memory_manager.on_session_switch( + agent.session_id or "", + parent_session_id=_boundary_parent, + reset=False, + reason="compression", + ) + except Exception as _me_err: + logger.debug("memory manager on_session_switch (compression): %s", _me_err) + + # Route via _emit_status so the warning reaches gateway platforms; store it on + # _compression_warning so a late-bound status_callback can replay it. + _cc = agent.context_compressor.compression_count + if _cc >= 2: + _cc_msg = ( + f"{agent.log_prefix}⚠️ Session compressed {_cc} times — " + f"accuracy may degrade. Consider /new to start fresh." + ) + agent._compression_warning = _cc_msg + agent._emit_status(_cc_msg) + + # session:compress lets hooks ingest the old session before it's lost; + # in_place=True tells them the same id was compacted rather than rotated. + if getattr(agent, "event_callback", None): + try: + agent.event_callback("session:compress", { + "platform": agent.platform or "", + "session_id": agent.session_id, + "old_session_id": _old_sid or "", + "in_place": in_place, + "compression_count": agent.context_compressor.compression_count, + }) + except Exception as e: + logger.debug("event_callback error on session:compress: %s", e) + + # Rotation-independent flag: the gateway uses it (not an id diff) to re-baseline + # transcript handling (history_offset=0 + rewrite on the same id) in-place. + agent._last_compression_attempt_in_place = compacted_in_place + agent._last_compaction_in_place = compacted_in_place + + # Diagnostics only, not provider usage: schema-heavy rough estimates can stay + # above threshold even after the next real request fits. + _compressed_est = estimate_request_tokens_rough( + compressed, + system_prompt=new_system_prompt or "", + tools=agent.tools or None, + ) + agent.context_compressor.last_compression_rough_tokens = _compressed_est + agent.context_compressor.last_prompt_tokens = -1 + agent.context_compressor.last_completion_tokens = 0 + agent.context_compressor.awaiting_real_usage_after_compression = True + # Transcript rewritten: invalidate the usage anchor's base snapshot explicitly + # (its structural check would fail closed anyway); estimate until re-anchored. + agent._usage_anchor = None + agent._turn_base_usage_anchor = None + # Arm the effectiveness verdict only after a completed rewrite crosses the + # boundary so later usage isn't charged to an attempt that changed nothing. + if compression_made_progress: + record_boundary = getattr( + type(agent.context_compressor), + "record_completed_compaction", + None, + ) + if callable(record_boundary): + record_boundary( + agent.context_compressor, + used_fallback=compression_used_fallback, + feasibility_skip=compression_feasibility_skip, + ) + else: + agent.context_compressor._verify_compaction_cleared_threshold = True + + # Clear file-read dedup cache: original read content was summarized away, so a + # re-read needs full content, not a "file unchanged" stub. + try: + from tools.file_tools import reset_file_dedup + reset_file_dedup(task_id) + except Exception: + pass + # Same for the skill_view repeat-view dedup: a post-compression + # re-view must return the full skill content again. + try: + from tools.skills_tool import reset_skill_view_dedup + reset_skill_view_dedup(task_id) + except Exception: + pass + return _compressed_est + + +def _candidate_rejected( + agent: Any, + compressed: Any, + messages: list, + messages_before_compression: list, + *, + attempt_generation: Any, + attempt_started_at: float, +) -> bool: + """Reject an unusable compression candidate before any session mutation. + + Order matters: compressor-reported abort, no progress, empty transcript, + superseded attempt. Each branch surfaces its own warning/telemetry; the + caller releases the lease and returns the input unchanged when True. + """ + # Aborted compression returns input unchanged: surface the error, skip rotation + # (no session ended); auto-compress callers detect no-op via equal lengths. + if getattr(agent.context_compressor, "_last_compress_aborted", False): + _err = getattr(agent.context_compressor, "_last_summary_error", None) or "unknown error" + if getattr(agent, "_last_compression_summary_warning", None) != _err: + agent._last_compression_summary_warning = _err + agent._emit_warning( + f"⚠ Compression aborted: {_err}. " + "No messages were dropped — conversation continues unchanged. " + "Run /compress to retry, or /new to start a fresh session." + ) + _emit_aborted_attempt_telemetry( + agent, + attempt_started_at, + ( + getattr(agent.context_compressor, "_last_summary_error", None) + and "summary_generation_aborted" + ), + ) + return True + + # Compare semantic state, not identity: engines may return an equal copy or + # mutate the live list. ``==`` first (subclass __eq__), then marker-insensitive. + if compressed == messages_before_compression or ( + _strip_marker_for_comparison(compressed) + == _strip_marker_for_comparison(messages_before_compression) + ): + if messages != messages_before_compression: + messages[:] = copy.deepcopy(messages_before_compression) + logger.info( + "Compression made no progress (session=%s) — skipping boundary rewrite.", + agent.session_id or "none", + ) + # Unchanged output would fail identically next turn; arm structural backoff so + # auto-compress stops re-firing each turn (success lifts it, force overrides). + try: + _no_progress_recorder = getattr( + agent.context_compressor, "_record_structural_no_op", None + ) + if callable(_no_progress_recorder): + _no_progress_recorder( + "compaction returned the transcript unchanged " + "(no_progress)" + ) + except Exception: + logger.debug( + "no-progress backoff arm failed", exc_info=True + ) + _emit_aborted_attempt_telemetry(agent, attempt_started_at, "no_progress") + return True + + if not compressed: + logger.error( + "context compression returned an empty transcript; refusing to " + "rotate session=%s so the parent remains resumable", + agent.session_id or "none", + ) + try: + agent._emit_warning( + "⚠ Compression returned an empty transcript. " + "No session split was performed; conversation continues unchanged." + ) + except Exception: + pass + return True + + # A newer attempt claiming this compressor supersedes us; discard the late + # candidate. Fence poison alone misses a successor that minted its own fence. + if not _compressor_attempt_is_current(agent.context_compressor, attempt_generation): + logger.warning( + "Discarding late compression candidate: attempt generation " + "%s was superseded by a newer attempt (current: %s) " + "(session=%s).", + attempt_generation, + getattr( + agent.context_compressor, + "_compression_attempt_generation", + None, + ), + agent.session_id or "none", + ) + _restore_messages_snapshot(messages, messages_before_compression) + agent._last_compaction_in_place = False + _emit_aborted_attempt_telemetry(agent, attempt_started_at, "attempt_superseded") + return True + return False + + def compress_context( agent: Any, messages: list, @@ -3317,46 +4128,16 @@ def compress_context( ) -> Tuple[list, str]: """Compress conversation context and split the session in SQLite. - Args: - agent: The owning :class:`AIAgent`. - messages: Current message history (will be summarised). - system_message: Current system prompt; used when compression needs a - rebuilt cached prompt. - approx_tokens: Pre-compression token estimate, logged for ops. - task_id: Tool task scope (used for clearing file-read dedup state). - focus_topic: Optional focus string for guided compression — the - summariser will prioritise preserving information related to - this topic. Inspired by Claude Code's ``/compact ``. - force: If True, bypass any active summary-failure cooldown. Set - by the manual ``/compress`` slash command so users can retry - immediately after an auto-compress abort. Auto-compress - callers use the default ``False``. - bypass_cooldown: If True, the automatic breaker gates ignore ONLY the - summary-failure cooldown for this attempt (#100661). Set by the - provider-proven overflow recovery path: the provider already - rejected the request, so deferring until the cooldown lapses - wedges the session. Unlike ``force`` it does not clear the - cooldown, and the ineffective/structural breakers still apply; - a failed attempt records its cooldown normally. - defer_context_engine_notification: Delay the existing context-engine - hook until a manual host commits its outer history transaction. - commit_fence: Optional cooperative fence for executor callers that - may time out. It prevents a late worker from mutating session state - after its caller has moved on. - - Returns: - ``(compressed_messages, new_system_prompt)`` tuple. When - compression aborts (aux LLM failed to produce a usable summary), - returns the original messages unchanged and the existing system - prompt — the session is NOT rotated. Callers should detect the - no-op via ``len(returned) == len(input)`` and stop the retry loop. + ``force`` (manual /compress) clears the summary-failure cooldown; + ``bypass_cooldown`` (provider-proven overflow) skips it once, breakers still + apply. ``commit_fence`` stops a timed-out worker mutating session state. + Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split. """ _compressor_attempt_snapshot = _snapshot_compressor_attempt_state( agent.context_compressor ) - # Claim attempt ownership: a detached, late-unwinding sibling attempt - # (stall-fallback overlap) must not restore its snapshot over ours or - # clear our cancellation consult (#96634 post-merge review). + # Claim attempt ownership so a late-unwinding sibling (stall-fallback overlap) + # cannot restore its snapshot over ours or clear our cancellation consult. _attempt_generation = _claim_compressor_attempt(agent.context_compressor) _durable_cooldown_authoritative: Optional[bool] = None _durable_cooldown_state: Optional[dict[str, Any]] = None @@ -3366,23 +4147,15 @@ def compress_context( ): raise RuntimeError("a compression notification is already pending") - # ``conversation_history_after_compression()`` needs the latest attempt's - # outcome, while ``_last_compaction_in_place`` remains the run-level signal - # read by gateway callers. ``None`` means this attempt aborted or made no - # boundary, so the previous flush baseline remains authoritative. + # Per-attempt outcome for conversation_history_after_compression(); None means + # aborted/no boundary, so the previous flush baseline stays authoritative. agent._last_compression_attempt_recorded = True agent._last_compression_attempt_in_place = None - # Clear the lock-skip signal at the VERY TOP, before the codex route and - # the breaker gates below can early-return (per-attempt state rule, - # #58630/#69853). A stale ``True``/holder value from a prior lock-skip - # must never make a later breaker/codex no-op look like lock contention - # to the automatic-path consumers (compression_deferred, #49874) — the - # second clear before lock acquisition below stays for the same reason - # it was added in #69870 and is simply idempotent now. + # Clear at the VERY TOP, before codex/breaker early-returns: a stale value must + # not make a later no-op look like lock contention to automatic-path consumers. agent._compression_skipped_due_to_lock = None - # Transient-block signal (#97488): cleared with the same per-attempt - # rule; set by the breaker gates below when a TRANSIENT guard (cooldown / - # structural backoff) no-ops this pass. + # Per-attempt transient-block signal, set when a cooldown/backoff guard no-ops + # this pass. agent._compression_blocked_transient = None _attempt_started_at = time.monotonic() @@ -3398,17 +4171,9 @@ def compress_context( except Exception: pass - # Codex app-server sessions: the codex agent owns the real thread context; - # Hermes' summarizer would only rewrite a local mirror without shrinking - # the actual thread (#36801). Route compaction to the app server's own - # thread/compact mechanism. Behavior is controlled by - # ``compression.codex_app_server_auto`` (native|hermes|off). - # The memory-provider context handoff below is intentionally Hermes-only: - # the app server does not expose its native summary prompt, so there is no - # truthful injection point for ``on_pre_compress()`` return text here. - # `is True` (not bool()): unit tests drive this path with bare MagicMock - # agents whose auto-created attributes are truthy; the gate must only arm - # on the explicit boolean set by agent_init from config. + # Codex owns the real thread; route compaction to its own compact (config + # compression.codex_app_server_auto). Memory handoff is Hermes-only: no native + # summary prompt to inject into. `is True`: MagicMock attributes are truthy. checkpoint_required = ( getattr(agent, "compression_checkpoint_required", False) is True ) @@ -3428,9 +4193,7 @@ def compress_context( agent.context_compressor, _compressor_attempt_snapshot, attempt_generation=_attempt_generation, ) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) + existing_prompt = _existing_system_prompt(agent, system_message) return messages, existing_prompt try: return _compress_context_via_codex_app_server( @@ -3445,9 +4208,8 @@ def compress_context( if _codex_fence_entered: commit_fence.finish_commit() - # Every automatic entrypoint must honor compressor-owned cooldown and - # breaker state. Gateway hygiene constructs a fresh AIAgent, so the - # persisted fallback streak is loaded by bind_session_state() before this. + # All automatic entrypoints honor compressor cooldown/breaker state; hygiene's + # fresh AIAgent loads the persisted streak via bind_session_state() first. if not force: _refresh_persisted_compression_guards(agent.context_compressor) blocked = getattr( @@ -3459,38 +4221,20 @@ def compress_context( blocked, agent.context_compressor, bypass_cooldown ): _mark_compression_blocked_transient(agent, agent.context_compressor) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) + existing_prompt = _existing_system_prompt(agent, system_message) return messages, existing_prompt - # Lazy feasibility check — run the auxiliary-provider probe + context - # length lookup just-in-time on the first compression attempt instead of - # at AIAgent.__init__. Saves ~400ms cold off every short session that - # never reaches the threshold (the vast majority of ``chat -q`` runs). - # The check itself sets ``agent._compression_warning`` so the - # status-callback replay machinery still emits the warning to the user - # the first time it would matter. + # Lazy feasibility probe (~400ms cold) on first attempt, not __init__; it sets + # _compression_warning so status replay still surfaces the warning. if not getattr(agent, "_compression_feasibility_checked", False): - # Mark as checked only after the probe completes. If the check - # raises (e.g. a fatal aux-context ValueError that aborts the - # session), leaving the flag unset is harmless; a non-fatal - # transient failure is swallowed inside the function so the flag - # is set normally on the next successful pass. + # Mark checked only after the probe completes; a raise leaves it unset + # harmlessly, transient failures are swallowed inside so it sets next pass. check_compression_model_feasibility(agent) agent._compression_feasibility_checked = True _pre_msg_count = len(messages) - # In-place compaction (config: compression.in_place, see #38763). When True, - # this compaction rewrites the message list and refreshes the system prompt - # when necessary, but keeps the SAME session_id — no end_session, no - # parent_session_id child, no - # `name #N` renumber, no contextvar/env/logging re-sync, no memory/context- - # engine session-switch. The conversation keeps one durable id for life, - # eliminating the session-rotation bug cluster. Default True (2107b86024). - # Default True matches DEFAULT_CONFIG / #38763. A missing attribute must - # NOT fall back to rotation mode — that re-enables the pre-lease drift - # path and can wedge busy sessions that never set the flag. + # In-place keeps the SAME session_id (no rotation/child/renumber/re-sync). A + # missing attribute must default True, not rotation, which can wedge sessions. in_place = bool(getattr(agent, "compression_in_place", True)) # Set True once the in-place DB write actually completes (the DB block can # raise and skip it). Surfaced to the gateway via agent._last_compaction_in_place. @@ -3515,387 +4259,29 @@ def compress_context( _compaction_status_emitted = bool(_compaction_status) if _compaction_status: agent._emit_status(_compaction_status) - _compaction_done_emitted = False - # Commit outcome of this attempt; rebound to "committed" on the success - # path just before returning the compressed history. The lifecycle - # closure reads it at call time, so any abort/exception path that skips - # that rebind keeps the terminal edge suppressed. - _commit_status = "aborted" + lifecycle = _CompactionLifecycle(agent, _compaction_status_emitted) - def _complete_compaction_lifecycle(*, force_terminal: bool = False) -> None: - nonlocal _compaction_done_emitted - if _compaction_done_emitted: - return - _compaction_done_emitted = True - # A suppressed start (quiet context engine) opened no visible - # compaction phase — emit no terminal edge either. Failure warnings - # go through agent._emit_warning and are never suppressed here. - # Aborts that never compacted (lock contender, cancelled commit - # fence) opt in via force_terminal: they still need the structured - # terminal edge so clients can retire their compaction phase. Chat - # surfaces filter this routine notice independently. - if _compaction_status_emitted and ( - _commit_status == "committed" or force_terminal - ): - _emit_compaction_done(agent) - - # ── Compression lock ──────────────────────────────────────────────── - # Atomic, state.db-backed lock per session_id. Without this, two - # AIAgent instances that share the same session_id (most commonly the - # parent-turn agent and its background-review fork — see - # ``agent/background_review.py``: ``review_agent.session_id = - # agent.session_id``) can each call compress() on overlapping - # snapshots of the same conversation. Both succeed, both rotate - # ``agent.session_id`` to a fresh id, both create child sessions in - # state.db parented to the same old id. The gateway's SessionEntry - # only catches one rotation, so the other child becomes an orphan - # that silently accumulates writes — Damien's repro shape. - # - # Acquire keyed on the OLD session_id (the rotation target's parent), - # because that's the id that competing paths see and read from - # SessionEntry at the start of their own compression attempt. - # - # If we can't acquire the lock, another path is mid-compression on - # this session. Aborting is correct: the messages are unchanged, the - # other path's rotation will produce the canonical new session_id, - # and our caller's auto-compress loop sees ``len(returned) == len(input)`` - # and stops retrying for this cycle. The session is NOT corrupted — - # we just sit out this round and let the winner finish. - _lock_db = getattr(agent, "_session_db", None) - _lock_sid = agent.session_id or "" - _lock_holder: Optional[str] = None - # Watermark captured at compression start (#75316); None = fall back to - # archive-everything (no concurrent-tail preservation this cycle). - _commit_watermark: Optional[int] = None - # Probe whether the lock subsystem is actually available on this - # SessionDB instance. A process running mismatched module versions can have - # this call site while its long-lived SessionDB instance predates the lock - # API. Only that structural absence is safe to fail open for: compression - # must make progress rather than spin forever after an update. Once the - # method has been resolved, every exception from its implementation fails - # closed because proceeding without a lock can fork the session lineage. - _try_acquire_lock = None - _lock_lookup_error: Optional[Exception] = None - _legacy_session_db_without_lock_api = False - # Clear any stale lock-skip signal from a prior call so this call's - # outcome alone determines what callers see. Without this an - # auto-compress lock-skip followed by a successful manual /compress - # would falsely report "Compression already in progress" and discard - # the compression results. - agent._compression_skipped_due_to_lock = None - if _lock_db is not None: - try: - _legacy_session_db_without_lock_api = _lock_api_is_absent_on_session_db( - _lock_db - ) - except Exception as exc: - _lock_lookup_error = exc - if _lock_lookup_error is None and not _legacy_session_db_without_lock_api: - try: - _try_acquire_lock = _lock_db.try_acquire_compression_lock - if not callable(_try_acquire_lock): - _lock_lookup_error = TypeError( - "compression lock API is present but not callable" - ) - except Exception as exc: - _lock_lookup_error = exc - try: - _lock_ttl = float(getattr(agent, "_compression_lock_ttl_seconds", 300.0) or 300.0) - except (TypeError, ValueError): - _lock_ttl = 300.0 - _lock_refresh_interval = getattr(agent, "_compression_lock_refresh_interval", None) - _lock_refresher: Optional[_CompressionLockLeaseRefresher] = None - # F4 (#76354, transplanted from PR #71569 by @ciabata-git): fence the - # durable-lock acquisition + release-hook publication so a host timeout - # can never win in the gap between acquiring the durable lock and having - # a holder-qualified way to release it. - _lock_setup_entered = False - - def _finish_lock_setup() -> None: - nonlocal _lock_setup_entered - if not _lock_setup_entered or commit_fence is None: - return - _lock_setup_entered = False - commit_fence.finish_lock_setup() - - if _lock_db is not None and _lock_sid: - _lock_holder = _compression_lock_holder(agent) - if _lock_lookup_error is not None: - # Attribute lookup itself failed for a reason other than a missing - # lock API. It is unsafe to proceed without a lock in that case. - _lock_holder = None - logger.warning( - "compression lock lookup raised unexpectedly for session=%s " - "(%s: %s) — skipping compression this cycle", - _lock_sid, type(_lock_lookup_error).__name__, _lock_lookup_error, - ) - _lock_acquired = False - elif _try_acquire_lock is None: - # The lock API itself is absent on this in-memory instance. Log once - # and proceed unlocked so an update-version skew cannot leave the - # outer auto-compression loop making no progress forever. - _lock_holder = None - if getattr(agent, "_last_compression_lock_error_sid", None) != _lock_sid: - agent._last_compression_lock_error_sid = _lock_sid - logger.warning( - "compression lock subsystem unavailable for session=%s " - "— proceeding without lock. This usually means a stale " - "in-memory module after an update; restart the process " - "(or `hermes update`) to resync.", - _lock_sid, - ) - _lock_acquired = True # acquired-but-unlocked compatibility path - else: - if commit_fence is not None: - _lock_setup_entered = commit_fence.begin_lock_setup() - if not _lock_setup_entered: - logger.info( - "Compression commit cancelled before lock acquisition " - "(session=%s).", - agent.session_id or "none", - ) - agent._last_compaction_in_place = False - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class="commit_fence_cancelled", - ) - _complete_compaction_lifecycle(force_terminal=True) - return messages, _existing_sp - try: - _lock_acquired = _try_acquire_lock( - _lock_sid, _lock_holder, ttl_seconds=_lock_ttl - ) - if _lock_acquired: - # Watermark (#75316): MAX(id) of active rows at compression - # START. Appends are NOT blocked while the slow provider - # summary runs — any row landing after this point is - # concurrent tail, and archive_and_compact() re-sequences - # it after the compacted set instead of archiving it. - try: - _commit_watermark = _lock_db.get_active_message_watermark( - _lock_sid - ) - # #97963: a captured watermark makes the eventual - # commit safe against rows appended after this - # point (they survive as cloned concurrent tail on - # BOTH commit paths — archive_and_compact and - # publish_compression_child). Tell the fence so a - # host at the turn-hold boundary can keep this - # attempt's commit admission instead of burning it. - if commit_fence is not None: - try: - commit_fence.mark_commit_watermark_fenced() - except AttributeError: - pass # test doubles without the method - except Exception as _wm_err: - # Watermark capture is safety-additive: without it the - # commit falls back to archive-everything (historical - # behavior), so failure here must not abort compression. - logger.warning( - "compression watermark capture failed for " - "session=%s (%s) — concurrent appends this cycle " - "will be archived with the snapshot", - _lock_sid, _wm_err, - ) - _commit_watermark = None - except Exception as _lock_err: - # The method exists and entered its implementation but failed. - # Do not mistake an internal AttributeError or TypeError for - # version skew: fail closed and preserve session lineage. A - # failure after SQLite committed the acquire can leave our - # holder row behind, so release it best-effort before returning - # unchanged messages; release is holder-qualified and safe when - # acquisition never succeeded. - try: - _lock_db.release_compression_lock(_lock_sid, _lock_holder) - except Exception as _release_err: - logger.debug( - "compression lock cleanup after failed acquire failed: %s", - _release_err, - ) - _lock_holder = None - logger.warning( - "compression lock acquisition raised unexpectedly for " - "session=%s (%s: %s) — skipping compression this cycle", - _lock_sid, type(_lock_err).__name__, _lock_err, - ) - _lock_acquired = False - if not _lock_acquired: - _finish_lock_setup() - try: - existing = _lock_db.get_compression_lock_holder(_lock_sid) - except Exception: - existing = None - logger.warning( - "compression skipped: another path is compressing session=%s " - "(holder=%s) — returning messages unchanged to avoid session fork", - _lock_sid, existing, - ) - _lock_holder = None # don't release a lock we don't own - # Signal to callers that this no-op is due to a concurrent lock, - # not a genuine "nothing to compress" or aux-model failure. - # Manual /compress callers can surface a clear status message - # instead of the misleading "No changes from compression" text. - agent._compression_skipped_due_to_lock = existing or True - # Surface to the user once — quiet for downstream auto-compress loops - if getattr(agent, "_last_compression_lock_warning_sid", None) != _lock_sid: - agent._last_compression_lock_warning_sid = _lock_sid - try: - agent._emit_warning( - "⚠ Skipping concurrent compression — another path " - "is already compressing this session. Will retry " - "after it finishes." - ) - except Exception: - pass - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - try: - if hasattr(agent.context_compressor, "_begin_compression_telemetry"): - agent.context_compressor._begin_compression_telemetry(current_tokens=approx_tokens) - except Exception: - pass - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class="lock_contended", - ) - _complete_compaction_lifecycle(force_terminal=True) - return messages, _existing_sp - _lock_released = False - _lock_release_guard = threading.Lock() - - def _release_lock_holder_only() -> None: - """Stop this holder's refresher and release only its durable lock. - - Holder-qualified and idempotent (#76354 F4, from PR #71569): safe for - the HOST to invoke after a timeout without an ABA race — the DB - release is scoped to this worker's holder token, so a NEW holder's - lease can never be deleted by this stale release. - """ - nonlocal _lock_released - with _lock_release_guard: - if _lock_released: - return - _lock_released = True - if getattr(agent, "_active_compression_lock_holder", None) == _lock_holder: - agent._active_compression_lock_holder = None - if _lock_refresher is not None: - try: - _lock_refresher.stop() - except Exception as _stop_err: - logger.debug("compression lock refresher stop failed: %s", _stop_err) - if _lock_db is not None and _lock_sid and _lock_holder: - try: - _lock_db.release_compression_lock(_lock_sid, _lock_holder) - except Exception as _rel_err: - logger.debug("compression lock release failed: %s", _rel_err) - - def _release_lock() -> None: - """Finish lifecycle cleanup and release the OLD session lock once.""" - try: - _complete_compaction_lifecycle() - finally: - try: - _release_lock_holder_only() - finally: - try: - if commit_fence is not None: - commit_fence.clear_cancelled_lock_release( - _release_lock_holder_only - ) - finally: - _finish_lock_setup() - - if _lock_holder is not None: - agent._active_compression_lock_holder = _lock_holder - if ( - commit_fence is not None - and commit_fence.register_cancelled_lock_release( - _release_lock_holder_only - ) - ): - # Cancellation already won while we were inside lock setup: the - # hook just ran synchronously, our lease is gone — abort before - # any summary work. - logger.info( - "Compression commit cancelled before summary dispatch " - "(session=%s).", - agent.session_id or "none", - ) - agent._last_compaction_in_place = False - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class="commit_fence_cancelled", - ) - _release_lock() - return messages, _existing_sp + lease, _abort_prompt = _acquire_compression_lease( + agent, + commit_fence=commit_fence, + lifecycle=lifecycle, + system_message=system_message, + approx_tokens=approx_tokens, + attempt_started_at=_attempt_started_at, + ) + if lease is None: + return messages, _abort_prompt # Publish the holder-qualified release hook before a timeout can win the # fence. If no durable lock was acquired there is no hook to publish. - _finish_lock_setup() + lease.finish_lock_setup() - # A delayed contender can acquire the parent lock after the winning path - # has released it and completed rotation. The lock serializes work but does - # not by itself prove that this stale agent still owns a live parent. - if _lock_db is not None and _lock_sid: - try: - _parent_already_rotated = _session_was_rotated_by_compression( - _lock_db, _lock_sid - ) - except Exception as _session_err: - logger.warning( - "compression session ownership lookup failed for session=%s " - "(%s: %s) - skipping compression this cycle", - _lock_sid, - type(_session_err).__name__, - _session_err, - ) - _release_lock() - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - return messages, _existing_sp - if _parent_already_rotated: - recovered_messages = _adopt_live_compression_child( - agent, _lock_db, _lock_sid - ) - _release_lock() - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - if recovered_messages is not None: - logger.warning( - "compression recovery: stale session=%s adopted live child=%s", - _lock_sid, - agent.session_id, - ) - return recovered_messages, _existing_sp - logger.warning( - "compression skipped: session=%s was already rotated by " - "another compression path, but no unique live child could be adopted", - _lock_sid, - ) - return messages, _existing_sp + _adopted = _adopt_if_parent_rotated(agent, lease, messages, system_message) + if _adopted is not None: + return _adopted - # Snapshot the authoritative durable cooldown only after this attempt owns - # the session lease. This runs for force=True too, but does not apply the - # automatic breaker gate: manual compression still retries immediately. + # Snapshot durable cooldown only once we own the lease. Runs for force=True + # too but skips the automatic breaker gate: manual compression retries now. _durable_cooldown_authoritative, _durable_cooldown_state = ( _capture_authoritative_cooldown_under_lease( agent.context_compressor, @@ -3903,20 +4289,14 @@ def compress_context( ) ) if _durable_cooldown_authoritative is False: - # A bound built-in compressor reached its durable getter and the read - # failed. Proceeding with force=True could clear an unknown newer row - # before cancellation has enough information to restore it. This is a - # persistence-safety abort, not automatic breaker gating. - _release_lock() - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) + # Durable cooldown read failed under a built-in compressor: force=True could + # clear an unknown newer row before cancellation could restore it. Abort. + lease.release() + existing_prompt = _existing_system_prompt(agent, system_message) return messages, existing_prompt - # The agent may have been constructed before another path completed an - # in-place compaction on the same session. Re-read durable breaker state - # after acquiring the session lock so this final gate cannot act on the - # stale snapshot loaded by bind_session_state(). + # Another path may have compacted this session in place since construction; + # re-read breaker state under the lock, not the bind_session_state() snapshot. if not force: compressor = agent.context_compressor _refresh_persisted_compression_guards( @@ -3932,332 +4312,53 @@ def compress_context( blocked, compressor, bypass_cooldown ): _mark_compression_blocked_transient(agent, compressor) - _release_lock() - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) + lease.release() + existing_prompt = _existing_system_prompt(agent, system_message) return messages, existing_prompt _activity_heartbeat: Optional[_CompressionActivityHeartbeat] = None messages_before_compression = None try: - if _lock_holder is not None: - _candidate_refresher = _CompressionLockLeaseRefresher( - _lock_db, - _lock_sid, - _lock_holder, - _lock_ttl, - _lock_refresh_interval, - ) - # Cancellation may release the holder after hook publication but - # before this refresher starts. Serialize that check/start with - # the idempotent release path so a refresher is never started for - # an already-released lock (#76354 F4 / PR #71569). - with _lock_release_guard: - if not _lock_released: - _lock_refresher = _candidate_refresher - _lock_refresher.start() + lease.start_refresher() - # The caller's history snapshot predates lease acquisition. Reload the - # durable parent after the lease is live; MORE durable rows than the - # snapshot carries means a frontend/background writer committed a turn - # in that window, so publishing from this snapshot would omit it. - # Deliberately a LENGTH check, not content equality: in-memory - # mutation of past turns is legal (multimodal compression, retry - # history replacement, think-tag stripping), and a content-equality - # abort would permanently wedge compression on such sessions — the - # #14694 failure shape. - # Rotation-only: in-place compaction (archive_and_compact) is - # non-destructive — pre-compaction rows are soft-archived (active=0, - # compacted=1), stay searchable and recoverable, so snapshot/durable - # drift cannot lose data there and must not abort compaction. - # - # When durable DID grow, ADOPT it and continue rather than aborting. - # Aborting returned the stale snapshot unchanged, so busy sessions - # (memory review / shared session_id writers) stayed permanently - # behind the DB: every /compress and auto-compress saw - # "changed before lease acquisition", surfaced as the misleading - # "No changes from compression", and never reclaimed tokens. - if not in_place and _lock_db is not None and _lock_sid: - durable_loader = getattr( - type(_lock_db), "get_messages_as_conversation", None - ) - if callable(durable_loader): - durable_parent = durable_loader(_lock_db, _lock_sid) - if isinstance(durable_parent, list) and len(durable_parent) > len(messages): - # The in-memory transcript carries the CURRENT turn's - # un-persisted user tail (anchored by - # _persist_user_message_idx) that the durable snapshot read - # above does not contain yet. Flush that tail through the - # normal rotation-boundary path (conversation_history = the - # already-durable prefix, #68196 boundary) BEFORE adopting, - # then re-read the durable parent so the adopted snapshot - # includes the live input. If the flush fails (or the - # anchor is unknown), skip adoption entirely: replacing - # the in-memory transcript with a snapshot that lacks the - # user's input would silently drop it from the summarized - # and rotated history (#adopt-live-tail). - _preflush_idx = getattr(agent, "_persist_user_message_idx", None) - _preflush_ok = False - if ( - isinstance(_preflush_idx, int) - and 0 <= _preflush_idx < len(messages) - ): - try: - _preflush_ok = agent._flush_messages_to_session_db( - messages, - conversation_history=messages[:_preflush_idx], - ) - except Exception: - _preflush_ok = False - else: - # No known un-persisted tail (anchor unset or already - # at the end of the snapshot): the in-memory transcript - # is fully durable, so adopting the longer parent - # cannot drop live input — keep the legacy - # adopt-directly behavior for that shape - # (test_compression_concurrent_fork). - _preflush_ok = True - if not _preflush_ok: - logger.warning( - "compression: session=%s grew before lease " - "(%d → %d msgs) but the pre-adoption flush of the " - "live tail failed; skipping durable-snapshot " - "adoption so un-persisted user input is kept", - _lock_sid, - len(messages), - len(durable_parent), - ) - else: - # Re-read after the flush so the adopted snapshot - # carries the just-persisted tail. - durable_parent = durable_loader(_lock_db, _lock_sid) - if ( - _preflush_ok - and isinstance(durable_parent, list) - and len(durable_parent) > len(messages) - ): - logger.info( - "compression: session=%s grew before lease " - "(%d → %d msgs); adopting durable snapshot", - _lock_sid, - len(messages), - len(durable_parent), - ) - messages = durable_parent - _pre_msg_count = len(messages) - # Token estimate was for the stale snapshot; clear it so - # the compressor re-derives from the adopted transcript - # instead of under-counting the newly visible rows. - approx_tokens = 0 - # The whole adopted list is durable (DB re-read plus - # the just-flushed tail). Re-anchor the persist index - # at the end so the rotation-boundary flush that runs - # after compression skips the adopted rows by identity - # (conversation_history=messages[:idx]) instead of - # re-appending the concurrent rows and the live tail. - # This rebind is the concrete divergence path that can - # leave `agent._session_messages` pointing at the old - # live list while `messages` now points at the adopted - # durable snapshot; the post-publish marker sync in - # run_agent.py keeps both views aligned. - agent._persist_user_message_idx = len(messages) + if not in_place: + _adopted_parent = _adopt_grown_durable_parent(agent, lease, messages) + if _adopted_parent is not None: + messages = _adopted_parent + _pre_msg_count = len(messages) + # Estimate was for the stale snapshot; force re-derivation from adopted rows. + approx_tokens = 0 + # Adopted list is fully durable: re-anchor persist idx at the end so the post- + # compression flush skips it; run_agent marker sync realigns _session_messages. + agent._persist_user_message_idx = len(messages) - # Notify external memory provider before compression discards context. - # The provider's on_pre_compress() may return a string of insights it - # wants surfaced inside the compression summary; capture and forward it - # instead of silently discarding the provider's return value. - memory_context = "" - memory_manager = getattr(agent, "_memory_manager", None) - # Raw messages remain the historical (API v1) provider contract; the - # normalized evidence list is handed only to API v2+ checkpoint - # providers inside MemoryManager.on_pre_compress(). - evidence_messages = _direct_messages_for_pre_compress_memory(messages) - if checkpoint_required: - supports_checkpoint = getattr( - memory_manager, "supports_pre_compress_checkpoint", None - ) - if memory_manager is None or not callable(supports_checkpoint): - raise _checkpoint_blocked( - f"no active provider implements checkpoint API " - f"v{PRE_COMPRESS_CHECKPOINT_API_VERSION}" - ) - try: - compatible = bool( - supports_checkpoint(PRE_COMPRESS_CHECKPOINT_API_VERSION) - ) - except Exception as exc: - raise _checkpoint_blocked("provider capability probe failed") from exc - if not compatible: - raise _checkpoint_blocked( - f"active provider does not implement checkpoint API " - f"v{PRE_COMPRESS_CHECKPOINT_API_VERSION}" - ) - try: - _maybe_ctx = memory_manager.on_pre_compress( - messages, - evidence_messages=evidence_messages, - require_checkpoint=True, - checkpoint_api_version=PRE_COMPRESS_CHECKPOINT_API_VERSION, - ) - except Exception as exc: - logger.warning( - "Required pre-compress checkpoint failed (%s)", - type(exc).__name__, - ) - raise _checkpoint_blocked( - f"provider checkpoint API v{PRE_COMPRESS_CHECKPOINT_API_VERSION} failed" - ) from exc - if isinstance(_maybe_ctx, str): - memory_context = sanitize_memory_context(_maybe_ctx) - elif memory_manager: - try: - _maybe_ctx = memory_manager.on_pre_compress( - messages, evidence_messages=evidence_messages - ) - if isinstance(_maybe_ctx, str): - memory_context = sanitize_memory_context(_maybe_ctx) - except Exception: - pass + memory_context = _pre_compress_memory_context(agent, messages, checkpoint_required) - compress_fn = agent.context_compressor.compress - compress_kwargs = _supported_compression_kwargs( - compress_fn, - current_tokens=approx_tokens, + compress_fn, compress_kwargs = _resolve_compress_call( + agent, + approx_tokens=approx_tokens, focus_topic=focus_topic, force=force, memory_context=memory_context, bypass_cooldown=bypass_cooldown, ) - if memory_context.strip() and "memory_context" not in compress_kwargs: - engine_name = getattr( - agent.context_compressor, - "name", - type(agent.context_compressor).__name__, - ) - if ( - getattr(agent, "_last_memory_context_unsupported_engine", None) - != engine_name - ): - agent._last_memory_context_unsupported_engine = engine_name - logger.warning( - "context engine %s does not accept memory_context; continuing " - "without provider-supplied summary context", - engine_name, - ) messages_before_compression = copy.deepcopy(messages) _activity_heartbeat = _CompressionActivityHeartbeat( agent, commit_fence=commit_fence ).start() - # Publish forward progress to the commit fence while the summary LLM - # call streams. Async hosts (gateway session hygiene) poll - # ``commit_fence.seconds_since_progress()`` to extend their deadline - # while tokens are moving — so a SLOW summary model is only killed - # when it is actually silent, not merely thorough. The hook is - # thread-local and the compress call is synchronous on this thread, - # so it cannot leak into unrelated auxiliary calls. - # - # Callers that pass no commit_fence install a no-op progress hook - # here. AIAgent._compress_context injects an owned fence for - # fenceless callers so the host-level progress-aware wait can - # extend on streamed tokens; gateway hygiene already passes its - # own fence. An ACTIVE hook (even a no-op) is what switches the - # summary call onto the streamed path — giving every compression - # path the same two guarantees: the configured timeout acts on - # inactivity (slow models finish), and a byte-trickling provider - # that keeps the connection alive forever is cut off at the - # streamed total ceiling (see _aux_stream_total_ceiling) instead of - # outliving the SDK's inactivity timeout indefinitely. - from agent.auxiliary_client import ( - aux_interrupt_protection, - aux_progress_hook, - aux_stream_deadline, - ) - _progress_hook = ( - commit_fence.touch_progress if commit_fence is not None - else (lambda: None) - ) - # #99692: the progress hook above is the worker -> host leg; this is the - # return leg. _compression_cancel_requested (below) releases the compression - # OWNER when the host gives up, but the isolated provider daemon that - # actually holds the socket keeps streaming to its own budget — - # ``_aux_stream_total_ceiling`` = max(600, 4 * aux_timeout), which is >= - # the host's total ceiling for every configured timeout and starts - # counting later (after admission, serialization, prompt build and TTFT). - # With ``auxiliary.compression.timeout: 600`` that is 2400s of an - # orphaned 500K-token summary the commit fence is already guaranteed to - # refuse: paid tokens, a pinned HTTP connection, and — since every new - # turn re-triggers compression on a session that never shrank — a fresh - # orphan stacked on top of the last one. Sharing the host's absolute - # deadline makes the stream stop when the host it serves stops waiting. - _host_stream_deadline = ( - commit_fence.deadline_monotonic if commit_fence is not None else None - ) - # F4 state-ordering (#76354): a LATE successful summary must not undo - # the timeout cooldown the host recorded. Install a cancellation - # check the compressor consults BEFORE clearing the failure cooldown; - # removed in the finally below so it cannot leak into later attempts - # (e.g. a manual /compress force-clear). - if commit_fence is not None: - _install_compression_cancelled_check( - agent.context_compressor, - lambda: commit_fence.is_cancelled, - _attempt_generation, - ) - # Incoming-message interrupts and active-turn redirects must not tear an - # atomic summary in half (#23975). Explicit stop surfaces set a separate - # Event atomically; never infer cause from the racy message fields. A - # host timeout also cancels the attempt's commit fence. Feed BOTH into - # the protected auxiliary-call seam so the compression owner unwinds - # promptly while an isolated provider stream finishes or closes in its - # daemon worker. Otherwise four timed-out streams retain all four shared - # compression-pool slots until the auxiliary stream's longer absolute - # ceiling expires. + # Interrupts/redirects must not tear a summary in half. Use the explicit stop + # Event (message fields race) + fence timeout so pool slots free promptly. _hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None) - - def _compression_cancel_requested() -> bool: - return bool( - ( - _hard_cancel_event is not None - and _hard_cancel_event.is_set() - ) - or ( - commit_fence is not None - and commit_fence.is_cancelled - ) - ) - - try: - # F6: never start expensive summary work for an already-cancelled - # fence (a stale queued job admitted after host departure). - if commit_fence is not None and commit_fence.is_cancelled: - logger.info( - "Compression cancelled before summary dispatch " - "(session=%s) — skipping summary work.", - agent.session_id or "none", - ) - compressed = messages - else: - with aux_progress_hook(_progress_hook), aux_stream_deadline( - _host_stream_deadline - ), aux_interrupt_protection( - cancel_check=_compression_cancel_requested - ): - compressed = compress_fn(messages, **compress_kwargs) - # Freeze a hard stop that arrived after the final provider - # attempt unwound but before this transaction can rotate - # session state. - if ( - _hard_cancel_event is not None - and _hard_cancel_event.is_set() - ): - raise AuxiliaryExplicitCancellation() - finally: - if commit_fence is not None: - _clear_compression_cancelled_check_if_owner( - agent.context_compressor, _attempt_generation - ) + compressed = _run_summary_dispatch( + agent, + messages, + compress_fn, + compress_kwargs, + commit_fence=commit_fence, + attempt_generation=_attempt_generation, + hard_cancel_event=_hard_cancel_event, + ) except AuxiliaryExplicitCancellation: try: _restore_compressor_attempt_state( @@ -4270,28 +4371,14 @@ def compress_context( except BaseException as _rollback_exc: # Compensation failure must surface, but it must not strand the # session lease or retain an in-memory transcript mutation. - if ( - messages_before_compression is not None - and messages != messages_before_compression - ): - messages[:] = copy.deepcopy(messages_before_compression) + _restore_messages_snapshot(messages, messages_before_compression) if _activity_heartbeat is not None: _activity_heartbeat.stop("context compression rollback failed") _activity_heartbeat = None - _release_lock() - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class=f"rollback:{type(_rollback_exc).__name__}", - ) + lease.release() + _emit_aborted_attempt_telemetry(agent, _attempt_started_at, f"rollback:{type(_rollback_exc).__name__}") raise - if ( - messages_before_compression is not None - and messages != messages_before_compression - ): - messages[:] = copy.deepcopy(messages_before_compression) + _restore_messages_snapshot(messages, messages_before_compression) # Record after restore so rollback cannot wipe a stall backoff, and # while the lease is still held so the next turn cannot race it. _stall_backoff = _record_stall_interrupted_backoff( @@ -4304,37 +4391,26 @@ def compress_context( if _activity_heartbeat is not None: _activity_heartbeat.stop("context compression cancelled") _activity_heartbeat = None - _release_lock() - _emit_compression_attempt_telemetry( + lease.release() + _emit_aborted_attempt_telemetry( agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class=( + _attempt_started_at, + ( STALL_INTERRUPTED_FAILURE_CLASS if _stall_backoff else "explicit_interrupt" ), ) - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) + _existing_sp = _existing_system_prompt(agent, system_message) return messages, _existing_sp except BaseException as _compress_exc: - # ANY exception after lock acquisition — memory hook, capability - # inspection, engine lookup, or compress() — must release the lock so - # the session isn't permanently blocked from future compression. + # Any failure after lock acquisition must release it or the session is + # permanently blocked from compression. if _activity_heartbeat is not None: _activity_heartbeat.stop("context compression failed") _activity_heartbeat = None - _release_lock() - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class=f"exception:{type(_compress_exc).__name__}", - ) + lease.release() + _emit_aborted_attempt_telemetry(agent, _attempt_started_at, f"exception:{type(_compress_exc).__name__}") raise finally: if _activity_heartbeat is not None: @@ -4342,10 +4418,8 @@ def compress_context( _commit_fence_entered = False try: - # Capture boundary quality before session-rotation callbacks run. Built-in - # and plugin lifecycle hooks may reset per-session compressor fields while - # rebinding to the child id; the completed attempt's verdict must survive - # that rebind and be recorded only after the full boundary commits. + # Capture the verdict before rotation callbacks: lifecycle hooks may reset + # compressor fields on rebind; record only after the full boundary commits. _compression_made_progress = bool( getattr(agent.context_compressor, "_last_compression_made_progress", False) ) @@ -4356,147 +4430,16 @@ def compress_context( getattr(agent.context_compressor, "_last_feasibility_skip", False) ) - # If compression aborted (aux LLM failed to produce a usable summary) - # the compressor returns the input messages unchanged. Surface the - # error to the user, skip the session-rotation work entirely (no - # session has logically ended), and let auto-compress callers detect - # the no-op via len(returned) == len(input). - if getattr(agent.context_compressor, "_last_compress_aborted", False): - try: - _err = getattr(agent.context_compressor, "_last_summary_error", None) or "unknown error" - if getattr(agent, "_last_compression_summary_warning", None) != _err: - agent._last_compression_summary_warning = _err - agent._emit_warning( - f"⚠ Compression aborted: {_err}. " - "No messages were dropped — conversation continues unchanged. " - "Run /compress to retry, or /new to start a fresh session." - ) - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class=( - getattr(agent.context_compressor, "_last_summary_error", None) - and "summary_generation_aborted" - ), - ) - return messages, _existing_sp - finally: - _release_lock() - - # Compare against the pre-dispatch semantic state, not object identity: - # legacy/plugin engines may return an equal copy for a no-op, or mutate - # the live list while returning an unchanged snapshot. Neither case may - # rotate or rewrite the session. The raw ``==`` leg runs FIRST so a - # list subclass returned by an engine keeps its ``__eq__`` semantics - # (tests seam on this); the marker-insensitive leg (#92231) then covers - # the cold-resume shape where the stamped snapshot differs from the - # marker-swept compress() output only by ``_db_persisted``. - if compressed == messages_before_compression or ( - _strip_marker_for_comparison(compressed) - == _strip_marker_for_comparison(messages_before_compression) + if _candidate_rejected( + agent, + compressed, + messages, + messages_before_compression, + attempt_generation=_attempt_generation, + attempt_started_at=_attempt_started_at, ): - if messages != messages_before_compression: - messages[:] = copy.deepcopy(messages_before_compression) - logger.info( - "Compression made no progress (session=%s) — skipping boundary rewrite.", - agent.session_id or "none", - ) - # Dead-loop breaker (#84371): a fired compaction that returns the - # transcript UNCHANGED will fail identically next turn unless the - # transcript changes — yet this path recorded telemetry only, so - # auto-compress re-fired every turn, each attempt burning a full - # aux summarization (6+/10min in the wild). Arm the transient - # structural backoff so the next attempts are deferred; any - # successful boundary lifts it, and manual /compress overrides it. - try: - _no_progress_recorder = getattr( - agent.context_compressor, "_record_structural_no_op", None - ) - if callable(_no_progress_recorder): - _no_progress_recorder( - "compaction returned the transcript unchanged " - "(no_progress)" - ) - except Exception: - logger.debug( - "no-progress backoff arm failed", exc_info=True - ) - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class="no_progress", - ) - _release_lock() - return messages, _existing_sp - - if not compressed: - logger.error( - "context compression returned an empty transcript; refusing to " - "rotate session=%s so the parent remains resumable", - agent.session_id or "none", - ) - try: - agent._emit_warning( - "⚠ Compression returned an empty transcript. " - "No session split was performed; conversation continues unchanged." - ) - except Exception: - pass - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _release_lock() - return messages, _existing_sp - - # Supersession guard (#97488): a NEWER attempt claiming this - # compressor (via _claim_compressor_attempt) supersedes this one — - # this attempt's late candidate must be discarded, never committed - # over the newer attempt's state. Checked for fenceless callers too: - # the fence poison alone cannot see a successor that minted its own - # fresh fence. - _attempt_superseded = not _compressor_attempt_is_current( - agent.context_compressor, _attempt_generation - ) - if _attempt_superseded: - logger.warning( - "Discarding late compression candidate: attempt generation " - "%s was superseded by a newer attempt (current: %s) " - "(session=%s).", - _attempt_generation, - getattr( - agent.context_compressor, - "_compression_attempt_generation", - None, - ), - agent.session_id or "none", - ) - if ( - messages_before_compression is not None - and messages != messages_before_compression - ): - messages[:] = copy.deepcopy(messages_before_compression) - agent._last_compaction_in_place = False - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class="attempt_superseded", - ) - _release_lock() + _existing_sp = _existing_system_prompt(agent, system_message) + lease.release() return messages, _existing_sp if commit_fence is not None: @@ -4509,11 +4452,7 @@ def compress_context( durable_cooldown_state=_durable_cooldown_state, attempt_generation=_attempt_generation, ) - if ( - messages_before_compression is not None - and messages != messages_before_compression - ): - messages[:] = copy.deepcopy(messages_before_compression) + _restore_messages_snapshot(messages, messages_before_compression) logger.info( "Compression commit cancelled before session mutation " "(session=%s).", @@ -4527,383 +4466,68 @@ def compress_context( messages=messages, approx_tokens=approx_tokens, ) - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( + _existing_sp = _existing_system_prompt(agent, system_message) + _emit_aborted_attempt_telemetry( agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class=( + _attempt_started_at, + ( STALL_INTERRUPTED_FAILURE_CLASS if _stall_backoff else "commit_fence_cancelled" ), ) - _release_lock() + lease.release() return messages, _existing_sp - summary_error = getattr(agent.context_compressor, "_last_summary_error", None) - if summary_error: - if getattr(agent, "_last_compression_summary_warning", None) != summary_error: - agent._last_compression_summary_warning = summary_error - agent._emit_warning( - f"⚠ Compression summary failed: {summary_error}. " - "Inserted a fallback context marker." - ) - else: - # No hard failure — but did the configured aux model error out - # and get recovered by retrying on main? Surface that so users - # know their auxiliary.compression.model setting is broken even - # though compression succeeded. - _aux_fail_model = getattr(agent.context_compressor, "_last_aux_model_failure_model", None) - _aux_fail_err = getattr(agent.context_compressor, "_last_aux_model_failure_error", None) - if _aux_fail_model: - # Dedup on (model, error) so we don't spam on every compaction - _aux_key = (_aux_fail_model, _aux_fail_err) - if getattr(agent, "_last_aux_fallback_warning_key", None) != _aux_key: - agent._last_aux_fallback_warning_key = _aux_key - agent._emit_warning( - f"ℹ Configured compression model '{_aux_fail_model}' failed " - f"({_aux_fail_err or 'unknown error'}). Recovered using main model — " - "check auxiliary.compression.model in config.yaml." - ) + _warn_summary_or_aux_fallback(agent) - todo_snapshot = agent._todo_store.format_for_injection() - # A non-empty store is authoritative even when every item is already - # completed/cancelled and format_for_injection() therefore returns an - # empty string. In that case remove the previous snapshot so completed - # work is not resurrected. A truly empty store is different: fresh - # gateway agents may be unable to rehydrate todo tool results after a - # prior compaction, so the retained snapshot is the only surviving - # record of pending work and must stay in place. - _todo_has_items = getattr(agent._todo_store, "has_items", None) - try: - _todo_store_is_authoritative = bool( - _todo_has_items() - ) if callable(_todo_has_items) else False - except Exception: - # A plugin/test double may implement only format_for_injection(). - # Unknown authority must preserve pending snapshot state rather than - # risk deleting it during compression. - _todo_store_is_authoritative = False - if _todo_store_is_authoritative: - for _todo_idx in range(len(compressed) - 1, -1, -1): - _todo_message = compressed[_todo_idx] - if not isinstance(_todo_message, dict) or _todo_message.get("role") != "user": - continue - _todo_content = _todo_message.get("content") - _todo_stripped = _strip_stale_todo_snapshot(_todo_content) - if _todo_stripped == _todo_content: - continue - if ( - _todo_message.get("_todo_snapshot_synthetic") - and _todo_snapshot_is_only_content( - _todo_content, _todo_stripped - ) - ): - compressed.pop(_todo_idx) - if _todo_idx < len(compressed): - # A standalone snapshot can move away from the tail - # after later turns arrive. Deleting it may expose two - # assistant rows; use the normal replay repair so their - # content/tool-call metadata is preserved consistently. - agent._repair_message_sequence(compressed) - else: - _replace_message_content(_todo_message, _todo_stripped) - # The row is no longer todo-only scaffolding. Other - # synthetic flags, if any, remain authoritative and - # _is_real_user_message() recomputes provenance from the - # surviving content plus those flags. - _todo_message.pop("_todo_snapshot_synthetic", None) - break - if todo_snapshot: - # Retention parity (#84718): the snapshot below re-injects the - # imperative verbatim. If this same boundary pruned skill bodies - # to [SKILL_PRUNED: ...] markers, the policy that governed those - # tasks is gone — couple a reload instruction to the snapshot so - # the imperative never crosses the boundary alone. Appended after - # TODO_INJECTION_HEADER, so the stale-snapshot strip removes both - # together at the next boundary. - _reload_notice = _pruned_skill_reload_notice(compressed) - if _reload_notice: - todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}" - # Fold the snapshot into a trailing REAL user message so - # compression never introduces a synthetic user/user pair. Any - # snapshot merged at an earlier boundary is stripped first so - # repeated compactions refresh rather than accumulate todo state - # (#26981). Scaffolding tails (continuation marker, summary - # handoff, a bare stale snapshot row) must never absorb the - # snapshot: merging would upgrade them to "real user" evidence - # and break zero-user provenance (#69292), so those keep the - # flagged standalone append and the real-user preservation pass - # continues to see todo scaffolding, not human intent. - from agent.context_compressor import _append_text_to_content - - merged = False - _tail = ( - compressed[-1] - if compressed and isinstance(compressed[-1], dict) - else None - ) - if _tail is not None and _tail.get("role") == "user": - _stripped = _strip_stale_todo_snapshot(_tail.get("content")) - _probe = { - key: value for key, value in _tail.items() if key != "content" - } - _probe["content"] = _stripped - if _is_real_user_message(_probe): - _snapshot_text = ( - f"\n\n{todo_snapshot}" - if isinstance(_stripped, str) and _stripped - else todo_snapshot - ) - _replace_message_content( - _tail, - _append_text_to_content(_stripped, _snapshot_text), - ) - merged = True - elif _stripped != _tail.get("content") and not _message_text( - {"role": "user", "content": _stripped} - ).strip(): - # The tail was nothing but an earlier snapshot row — - # refresh it in place instead of stacking a duplicate. - _replace_message_content(_tail, todo_snapshot) - _tail["_todo_snapshot_synthetic"] = True - merged = True - if not merged: - compressed.append({ - "role": "user", - "content": todo_snapshot, - "_todo_snapshot_synthetic": True, - }) + _fold_todo_snapshot(agent, compressed) compressed_user_turn_outcome = _ensure_compressed_has_user_turn( messages, compressed ) - cached_system_prompt = agent._cached_system_prompt - agent._invalidate_system_prompt() - - # Refresh dynamic tool schemas at the same admitted-commit boundary - # that rebuilds the system prompt (maintainer-directed, #95681 arc): - # forever-sessions (Bot Mode chats, gateway channels) never restart, - # so compaction is the ONLY point where a config change — image - # model swap, delegation depth, code_execution mode — can reach - # agent.tools. The prompt cache is already broken here, so the - # refresh is free; when nothing changed the snapshot is byte-equal - # and we keep the existing list object (identity matters to - # provider-side tool-block caching on some backends). - try: - _refresh_agent_tool_definitions(agent) - except Exception: # noqa: BLE001 - logger.warning( - "compaction tool-definition refresh failed; keeping the " - "session's existing tool snapshot", - exc_info=True, - ) - - # ALWAYS rebuild the prompt at the admitted-commit boundary - # (maintainer-directed, #95681 arc). The previous "keep-prompt" - # containment branch put the OLD bytes back whenever the reloaded - # memory blocks were already embedded — which meant prompt-builder - # changes (guidance diets, new blocks, renames) NEVER reached a - # long-lived session. The cache argument for keeping bytes was - # hollow: when nothing changed, the rebuild is byte-identical and - # local KV prefixes survive on equality; when something changed, - # the cache was stale by definition and propagation is the point. - # Preserve OBJECT identity on byte-equality for backends that key - # on it. - rebuilt_system_prompt = agent._build_system_prompt(system_message) - if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt: - new_system_prompt = cached_system_prompt - agent._cached_system_prompt = cached_system_prompt - from agent.system_prompt import reconstruct_static_prefix - - reconstruct_static_prefix( - agent, - system_message=system_message, - log_label="compression keep-prompt", - ) - else: - new_system_prompt = rebuilt_system_prompt - agent._cached_system_prompt = new_system_prompt - if cached_system_prompt is not None: - logger.info( - "Compaction rebuilt a drifted system prompt " - "(session=%s, %d -> %d chars): builder output changed " - "since the stored snapshot (update, config change, or " - "memory/skills growth)", - agent.session_id or "none", - len(cached_system_prompt), - len(new_system_prompt), - ) + new_system_prompt = _rebuild_system_prompt_at_boundary(agent, system_message) _session_commit_succeeded = False _commit_started_at = time.monotonic() split_status = "not_applicable" + old_session_id: Optional[str] = None # bound only once rotation begins if agent._session_db: split_status = "pending" try: - # Trigger memory extraction on the current session before the - # transcript is rewritten (runs in BOTH modes — the logical - # conversation's pre-compaction turns are about to be summarized - # away regardless of whether the id rotates). + # Memory extraction runs in BOTH modes: pre-compaction turns are summarized + # away whether or not the id rotates. agent.commit_memory_session(messages) - # Pop the #86366 carried-forward-tail tags BEFORE the size - # estimate and BEFORE any non-in-place path can see them: - # the private `_compaction_tail` key must not inflate the - # anti-growth token estimate (it tipped break-even - # transcripts into a false "would grow" refusal) and must - # not ride rotation handoffs into the provider payload. - # Remember the tagged dicts by identity — the salvage pass - # below may swap `compressed` for a subset, and tail_count - # must describe the FINAL committed list. + # Pop _compaction_tail tags before the size estimate / rotation: they must not + # inflate anti-growth or reach the provider. Track ids: salvage may subset list. _tail_tagged_ids = { id(m) for m in compressed if isinstance(m, dict) and m.pop("_compaction_tail", None) } - # Anti-growth guard at the COMMIT SITE: never persist a - # compression that makes the transcript larger (observed: - # 379K -> 687K when the generated summary plus retained - # reasoning exceeded what it replaced). Compare like-for-like - # (both rough estimates of the same message shape) so an - # "actual vs estimate" measurement mismatch cannot produce a - # false verdict. The gateway has a rotation-path-only guard - # (#83339), but in-place compaction commits inside this method - # via archive_and_compact — before the gateway can inspect the - # result — so the guard must live here to protect both paths. - # On growth, treat the attempt as a no-op: the original - # transcript stays untouched and durable. - _rough_in = estimate_messages_tokens_rough(messages) - _rough_out = estimate_messages_tokens_rough(compressed) - if _rough_out > _rough_in: - # Todo refresh and user-turn anchoring happen after the - # compressor's own size check, so they can tip a break-even - # candidate over. Give it one mechanical salvage pass. - from agent.context_compressor import salvage_grown_transcript - - _salvaged = salvage_grown_transcript( - messages, compressed, budget=_rough_in - ) - if _salvaged is not None: - _salv_est = estimate_messages_tokens_rough(_salvaged) - if _salv_est < _rough_in: - logger.info( - "Compression salvage recovered a shrinking " - "transcript (session=%s, ~%s -> ~%s tokens)", - agent.session_id or "none", - f"{_rough_in:,}", - f"{_salv_est:,}", - ) - compressed = _salvaged - _rough_out = _salv_est - if _rough_out > _rough_in: - logger.warning( - "Compression refused: compressed transcript would be " - "larger than the original (session=%s, ~%s -> ~%s " - "tokens); keeping the original transcript unchanged", - agent.session_id or "none", - f"{_rough_in:,}", - f"{_rough_out:,}", - ) - # Flag the refusal on the compressor state so manual - # /compress feedback can report it honestly. Without this, - # the CLI compared the returned list against its pre-call - # snapshot, saw a difference (durable-snapshot adoption can - # legitimately change the count), and printed - # "✅ Compressed: 8 → 14 messages" directly under the - # refusal warning (Aug 2026 full-surface CLI QA sweep). - try: - agent.context_compressor._last_compress_refused_would_grow = True - except Exception: - pass - try: - agent._emit_warning( - "⚠️ Compression refused: the generated summary " - "would have GROWN the conversation instead of " - "shrinking it. No messages were dropped — " - "conversation continues unchanged." - ) - except Exception: - pass - _existing_sp = getattr(agent, "_cached_system_prompt", None) - if not _existing_sp: - _existing_sp = agent._build_system_prompt(system_message) - _emit_compression_attempt_telemetry( - agent, - started_at=_attempt_started_at, - commit_status="aborted", - split_status="aborted", - failure_class="would_grow", - ) - # Record the rejected attempt as an ineffective - # compaction strike so the anti-thrash breaker latches - # after the normal threshold. Without this, the unchanged - # transcript stays over the compression threshold and - # automatic compression retries the identical summary - # request on every turn (#88568). Manual /compress keeps - # bypassing the latch (force=True skips the guards). - try: - agent.context_compressor.record_rejected_compaction() - except Exception: - logger.debug( - "could not record rejected-compaction strike", - exc_info=True, - ) - # Restore ONLY the prune runway (same rationale as the - # rotation-failure rollback below): compress()'s successful - # tail already zeroed _proactive_prune_rearm_tokens in - # memory, but this refusal keeps the ORIGINAL transcript — - # whose cached prefix is intact. Leaving the runway at 0 - # disarms the #79640 throttle, so the very next iteration's - # proactive prune rewrites history and breaks the prompt - # cache without the required regrowth interval (#91830). - # The durable copy was never cleared (that clear only rides - # the archive_and_compact / child-row commit that never - # ran), so restoring the snapshot re-aligns memory with - # disk. - if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot: - agent.context_compressor._proactive_prune_rearm_tokens = ( - _compressor_attempt_snapshot[ - "_proactive_prune_rearm_tokens" - ] - ) - _release_lock() - return messages, _existing_sp + compressed, _refused_sp = _salvage_or_refuse_grown_transcript( + agent, + messages, + compressed, + system_message=system_message, + attempt_started_at=_attempt_started_at, + attempt_snapshot=_compressor_attempt_snapshot, + ) + if compressed is None: + lease.release() + return messages, _refused_sp if in_place: - # ── In-place compaction: keep the same session_id ────────── - # No end_session, no new row, no parent_session_id, no title - # renumber, no contextvar/env/logging re-sync. The session's - # id, title, cwd, /goal, and gateway routing all stay put. - # - # Durable, NON-DESTRUCTIVE replace: soft-archive the - # pre-compaction turns (active=0, kept on disk + FTS-searchable + - # recoverable) and insert `compressed` as the new live (active=1) - # set, atomically. `compressed` already carries the surviving - # tail (current-turn messages the compressor kept via - # protect_last_n), so we DON'T pre-flush here — a flush would - # INSERT current-turn rows that archive_and_compact would then - # archive alongside the rest (harmless but wasted writes). The - # live-context load filters active=1, so a resume reloads ONLY - # the compacted set; the original turns remain under the SAME id - # for search/recovery (Teknium review — keep one durable id - # WITHOUT destroying history, unlike a hard replace_messages). - # See #38763. + # In-place compaction: same session_id; soft-archive old turns (active=0, still + # searchable) + insert `compressed` atomically; no pre-flush (tail already in). from agent.context_compressor import ( PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY, ) - # Tail rows carried the _compaction_tail tag from - # compress() (#86366; popped above, before the size - # estimate): their originals must be archived as - # superseded duplicates (rewind-style), not compacted=1. - # Count against the FINAL list — salvage may have - # dropped rows. + # Tail rows tagged by compress() are archived as superseded duplicates, not + # compacted=1. Count against the FINAL list — salvage may have dropped rows. _tail_count = sum( 1 for m in compressed if id(m) in _tail_tagged_ids ) @@ -4913,390 +4537,37 @@ def compress_context( model_config_patch={ PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY: None, }, - watermark=_commit_watermark, - lock_holder=_lock_holder, + watermark=lease.watermark, + lock_holder=lease.holder, tail_count=_tail_count, ) split_status = "in_place_committed" - # Post-commit contract (#98450, mirrors - # _sync_micro_compact_to_db): archive_and_compact just - # durably wrote every dict in `compressed` as the new - # active set, but compress() returned marker-swept COPIES - # (_strip_persistence_markers, #57491). These exact dict - # instances become the live message list the caller keeps, - # so without the stamp the next _persist_session → - # _flush_messages_to_session_db_unlocked walk treats the - # whole compacted transcript as unpersisted and re-INSERTs - # it — the live set doubles on every compaction - # (~58K → ~512K tokens in production). + # compress() returned marker-swept copies; stamp them as persisted or the next + # flush re-INSERTs the whole compacted transcript, doubling the live set. from agent.context_compressor import ( stamp_db_persisted_markers, ) stamp_db_persisted_markers(compressed) - # Reset the flush identity set so the next turn's appends are - # diffed against the COMPACTED transcript: the compacted dicts - # are passed as conversation_history next turn and skipped by - # identity, so only genuinely new turn messages get appended - # (no dup of the summary, no resurrection of dropped turns). + # Reset flush identity set so next turn diffs against the COMPACTED transcript: + # only genuinely new messages append (no summary dup, no resurrected turns). agent._flushed_db_message_ids = set() - # Rotation-independent signal: the conversation was compacted in - # place (id unchanged). The gateway reads this (NOT an id-change - # diff) to re-baseline transcript handling. + # Rotation-independent signal; the gateway reads this (not an id diff) to + # re-baseline transcript handling. compacted_in_place = True else: - # ── Rotation (legacy): end this session, fork a continuation ─ - # Flush any un-persisted current-turn messages to the OLD - # session before ending it, so they survive in the preserved - # parent transcript (#47202). (In-place skips this — see above.) - # - # Pass the already-durable prefix as conversation_history so - # the flush skips it by identity (#68196). Preflight - # compression runs BEFORE the normal turn flush has stamped - # the cold-resumed history dicts with _DB_PERSISTED_MARKER, so - # without a boundary _flush_messages_to_session_db treats every - # restored row as new and re-appends the whole transcript to - # the parent. turn_context anchors _persist_user_message_idx at - # the current-turn user message before preflight runs, so - # messages[:idx] is exactly the persisted prefix; only the - # current turn's new messages get written. - # - # Bound to old_session_id, hoisted above the flush: the - # ``except`` handler below keys its in-memory rollback off - # this name, so anything that fails from here on rolls the - # transcript back instead of leaving the failed attempt's - # compacted snapshot in place. + # Bind old_session_id first: it is the rollback key in the handler below. old_session_id = agent.session_id - current_idx = getattr(agent, "_persist_user_message_idx", None) - persisted_history = ( - messages[:current_idx] - if isinstance(current_idx, int) - and 0 <= current_idx <= len(messages) - else None + _publish_rotated_compaction( + agent, + messages, + compressed, + new_system_prompt=new_system_prompt, + lease=lease, + old_session_id=old_session_id, + compressed_user_turn_outcome=compressed_user_turn_outcome, ) - # The #47202 flush below is a DURABLE append to the parent - # and it is NOT undone when the rotation aborts: the - # ``except`` handler restores the in-memory transcript and - # keeps agent.session_id on the parent, but the rows it just - # wrote stay. Survivable for a one-off failure; pathological - # for a STICKY one. A parent row that already carries - # ``ended_at`` fails publish_compression_child on every - # attempt and nothing in this path clears it, so each - # auto-compaction appends another copy of the current turn to - # the transcript it was supposed to shrink — the session grows - # until the provider rejects the request outright (#88197: - # 303 unique messages stored as 2,611 rows after 7 aborted - # attempts, ~1.66M tokens, HTTP 400). - # - # So check that one precondition BEFORE writing. It is a plain - # read of the row the publish is about to read anyway, it - # raises the publish's own message so the log line, telemetry - # and rollback path are all unchanged, and it cannot mask a - # real rotation — a live parent reaches the flush exactly as - # before. Deliberately NOT extended to the compression lease: - # a lease is re-acquirable, so a transient miss here would - # abort a rotation that would otherwise have committed. - # - # AUTOMATIC stamps (tui_shutdown, ws_disconnect, orphan - # reap, idle/LRU evict — is_automatic_end_reason) do NOT - # trip this guard: publish_compression_child treats them - # as stale-by-construction and clears them in its own - # transaction (#88197 Bug 1), so aborting here would keep - # rotation wedged on exactly the stamp the publish can - # heal. Only deliberate boundaries (compression, - # session_reset, explicit close) abort before the flush. - _parent_row_reader = getattr(agent._session_db, "get_session", None) - _parent_already_ended = False - if callable(_parent_row_reader): - try: - from hermes_state_common import is_automatic_end_reason - - _parent_row = _parent_row_reader(old_session_id) or {} - _parent_already_ended = ( - _parent_row.get("ended_at") is not None - and not is_automatic_end_reason( - _parent_row.get("end_reason") - ) - ) - except Exception: - # Fail OPEN: an unreadable row must not turn a cheap - # guard into a new way to lose compression. - _parent_already_ended = False - if _parent_already_ended: - raise RuntimeError( - f"Compression parent already ended: {old_session_id}" - ) - # Foreign-tail ceiling (#75316): the flush below writes OUR - # OWN input transcript to the parent — those rows are - # already represented in the compacted handoff and must - # not be cloned into the child. Everything at or below - # this MAX(id) but above the start-watermark is a foreign - # concurrent append; everything above it is our flush. - try: - _foreign_tail_ceiling = ( - agent._session_db.get_active_message_watermark( - agent.session_id - ) - ) - except Exception: - # Without a trustworthy ceiling the clone could - # duplicate the handoff — fall back to historical - # behavior (no tail preservation this rotation). - _foreign_tail_ceiling = None - try: - agent._flush_messages_to_session_db( - messages, - conversation_history=persisted_history, - ) - except Exception: - pass # best-effort — don't block compression on a flush error - # Publish parent closure + child row + compacted handoff in - # one transaction. No reader can observe a missing/empty child. - # The rotation child must stay on the parent's profile — - # mirror _ensure_db_session's stamp ("default" persists as - # NULL). publish_compression_child additionally COALESCEs - # from the parent row, covering app-global remote sessions - # whose thread lacks the HERMES_HOME context. - try: - from hermes_cli.profiles import get_active_profile_name - - _profile_for_child = get_active_profile_name() - if _profile_for_child == "default": - _profile_for_child = None - except Exception: - _profile_for_child = None - old_title = agent._session_db.get_session_title(agent.session_id) - new_session_id = ( - f"{datetime.now().strftime('%Y%m%d_%H%M%S')}_" - f"{uuid.uuid4().hex[:6]}" - ) - from agent.context_compressor import _DB_PERSISTED_MARKER - agent._session_db.publish_compression_child( - parent_session_id=old_session_id, - child_session_id=new_session_id, - source=agent.platform - or os.environ.get("HERMES_SESSION_SOURCE", "cli"), - model=agent.model, - model_config=agent._session_init_model_config, - system_prompt=new_system_prompt, - messages=compressed, - cwd=getattr(agent, "working_directory", None), - profile_name=_profile_for_child, - compression_lock_holder=_lock_holder, - require_compression_lease=_lock_holder is not None, - require_lease_refresh=_lock_holder is not None, - lease_ttl_seconds=_lock_ttl, - watermark=( - _commit_watermark - if _foreign_tail_ceiling is not None - else None - ), - watermark_ceiling=_foreign_tail_ceiling, - ) - # For the `already_present` outcome the live-dict stamping is - # handled by the run_agent _compress_context wrapper's - # _sync_persisted_markers (it mirrors the handoff stamps back - # to the live lists by scoped identity). This branch only - # covers inserted/merged; compress_context is intentionally - # NOT self-contained for already_present — a future direct - # caller of compress_context must go through that wrapper. - if compressed_user_turn_outcome in {"inserted", "merged"}: - # The published child represents exactly one live row: - # the anchor source _ensure_compressed_has_user_turn - # inserted (verbatim) or merged (as the leading - # content) into the handoff. Stamp THAT row, not an - # index that may have drifted (adoption rebind at - # :3028/:3045 rebinds _persist_user_message_idx to - # len(messages) — deliberately OUT OF RANGE — or - # reanchor_current_turn_user_idx falling back to a - # user-role neighbor). The persist index is - # deliberately NOT read here: trusting it is the - # drift vector. _messages_match_scoped_identity is - # intentionally NOT used against the HANDOFF row: for - # `merged` the handoff is a superset (anchor + - # scaffolding), so a verbatim match would break the - # legitimate merged stamp. Live views are compared - # against the ANCHOR SOURCE only. - _compressed_anchor_source = None - for _reversed_message in reversed(messages): - if _is_real_user_message(_reversed_message): - _compressed_anchor_source = _reversed_message - break - if isinstance(_compressed_anchor_source, dict): - _compressed_anchor_source[_DB_PERSISTED_MARKER] = True - _session_messages = getattr( - agent, "_session_messages", None - ) - if ( - isinstance(_session_messages, list) - and _session_messages is not messages - ): - # Adoption divergence: `messages` now points - # at the adopted durable snapshot while - # agent._session_messages may still point at - # the pre-adoption live list (comment - # :3040-3044), and the persist index is OUT - # OF RANGE for the twin by design. The - # post-publish _sync_persisted_markers cannot - # mirror a MERGED handoff either (anchor + - # scaffolding superset has no scoped match - # with the standalone live row). Locate the - # twin's corresponding live row(s) by scoped - # identity AGAINST THE ANCHOR SOURCE and - # stamp them — mirroring the wrapper's "stamp - # every scoped match" pattern for - # timestamp-less ambiguity - # (run_agent.py:8237-8281). - _anchor_timestamp = _compressed_anchor_source.get( - "timestamp" - ) - _found_exact_timestamp_candidate = False - if _anchor_timestamp is not None: - for _twin_message in _session_messages: - if ( - isinstance(_twin_message, dict) - and _twin_message.get("timestamp") - == _anchor_timestamp - and _messages_match_scoped_identity( - _twin_message, - _compressed_anchor_source, - ) - ): - # An exact scoped twin EXISTS — - # count the hit REGARDLESS of the - # marker. An already-stamped exact - # twin (stamped by a prior - # flush/rotation) must still - # suppress the broad fallback: - # otherwise a timestamp-less - # historical duplicate with the - # same content would be stamped as - # a "twin" of the anchor — the - # exact wrong-row path the - # reviewer flagged. The stamp - # itself stays idempotent (skip - # rows already carrying the - # marker); only the HIT is - # marker-independent. - _found_exact_timestamp_candidate = True - if not _twin_message.get( - _DB_PERSISTED_MARKER - ): - _twin_message[ - _DB_PERSISTED_MARKER - ] = True - if not _found_exact_timestamp_candidate: - # No exact scoped twin exists anywhere (or - # the anchor itself is timestamp-less): - # stamp every scoped match — the wrapper's - # documented ambiguity policy for - # timestamp-less anchors. This branch is - # reached ONLY when an exact hit is - # provably absent; an exact hit that is - # already stamped does NOT open this - # branch. - for _twin_message in _session_messages: - if ( - isinstance(_twin_message, dict) - and not _twin_message.get( - _DB_PERSISTED_MARKER - ) - and _messages_match_scoped_identity( - _twin_message, - _compressed_anchor_source, - ) - ): - _twin_message[ - _DB_PERSISTED_MARKER - ] = True - for _handoff_message in compressed: - if isinstance(_handoff_message, dict): - _handoff_message[_DB_PERSISTED_MARKER] = True - agent.session_id = new_session_id - agent._db_flush_scan_prefix = None - try: - from gateway.session_context import set_current_session_id - - set_current_session_id(agent.session_id) - except Exception: - os.environ["HERMES_SESSION_ID"] = agent.session_id - try: - from hermes_logging import set_session_context - - set_session_context(agent.session_id) - except Exception: - pass - agent._session_db_created = True split_status = "rotated_committed" - # Carry a persistent /goal onto the continuation session. - # Compression mints a fresh child id; load_goal does a flat - # per-session lookup with no parent walk, so without this an - # active goal silently dies at the boundary (#33618). - try: - from hermes_cli.goals import migrate_goal_to_session - migrate_goal_to_session(old_session_id, agent.session_id, reason="compression") - except Exception as _goal_err: - logger.debug("Could not migrate goal on compression: %s", _goal_err) - # Same boundary hazard for /heartbeat state — carry it too. - try: - from hermes_cli.heartbeat import migrate_heartbeat_to_session - migrate_heartbeat_to_session(old_session_id, agent.session_id) - except Exception as _hb_err: - logger.debug("Could not migrate heartbeat on compression: %s", _hb_err) - # Same boundary hazard for a persistent /loop — carry it - # onto the continuation session so the recurring wakeups - # survive compression. - try: - from hermes_cli.loops import migrate_loop_to_session - migrate_loop_to_session(old_session_id, agent.session_id, reason="compression") - except Exception as _loop_err: - logger.debug("Could not migrate loop on compression: %s", _loop_err) - # Carry the title across the compression boundary unchanged. - # - # This used to renumber ("Fix X" → "Fix X #2") on every - # rotation, which is why a long conversation ended up as - # "Smallville Map Architecture Plan #10" — ten forks of ONE - # session, each looking like a separate piece of work in the - # sidebar. Compression is an internal implementation detail; - # the user's conversation did not change topic, so its name - # must not change either. Uniqueness still holds because - # _set_session_title transfers the title off a hidden - # compression ancestor rather than raising on the conflict. - if old_title: - # Read provenance BEFORE the write: transferring the - # title off a hidden compression ancestor clears the - # ancestor's row, so reading afterwards always returns - # None and the child would be stamped "user" — freezing - # an auto-title that should still be upgradeable. - _src = None - try: - _src = agent._session_db.get_session_title_source( - old_session_id - ) - except Exception as _src_err: - logger.debug( - "Could not read title provenance: %s", _src_err - ) - try: - agent._session_db.set_session_title( - agent.session_id, old_title - ) - except (ValueError, Exception) as e: - logger.debug("Could not propagate title on compression: %s", e) - else: - # set_session_title() records "user"; restore the - # original authority so an inherited auto-title - # stays upgradeable and a manual one stays pinned. - if _src is not None: - try: - agent._session_db.set_session_title_source( - agent.session_id, _src - ) - except Exception as _src_err: - logger.debug( - "Could not propagate title provenance: %s", - _src_err, - ) # In-place mode still updates/replaces the current row here. # Rotation already published prompt + compacted handoff atomically. @@ -5312,108 +4583,49 @@ def compress_context( except Exception as e: if ( not in_place - and locals().get("old_session_id") + and old_session_id and agent.session_id == old_session_id ): # Atomic publication failed (including lease loss): keep the # parent live and discard the stale compacted snapshot. old_session_id = None - # NOTE: _db_flush_scan_prefix is intentionally NOT cleared - # here. The flush's bounded scan is identity-based - # (messages[i] is prefix[i]); the deepcopy rollback below - # replaces every row, so a stale prefix can never - # identity-match again. A failed parent flush clears its own - # prefix, and on the snapshot path the live list is - # untouched — do not "restore" a clear here without - # re-checking those invariants. + # _db_flush_scan_prefix is intentionally NOT cleared: the scan is identity-based + # and the deepcopy replaces every row. A failed parent flush clears its own; the + # snapshot path leaves the live list untouched. Recheck both before adding one. messages[:] = copy.deepcopy(messages_before_compression) compressed = messages _compression_made_progress = False - # Restore ONLY the prune runway, not the full attempt - # snapshot: _restore_compressor_attempt_state is reserved - # for pre-commit cancels (fence deny / explicit cancel), - # while this branch is post-attempt — the other snapshot - # fields (telemetry, aborted flags) must keep the failed - # attempt's values. The runway is a property of transcript - # state, and the transcript was just rolled back to its - # pre-compression copy, so the runway rolls back with it. - # (compress() zeroed it in-memory on summary success; the - # durable copy was never cleared — that clear only rides - # the atomic archive_and_compact / child-row publication - # that just failed.) - if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot: - agent.context_compressor._proactive_prune_rearm_tokens = ( - _compressor_attempt_snapshot[ - "_proactive_prune_rearm_tokens" - ] - ) + # Only the runway rolls back: the full snapshot restore is for pre-commit + # cancels (telemetry keeps failed values). + _restore_prune_rearm_tokens(agent.context_compressor, _compressor_attempt_snapshot) elif ( in_place and split_status != "in_place_committed" and messages_before_compression is not None ): - # In-place sibling of the rotation rollback above (#99477). - # archive_and_compact() is atomic, so a raise before it - # returned means EVERY pre-compaction row is still - # ``active = 1`` in state.db — nothing was archived and the - # compacted set was never inserted. But ``compressed`` is - # the marker-swept output of compress() - # (_strip_persistence_markers, #57491) and the post-commit - # ``stamp_db_persisted_markers`` never ran, so handing it - # back makes the next append-only flush treat the whole - # compacted transcript as new and INSERT it ON TOP of the - # rows it was supposed to replace. The active set then holds - # the summary AND the turns it summarized; the next resume - # reloads both, the token count goes UP, preflight fires - # again, and each failed attempt appends another copy of the - # protected head + tail (#99477: ~15 real turns stored as - # 3,814 rows, the first user message repeated 893 times). - # - # Gate on ``split_status`` rather than ``compacted_in_place``: - # it is assigned on the statement immediately after the - # atomic commit returns, so a committed compaction can never - # be rolled back into a live/durable mismatch of the - # opposite sign. - # - # The deepcopy carries each row's _DB_PERSISTED_MARKER from - # the pre-compression snapshot, so the restored transcript is - # correctly skipped by the flush, and replacing every dict - # breaks _db_flush_scan_prefix identity (same reasoning as - # the rotation branch — no explicit clear needed). + # In-place rollback: archive_and_compact is atomic so old rows stay active, but + # marker-swept `compressed` would re-INSERT on top of them (doubling each try). + # Gate on split_status (set right after commit); deepcopy keeps markers/identity messages[:] = copy.deepcopy(messages_before_compression) compressed = messages _compression_made_progress = False - # Runway rolls back with the transcript, exactly as above: - # compress() zeroed it in memory, and the durable clear only - # rides the archive_and_compact that just failed. - if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot: - agent.context_compressor._proactive_prune_rearm_tokens = ( - _compressor_attempt_snapshot[ - "_proactive_prune_rearm_tokens" - ] - ) + _restore_prune_rearm_tokens(agent.context_compressor, _compressor_attempt_snapshot) split_status = ( "aborted" - if locals().get("old_session_id") is None and not in_place + if old_session_id is None and not in_place else "failed_not_indexed" ) - # If the rotation rolled back to the parent (orphan-avoidance - # above), agent.session_id is the still-indexed parent and - # old_session_id was cleared — so this is recovery, not an - # un-indexed orphan. Otherwise an earlier step failed before the - # child was created and the warning's original meaning holds. - if locals().get("old_session_id") is None and not in_place: + # If rotation rolled back to the parent, agent.session_id is the indexed parent + # and old_session_id was cleared: recovery, not an un-indexed orphan. + if old_session_id is None and not in_place: logger.warning( "Compression rotation aborted and rolled back to the " "parent session (%s): %s", agent.session_id or "?", e, ) else: logger.warning("Session DB compression split failed — new session will NOT be indexed: %s", e) - # Arm the failure cooldown so the next turn cannot immediately - # re-run the identical doomed compression (#97948 symptom B). - # try/except mirrors the sibling record_rejected_compaction - # call above: this runs inside the split-failure handler, and - # a stub compressor must not mask the original error. + # Arm the failure cooldown so the next turn can't rerun the doomed compression; + # try/except so a stub compressor can't mask the original error in this handler. try: agent.context_compressor._record_compression_failure_cooldown( _SPLIT_FAILURE_COOLDOWN_SECONDS, @@ -5425,181 +4637,31 @@ def compress_context( exc_info=True, ) - # Compaction-boundary bookkeeping, computed once. `old_session_id` is only - # bound in the rotation branch; in-place leaves it unset. `_boundary_parent` - # is the id the boundary notifications attribute the prior state to: the old - # id on rotation, the (unchanged) current id in-place. - _old_sid = locals().get("old_session_id") - _is_boundary = bool(_old_sid) or in_place - _context_engine_boundary_committed = _session_commit_succeeded and ( - bool(_old_sid) or compacted_in_place - ) - _boundary_parent = _old_sid or agent.session_id or "" - - # Round-2 #4: the activity heartbeat's terminal "context compression - # completed" stamp landed on the PARENT row (force-persisted before - # the rotation re-pointed agent.session_id at the child). Without a - # cleanup, the archived parent advertises a fresh last_activity_at + - # "context compression completed" forever — a permanent false-fresh - # row for any activity consumer that scans ended sessions. Clear the - # labels on the parent best-effort (keeps last_activity_at so idle - # clocks stay continuous; the CHILD carries the live labels). - if _old_sid and _session_commit_succeeded: - try: - _labels_db = getattr(agent, "_session_db", None) - _clear_labels = getattr( - type(_labels_db) if _labels_db is not None else None, - "clear_session_activity_labels", - None, - ) - if callable(_clear_labels): - _clear_labels(_labels_db, _old_sid) - except Exception: - logger.debug( - "failed to clear archived compression parent's activity " - "labels (ignored)", - exc_info=True, - ) - - # Notify the context engine that a compaction boundary occurred. Plugin - # engines (e.g. hermes-lcm) use boundary_reason="compression" to preserve - # DAG lineage / checkpoint per-session state across the boundary instead of - # re-initializing fresh. See hermes-lcm#68. Built-in ContextCompressor - # ignores kwargs. Fires in BOTH modes: rotation passes old→new ids; in-place - # passes the SAME id (the boundary is real even though the id didn't move). - if _context_engine_boundary_committed: - if defer_context_engine_notification: - _queue_context_engine_compression_notification( - agent, - new_session_id=agent.session_id or "", - old_session_id=_boundary_parent, - ) - else: - _notify_context_engine_compression_complete( - agent, - new_session_id=agent.session_id or "", - old_session_id=_boundary_parent, - ) - - # Notify memory providers of the compaction boundary so provider-cached - # per-session state (Hindsight's _document_id, accumulated turn buffers, - # counters) refreshes. reset=False because the logical conversation - # continues. See #6672. Fires in BOTH modes: in-place uses the same id as - # parent (the conversation didn't fork, but the buffer must still be told - # the transcript was compacted so it doesn't double-count dropped turns). - try: - if _is_boundary and agent._memory_manager: - agent._memory_manager.on_session_switch( - agent.session_id or "", - parent_session_id=_boundary_parent, - reset=False, - reason="compression", - ) - except Exception as _me_err: - logger.debug("memory manager on_session_switch (compression): %s", _me_err) - - # Warn on repeated compressions (quality degrades with each pass). - # Route through _emit_status (like the other compression warnings above) - # so the warning reaches the TUI / Telegram / Discord via status_callback, - # not just CLI stdout. _emit_status still _vprints for the CLI, and - # storing it on _compression_warning lets replay_compression_warning - # re-deliver it once a late-bound gateway status_callback is wired (#36908). - _cc = agent.context_compressor.compression_count - if _cc >= 2: - _cc_msg = ( - f"{agent.log_prefix}⚠️ Session compressed {_cc} times — " - f"accuracy may degrade. Consider /new to start fresh." - ) - agent._compression_warning = _cc_msg - agent._emit_status(_cc_msg) - - # Emit session:compress event so hooks (e.g. MemPalace sync) can ingest - # the completed old session before its details are lost. In in-place mode - # there is no old id (same session); ``in_place=True`` tells hooks the - # transcript was compacted on the same id rather than rotated. - if getattr(agent, "event_callback", None): - try: - agent.event_callback("session:compress", { - "platform": agent.platform or "", - "session_id": agent.session_id, - "old_session_id": _old_sid or "", - "in_place": in_place, - "compression_count": agent.context_compressor.compression_count, - }) - except Exception as e: - logger.debug("event_callback error on session:compress: %s", e) - - # Surface the compaction mode to the caller (run_conversation / gateway) - # via a rotation-independent flag. The gateway uses this — NOT an - # id-change diff — to re-baseline transcript handling (history_offset=0 + - # rewrite on the same id) when compaction happened in place. See #38763. - agent._last_compression_attempt_in_place = compacted_in_place - agent._last_compaction_in_place = compacted_in_place - - # Keep the post-compression rough estimate for diagnostics, but do not - # treat it as provider-reported prompt usage. Schema-heavy rough estimates - # can remain above threshold even after the next real API request fits. - _compressed_est = estimate_request_tokens_rough( + _compressed_est = _finish_compaction_boundary( + agent, compressed, - system_prompt=new_system_prompt or "", - tools=agent.tools or None, + new_system_prompt=new_system_prompt, + old_session_id=old_session_id, + in_place=in_place, + compacted_in_place=compacted_in_place, + session_commit_succeeded=_session_commit_succeeded, + defer_context_engine_notification=defer_context_engine_notification, + compression_made_progress=_compression_made_progress, + compression_used_fallback=_compression_used_fallback, + compression_feasibility_skip=_compression_feasibility_skip, + task_id=task_id, ) - agent.context_compressor.last_compression_rough_tokens = _compressed_est - agent.context_compressor.last_prompt_tokens = -1 - agent.context_compressor.last_completion_tokens = 0 - agent.context_compressor.awaiting_real_usage_after_compression = True - # Compaction rewrote the transcript, so the usage anchor's base - # message-list snapshot no longer describes what will be sent — - # invalidate it. Context checks fall back to full estimation until - # the next response with usage re-anchors (its structural id/index - # check would also fail closed, but explicit is safer). - agent._usage_anchor = None - agent._turn_base_usage_anchor = None - # Arm the effectiveness verdict only after a completed rewrite crosses - # the full compaction boundary. Exceptions, aborts, and no-op attempts - # leave this false, so unrelated later usage cannot be charged to an - # attempt that never changed the transcript. - if _compression_made_progress: - record_boundary = getattr( - type(agent.context_compressor), - "record_completed_compaction", - None, - ) - if callable(record_boundary): - record_boundary( - agent.context_compressor, - used_fallback=_compression_used_fallback, - feasibility_skip=_compression_feasibility_skip, - ) - else: - agent.context_compressor._verify_compaction_cleared_threshold = True - - # Clear the file-read dedup cache. After compression the original - # read content is summarised away — if the model re-reads the same - # file it needs the full content, not a "file unchanged" stub. - try: - from tools.file_tools import reset_file_dedup - reset_file_dedup(task_id) - except Exception: - pass - # Same for the skill_view repeat-view dedup: a post-compression - # re-view must return the full skill content again. - try: - from tools.skills_tool import reset_skill_view_dedup - reset_skill_view_dedup(task_id) - except Exception: - pass logger.info( "context compression done: session=%s messages=%d->%d rough_tokens=~%s awaiting_real_usage=true", agent.session_id or "none", _pre_msg_count, len(compressed), f"{_compressed_est:,}", ) - _commit_status = "committed" if split_status in {"not_applicable", "in_place_committed", "rotated_committed"} else "aborted" + lifecycle.commit_status = "committed" if split_status in {"not_applicable", "in_place_committed", "rotated_committed"} else "aborted" _emit_compression_attempt_telemetry( agent, started_at=_attempt_started_at, - commit_status=_commit_status, + commit_status=lifecycle.commit_status, split_status=split_status, failure_class=( "session_split_failed" @@ -5610,13 +4672,10 @@ def compress_context( ) return compressed, new_system_prompt finally: - # Release the lock on the OLD session_id only AFTER rotation completed - # and all post-rotation bookkeeping (memory manager, context engine, - # file dedup) ran. A concurrent path that wakes up the moment we - # release will see the NEW session_id in state.db / SessionEntry and - # acquire on that — no race against our just-finished work. + # Release the OLD session's lock only after rotation and all post-rotation + # bookkeeping; a waking contender then sees the NEW id and acquires on that. try: - _release_lock() + lease.release() finally: if _commit_fence_entered: commit_fence.finish_commit() @@ -5642,13 +4701,10 @@ def _codex_compaction_cooldown_remaining(agent: Any) -> float: def _record_codex_compaction_failure(agent: Any, error: str) -> None: - """Arm the shared compression-failure cooldown after a failed compaction. + """Arm the shared compression-failure cooldown after a failed codex compaction. - The codex path returns the transcript unchanged on failure, so the session - is still above threshold and the next turn retries immediately. Every other - compression path records a cooldown, an ineffective-compression strike, or - both; this one recorded neither, so an interrupted compaction retried once - per turn for as long as the condition persisted. + The codex path returns the transcript unchanged, so without a cooldown the + still-over-threshold session would retry every turn. """ from agent.context_compressor import _SUMMARY_FAILURE_COOLDOWN_SECONDS @@ -5673,68 +4729,38 @@ def _compress_context_via_codex_app_server( ) -> Tuple[list, str]: """Route compaction to Codex app-server for Codex-owned threads. - Hermes' normal compressor rewrites the local OpenAI-style transcript. - That does not shrink the actual Codex app-server thread context. For this - runtime, ask Codex to compact its own thread and keep Hermes' transcript - unchanged. + Rewriting the local transcript would not shrink the Codex thread, so Codex + compacts its own thread and Hermes' transcript is left unchanged. """ + _sid = getattr(agent, "session_id", None) or "none" + _tokens = f"{approx_tokens:,}" if approx_tokens else "unknown" auto_mode = str( getattr(agent, "codex_app_server_auto_compaction", "native") or "native" ).lower() if auto_mode not in {"native", "hermes", "off"}: auto_mode = "native" + skip_reason = None if not force and auto_mode != "hermes": - logger.info( - "codex app-server compaction skipped: mode=%s force=false " - "(session=%s messages=%d tokens=~%s)", - auto_mode, - getattr(agent, "session_id", None) or "none", - len(messages), - f"{approx_tokens:,}" if approx_tokens else "unknown", - ) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) - return messages, existing_prompt - - # Automatic entrypoints must honor the compressor-owned cooldown, the same - # way the Hermes path below does. An active cooldown means a recent - # compaction already failed; retrying every turn is what thrashes. - if not force: + skip_reason = f"mode={auto_mode} force=false" + elif not force: + # Automatic entrypoints honor the compressor-owned cooldown: a recent compaction + # failed, and retrying every turn is what thrashes. _cooldown_remaining = _codex_compaction_cooldown_remaining(agent) if _cooldown_remaining > 0: - logger.info( - "codex app-server compaction skipped: failure cooldown active " - "for %.0fs (session=%s messages=%d tokens=~%s)", - _cooldown_remaining, - getattr(agent, "session_id", None) or "none", - len(messages), - f"{approx_tokens:,}" if approx_tokens else "unknown", - ) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) - return messages, existing_prompt - + skip_reason = f"failure cooldown active for {_cooldown_remaining:.0f}s" codex_session = getattr(agent, "_codex_session", None) - if codex_session is None: + if skip_reason is None and codex_session is None: + skip_reason = "no active codex thread" + if skip_reason is not None: logger.info( - "codex app-server compaction skipped: no active codex thread " - "(session=%s messages=%d tokens=~%s)", - getattr(agent, "session_id", None) or "none", - len(messages), - f"{approx_tokens:,}" if approx_tokens else "unknown", + "codex app-server compaction skipped: %s (session=%s messages=%d tokens=~%s)", + skip_reason, _sid, len(messages), _tokens, ) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) - return messages, existing_prompt + return messages, _existing_system_prompt(agent, system_message) logger.info( "codex app-server compaction started: session=%s messages=%d tokens=~%s", - getattr(agent, "session_id", None) or "none", - len(messages), - f"{approx_tokens:,}" if approx_tokens else "unknown", + _sid, len(messages), _tokens, ) try: agent._emit_status(COMPACTION_STATUS) @@ -5775,10 +4801,7 @@ def _compress_context_via_codex_app_server( agent, str(getattr(result, "error", None) or "compaction interrupted"), ) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) - return messages, existing_prompt + return messages, _existing_system_prompt(agent, system_message) try: from agent.codex_runtime import ( @@ -5792,10 +4815,8 @@ def _compress_context_via_codex_app_server( approx_tokens=approx_tokens, force=True, ) - # An empty usage report must consume the pending post-compaction verdict - # rather than leaving preflight deferral armed until some unrelated later - # Codex turn supplies usage. Minimal external test engines may not expose - # the ContextEngine update hook; preserve their existing bookkeeping. + # An empty usage report must consume the pending verdict, not leave deferral + # armed until a later turn; minimal test engines may lack update_from_response. if hasattr(agent.context_compressor, "update_from_response"): _record_codex_app_server_usage(agent, result) except Exception: @@ -5810,13 +4831,11 @@ def _compress_context_via_codex_app_server( logger.info( "codex app-server compaction done: session=%s thread=%s turn=%s", - getattr(agent, "session_id", None) or "none", + _sid, getattr(result, "thread_id", None) or "", getattr(result, "turn_id", None) or "", ) - existing_prompt = getattr(agent, "_cached_system_prompt", None) - if not existing_prompt: - existing_prompt = agent._build_system_prompt(system_message) + existing_prompt = _existing_system_prompt(agent, system_message) # Terminal edge only on success — failure/interrupt paths above return # without it, matching the main compress_context() gating. _emit_compaction_done(agent) @@ -5828,24 +4847,11 @@ def try_shrink_image_parts_in_messages( *, max_dimension: int = 8000, ) -> bool: - """Re-encode all native image parts at a smaller size to recover from - image-too-large errors (Anthropic 5 MB, unknown other providers). + """Re-encode oversized native image parts to recover from image-too-large errors. - Mutates ``api_messages`` in place. Returns True if any image part was - actually replaced, False if there were no image parts to shrink or - Pillow couldn't help (caller should surface the original error). - - Strategy: look for ``image_url`` / ``input_image`` parts carrying a - ``data:image/...;base64,...`` payload, plus Anthropic-native - ``{"type": "image", "source": {"type": "base64", ...}}`` blocks. - For each one whose encoded size exceeds 4 MB (a safe target that slides - under Anthropic's 5 MB ceiling with header overhead) or whose longest side - exceeds ``max_dimension``, write the base64 to a tempfile, call - ``vision_tools._resize_image_for_vision`` to produce a smaller data - URL, and substitute it in place. - - Non-data-URL images (http/https URLs) are not touched — the provider - fetches those itself and the size limit is different. + Mutates ``api_messages`` in place. Returns True if any part was replaced, + False if nothing to shrink or Pillow could not help. Targets data-URL parts + over 4 MB or ``max_dimension``; http(s) image URLs are left untouched. """ if not api_messages: return False @@ -5856,28 +4862,21 @@ def try_shrink_image_parts_in_messages( logger.warning("image-shrink recovery: vision_tools unavailable — %s", exc) return False - # 4 MB target leaves comfortable headroom under Anthropic's 5 MB. - # Non-Anthropic providers we haven't observed rejecting are fine with - # much larger; shrinking to 4 MB here loses quality but only fires - # after a confirmed provider rejection, so the alternative is failure. + # 4 MB leaves headroom under Anthropic's 5 MB; shrinking loses quality but only + # runs after a confirmed provider rejection, so the alternative is failure. target_bytes = 4 * 1024 * 1024 - # Anthropic enforces an 8000px per-side dimension cap independently of - # the 5 MB byte cap. In many-image requests, the provider can report a - # lower cap (observed: 2000px). The caller passes that parsed ceiling - # when the rejection includes it. + # Anthropic also caps per-side pixels (8000, or lower in many-image requests); + # the caller passes the parsed ceiling when the rejection includes it. changed_count = 0 - # Track parts that are over the target but could NOT be shrunk under it. - # If any survive, retrying is pointless — the same oversized payload will - # be re-sent and rejected again, wasting the single retry budget. We only - # report success (caller retries) when every over-threshold image was - # actually brought under the target. + # Track over-target parts that could not be shrunk: if any remain, a retry + # re-sends the same payload and wastes the single retry budget. unshrinkable_oversized = 0 def _decode_pixels(data_url: str) -> Optional[tuple]: """Return ``(width, height)`` of a base64 data URL, or None on failure. - Soft-depends on Pillow; returns None (caller falls back to a - bytes-only check) if Pillow is missing or the payload is corrupt. + None when Pillow is missing or the payload is corrupt; caller falls back to a + bytes-only check. """ try: import base64 as _b64_dim @@ -5894,29 +4893,19 @@ def try_shrink_image_parts_in_messages( def _shrink_data_url(url: str) -> tuple: """Return ``(resized_url, unshrinkable)`` for a data URL. - ``resized_url`` is a smaller/dimension-correct data URL, or None when - no rewrite was applied. ``unshrinkable`` is True only when the image - exceeded a constraint (byte-size or dimensions) and the resize failed - to satisfy *that same* constraint — so the caller knows retrying is - pointless even if a different image in the request shrank. + ``resized_url`` is None when no rewrite applied. ``unshrinkable`` is True only + when the image violated a constraint and resizing failed to satisfy that same + constraint, so the caller knows a retry is pointless. """ if not isinstance(url, str) or not url.startswith("data:"): return None, False - # Determine which constraint is binding. The accept/reject gate below - # MUST be checked against the same axis that triggered the shrink: a - # downscaled screenshot PNG routinely re-encodes to *more* bytes than - # the original (PNG compression is non-monotonic in image size — a - # smaller raster with LANCZOS resampling noise compresses worse than a - # larger smooth one). Rejecting a pixel-correct downscale purely - # because its bytes grew permanently wedges sessions on the Anthropic - # many-image 2000px path (#48013). + # The accept gate MUST use the axis that triggered the shrink: a pixel downscale + # can re-encode to MORE bytes (PNG non-monotonic); byte-only reject wedges. needs_shrink = len(url) > target_bytes # over byte budget triggered_by = "bytes" if needs_shrink else None if not needs_shrink: - # Bytes are fine — check pixel dimensions against the provider's - # reported per-side cap. A screenshot can be tiny in bytes yet - # too large in pixels. + # Bytes fine; check pixels against the provider cap (tiny bytes, huge pixels). dims = _decode_pixels(url) if dims is None: # Pillow missing or corrupt data — fall back to byte-only. @@ -5963,32 +4952,21 @@ def try_shrink_image_parts_in_messages( # Byte budget is the binding constraint — bytes must shrink. if len(resized) >= len(url): return None, True # re-encode made it bigger - # The per-side dimension cap is ALSO an active provider - # constraint on this request (the caller passes the parsed cap - # to both this helper and the resizer). _resize_image_for_vision - # returns a best-effort, possibly-over-cap blob when it - # exhausts its halving budget — it freezes the long side once - # the short side hits its 64px floor, so a very-high-aspect - # image can stay over the cap even after bytes shrank. If the - # output is still over the cap, retrying would re-400 on - # dimensions; treat it as unshrinkable. (Skip when dims can't - # be decoded — preserves historical byte-only behaviour.) + # The resizer may return an over-cap blob (long side freezes at the 64px short- + # side floor); still over cap → re-400, so unshrinkable. Undecodable dims: skip. new_dims = _decode_pixels(resized) if new_dims is not None and max(new_dims) > max_dimension: return None, True return resized, False - # triggered_by == "dimension": the per-side cap is binding. The - # re-encode may have grown in bytes; accept it as long as it is now - # within the dimension cap. Verify the new dimensions when we can. + # Dimension cap is binding: accept a byte-larger re-encode if now within cap. new_dims = _decode_pixels(resized) if new_dims is not None: if max(new_dims) <= max_dimension: return resized, False # Still over the per-side cap — the resize didn't satisfy it. return None, True - # Couldn't verify the re-encode's dimensions (corrupt output or - # Pillow gone mid-call). Fall back to the historical "bytes must - # shrink" gate so we never accept an unverifiable, byte-larger blob. + # Can't verify dimensions: fall back to the bytes-must-shrink gate so we never + # accept an unverifiable byte-larger blob. if len(resized) >= len(url): return None, True return resized, False @@ -6010,12 +4988,8 @@ def try_shrink_image_parts_in_messages( def _write_data_url_to_source(source: dict, data_url: str) -> dict: """Return a NEW source dict carrying the re-encoded payload. - Copy-on-write: content parts on the per-call ``api_messages`` list may - be shared references into the persistent conversation history (the - per-message copy is shallow, and cache decoration only deep-copies the - marked messages). Mutating the existing dict would rewrite the stored - transcript with the degraded image — so the caller replaces the part, - never edits it in place. + Copy-on-write: parts may be shared with the persistent history, so mutating + in place would store the degraded image; the caller replaces the part. """ header, _, data = data_url.partition(",") media_type = "image/jpeg" @@ -6036,11 +5010,8 @@ def try_shrink_image_parts_in_messages( content = msg.get("content") if not isinstance(content, list): continue - # Copy-on-write per message: never mutate part/source dicts in place — - # they can alias the stored conversation history (see - # _write_data_url_to_source). Build a replacement content list on the - # first shrunken part and reassign msg["content"] (a top-level write on - # the per-call message copy, which never reaches history). + # Copy-on-write: part/source dicts can alias stored history, so build a new + # content list and reassign msg["content"] on the per-call copy. new_content: list | None = None for part_idx, part in enumerate(content): if not isinstance(part, dict): @@ -6097,10 +5068,8 @@ def try_shrink_image_parts_in_messages( changed_count, target_bytes / (1024 * 1024), ) if unshrinkable_oversized: - # At least one oversized image could not be shrunk under the target. - # Retrying would re-send it and fail identically, so signal "no - # progress" even if other parts shrank — the caller will surface the - # original error rather than burning its single retry on a no-op. + # An unshrinkable oversized image makes retry pointless; signal no progress even + # if others shrank so the caller surfaces the original error. logger.warning( "image-shrink recovery: %d oversized image part(s) could not be " "shrunk under %.0f MB — not retrying (would re-send rejected payload)",