From 806d2c32746cea026622b08053b631dfbdbd7349 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:24:29 -0700 Subject: [PATCH 01/27] refactor(agent/context_compressor): dedupe image-part stripping, summarizer one-liners into dispatch table, alias coerce helper --- agent/context_compressor.py | 148 ++++++++++-------------------------- 1 file changed, 39 insertions(+), 109 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 99c839739e..11fd67dbe4 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -194,8 +194,6 @@ def _is_summary_access_or_quota_error(exc: Exception) -> bool: if status in {401, 402, 403}: return True - if classified.reason is FailoverReason.billing: - return any(marker in err_text for marker in _SUMMARY_PERMANENT_QUOTA_MARKERS) return any(marker in err_text for marker in _SUMMARY_PERMANENT_QUOTA_MARKERS) @@ -1336,26 +1334,11 @@ def _append_text_to_content(content: Any, text: str, *, prepend: bool = False) - return text + rendered if prepend else rendered + text -def _strip_image_parts_from_parts(parts: Any) -> Any: - """Strip image parts from an OpenAI-style content-parts list. - - Returns a new list with text placeholders, or None if the list had no images. - """ - if not isinstance(parts, list): +def _replace_image_parts(parts: Any, placeholder: str) -> Optional[List[Any]]: + """New parts list with every image part replaced by a text placeholder; None if no images.""" + if not isinstance(parts, list) or not any(_is_image_part(p) for p in parts): return None - had_image = False - out = [] - for part in parts: - if not isinstance(part, dict): - out.append(part) - continue - ptype = part.get("type") - if ptype in {"image", "image_url", "input_image"}: - had_image = True - out.append({"type": "text", "text": "[screenshot removed to save context]"}) - else: - out.append(part) - return out if had_image else None + return [{"type": "text", "text": placeholder} if _is_image_part(p) else p for p in parts] def _tool_content_has_images(content: Any) -> bool: @@ -1377,7 +1360,7 @@ def _strip_images_from_tool_msg(msg: Dict[str, Any]) -> Optional[Dict[str, Any]] new_msg = {**msg, "content": f"[screenshot removed] {str(summary)[:200]}"} drop_stale_api_content(new_msg) return new_msg - stripped = _strip_image_parts_from_parts(content) + stripped = _replace_image_parts(content, "[screenshot removed to save context]") if stripped is None: return None new_msg = {**msg, "content": stripped} @@ -1459,25 +1442,9 @@ def _content_has_images(content: Any) -> bool: def _strip_images_from_content(content: Any) -> Any: - """Return a copy of ``content`` with every image part replaced by a text placeholder. - - Non-list content is returned unchanged. Input is never mutated. - """ - if not isinstance(content, list): - return content - if not any(_is_image_part(p) for p in content): - return content - - new_parts: List[Any] = [] - for p in content: - if _is_image_part(p): - new_parts.append({ - "type": "text", - "text": "[Attached image — stripped after compression]", - }) - else: - new_parts.append(p) - return new_parts + """``content`` with image parts replaced by placeholders; unchanged (same object) when none.""" + stripped = _replace_image_parts(content, "[Attached image — stripped after compression]") + return content if stripped is None else stripped def _strip_historical_media(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: @@ -1490,33 +1457,18 @@ def _strip_historical_media(messages: List[Dict[str, Any]]) -> List[Dict[str, An if not messages: return messages - # Anchor on image-bearing user messages (not all) so a text follow-up still - # strips the old image. - anchor = -1 - for i in range(len(messages) - 1, -1, -1): - msg = messages[i] - if not isinstance(msg, dict): - continue - if msg.get("role") != "user": - continue - if _content_has_images(msg.get("content")): - anchor = i - break + def _newest(role: str, has_images) -> int: + for i in range(len(messages) - 1, -1, -1): + msg = messages[i] + if isinstance(msg, dict) and msg.get("role") == role and has_images(msg.get("content")): + return i + return -1 - # Tool-result images age on their own timeline: keep only the newest one, - # wherever it sits (the user anchor never protects stale ones). - tool_anchor = -1 - for i in range(len(messages) - 1, -1, -1): - msg = messages[i] - if not isinstance(msg, dict): - continue - if msg.get("role") != "tool": - continue - # Envelope-aware matcher so the native {_multimodal: True} dict shape - # anchors here too, otherwise rule 2 strips it as stale. - if _tool_content_has_images(msg.get("content")): - tool_anchor = i - break + # Anchor on image-bearing user messages (not all) so a text follow-up still strips the old image. + anchor = _newest("user", _content_has_images) + # Tool-result images age on their own timeline: keep only the newest one, wherever it sits. + # Envelope-aware matcher so the native {_multimodal: True} dict shape anchors too. + tool_anchor = _newest("tool", _tool_content_has_images) if anchor <= 0 and tool_anchor < 0: # Nothing to strip under any rule. @@ -1615,10 +1567,6 @@ def _sum_terminal(name, args, content, content_len, line_count): return f"[terminal] ran `{cmd}` -> exit {exit_code}, {line_count} lines output" -def _sum_read_file(name, args, content, content_len, line_count): - return f"[read_file] read {args.get('path', '?')} from line {args.get('offset', 1)} ({content_len:,} chars)" - - def _sum_write_file(name, args, content, content_len, line_count): written_lines = _str_arg(args, "content").count("\n") + 1 if args.get("content") else "?" return f"[write_file] wrote to {args.get('path', '?')} ({written_lines} lines)" @@ -1633,10 +1581,6 @@ def _sum_search_files(name, args, content, content_len, line_count): ) -def _sum_patch(name, args, content, content_len, line_count): - return f"[patch] {args.get('mode', 'replace')} in {args.get('path', '?')} ({content_len:,} chars result)" - - def _sum_browser(name, args, content, content_len, line_count): url = args.get("url", "") ref = args.get("ref", "") @@ -1644,10 +1588,6 @@ def _sum_browser(name, args, content, content_len, line_count): return f"[{name}]{detail} ({content_len:,} chars)" -def _sum_web_search(name, args, content, content_len, line_count): - return f"[web_search] query='{args.get('query', '?')}' ({content_len:,} chars result)" - - def _sum_web_extract(name, args, content, content_len, line_count): urls = args.get("urls", []) first = urls[0] if isinstance(urls, list) and urls else "?" @@ -1686,18 +1626,6 @@ def _sum_skill_view(name, args, content, content_len, line_count): return f"[skill_view] name={skill} ({content_len:,} chars)" -def _sum_named(name, args, content, content_len, line_count): - return f"[{name}] name={args.get('name', '?')} ({content_len:,} chars)" - - -def _sum_vision_analyze(name, args, content, content_len, line_count): - return f"[vision_analyze] '{_str_arg(args, 'question')[:50]}' ({content_len:,} chars)" - - -def _sum_memory(name, args, content, content_len, line_count): - return f"[memory] {args.get('action', '?')} on {args.get('target', '?')}" - - def _sum_clarify(name, args, content, content_len, line_count): response_prefix = "[clarify] user responded: " # Strictly below _PRUNE_MIN_CHARS so the summary survives later prune passes via the @@ -1731,38 +1659,48 @@ def _sum_clarify(name, args, content, content_len, line_count): return "[clarify] asked user a question" -def _sum_process_manage(name, args, content, content_len, line_count): - return f"[process] {args.get('action', '?')} session={args.get('session_id', '?')}" +def _sum_named(name, args, content, content_len, line_count): + return f"[{name}] name={args.get('name', '?')} ({content_len:,} chars)" # tool_name -> (name, args, content, content_len, line_count) -> one-line summary. _TOOL_RESULT_SUMMARIZERS = { "terminal": _sum_terminal, - "read_file": _sum_read_file, + "read_file": lambda name, args, content, content_len, line_count: ( + f"[read_file] read {args.get('path', '?')} from line {args.get('offset', 1)} ({content_len:,} chars)" + ), "write_file": _sum_write_file, "search_files": _sum_search_files, - "patch": _sum_patch, + "patch": lambda name, args, content, content_len, line_count: ( + f"[patch] {args.get('mode', 'replace')} in {args.get('path', '?')} ({content_len:,} chars result)" + ), **{ _b: _sum_browser for _b in ("browser_navigate", "browser_click", "browser_snapshot", "browser_type", "browser_scroll", "browser_vision") }, - "web_search": _sum_web_search, + "web_search": lambda name, args, content, content_len, line_count: ( + f"[web_search] query='{args.get('query', '?')}' ({content_len:,} chars result)" + ), "web_extract": _sum_web_extract, "delegate_task": _sum_delegate_task, "execute_code": _sum_execute_code, "skill_view": _sum_skill_view, "skills_list": _sum_named, "skill_manage": _sum_named, - "vision_analyze": _sum_vision_analyze, - "memory": _sum_memory, + "vision_analyze": lambda name, args, content, content_len, line_count: ( + f"[vision_analyze] '{_str_arg(args, 'question')[:50]}' ({content_len:,} chars)" + ), + "memory": lambda name, args, *_: f"[memory] {args.get('action', '?')} on {args.get('target', '?')}", "todo_list": lambda *a: "[todo] updated task list", "clarify": _sum_clarify, "text_to_speech": lambda name, args, content, content_len, line_count: ( f"[text_to_speech] generated audio ({content_len:,} chars)" ), "cronjob_manage": lambda name, args, *_: f"[cronjob] {args.get('action', '?')}", - "process_manage": _sum_process_manage, + "process_manage": lambda name, args, *_: ( + f"[process] {args.get('action', '?')} session={args.get('session_id', '?')}" + ), } @@ -2578,16 +2516,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return None return ivalue if ivalue > 0 else None - @staticmethod - def _coerce_threshold_tokens_cap(value: Any) -> int | None: - """Normalize a threshold_tokens cap to a positive int, or None for "no cap".""" - if value is None: - return None - try: - ivalue = int(value) - except (TypeError, ValueError): - return None - return ivalue if ivalue > 0 else None + # Same normalization: a threshold_tokens cap is a positive int, or None for "no cap". + _coerce_threshold_tokens_cap = _coerce_max_tokens def _apply_threshold_tokens_cap(self) -> None: """Clamp threshold_tokens to the configured cap (itself clamped to the context length).""" From 6bbdb45df171bcfe802272da6caa7e77038b5682 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:27:34 -0700 Subject: [PATCH 02/27] refactor(agent/context_compressor): unify durable read/write/load helpers and cooldown persistence paths --- agent/context_compressor.py | 131 ++++++++++++------------------------ 1 file changed, 44 insertions(+), 87 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 11fd67dbe4..f7b74317b5 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -2104,10 +2104,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if isinstance(stored, (int, float, str)): return True, max(default, coerce(stored)) return True, None - except (TypeError, ValueError, json.JSONDecodeError, sqlite3.Error) as exc: - logger.debug("%s lookup failed: %s", label, exc) except Exception as exc: - logger.debug("%s lookup failed (non-sqlite): %s", label, exc) + suffix = "" if isinstance(exc, (TypeError, ValueError, sqlite3.Error)) else " (non-sqlite)" + logger.debug("%s lookup failed%s: %s", label, suffix, exc) return False, default def _durable_write(self, method: str, label: str, *args) -> bool: @@ -2120,27 +2119,30 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): try: setter(session_id, *args) return True - except sqlite3.Error as exc: - logger.debug("%s persist failed: %s", label, exc) except Exception as exc: - logger.debug("%s persist failed (non-sqlite): %s", label, exc) + suffix = "" if isinstance(exc, sqlite3.Error) else " (non-sqlite)" + logger.debug("%s persist failed%s: %s", label, suffix, exc) return False + def _load_durable(self, attr: str, method: str, label: str, coerce, default, *args) -> None: + """Restore ``self.`` from the bound row; a non-numeric row resets it to ``default``.""" + found, value = self._durable_read(method, label, coerce, default, *args) + if found: + setattr(self, attr, default if value is None else value) + def _load_fallback_compression_streak(self) -> None: - found, value = self._durable_read( + self._load_durable( + "_fallback_compression_streak", "get_compression_fallback_streak", "compression fallback streak", int, 0, ) - if found: - self._fallback_compression_streak = 0 if value is None else value def _load_proactive_prune_rearm_tokens(self) -> None: """Restore the cache-boundary runway for a resumed durable session.""" - found, value = self._durable_read( + self._load_durable( + "_proactive_prune_rearm_tokens", "get_session_model_config_value", "proactive prune runway", int, 0, PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY, 0, ) - if found: - self._proactive_prune_rearm_tokens = 0 if value is None else value def _clear_durable_proactive_prune_rearm(self) -> None: """Best-effort removal of the persisted prune-runway key; transcript untouched.""" @@ -2157,11 +2159,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _load_ineffective_compression_count(self) -> None: """Load the durable anti-thrash strike count so a restart never disarms a guard.""" - found, value = self._durable_read( + self._load_durable( + "_ineffective_compression_count", "get_compression_ineffective_count", "compression ineffective count", int, 0, ) - if found: - self._ineffective_compression_count = 0 if value is None else value def _persist_ineffective_compression_count(self) -> None: self._durable_write( @@ -2171,11 +2172,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _load_anti_thrash_recovery_deadline(self) -> None: """Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed.""" - found, value = self._durable_read( + self._load_durable( + "_anti_thrash_recovery_deadline", "get_compression_recovery_deadline", "compression recovery deadline", float, 0.0, ) - if found: - self._anti_thrash_recovery_deadline = 0.0 if value is None else value def _set_anti_thrash_recovery_deadline(self, deadline: float) -> None: """Set the recovery deadline, persisting on change only (0 = disarmed).""" @@ -2288,30 +2288,19 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return local_state try: state = getter(session_id) - except sqlite3.Error as exc: - if refresh: - self._last_cooldown_refresh_was_authoritative = False - logger.debug("compression failure cooldown lookup failed: %s", exc) - return local_state - except Exception: + except Exception as exc: if refresh: self._last_cooldown_refresh_was_authoritative = False + if isinstance(exc, sqlite3.Error): + logger.debug("compression failure cooldown lookup failed: %s", exc) return local_state if refresh: self._last_cooldown_refresh_was_authoritative = True - if not state: - if refresh: - if local_state is not None and self._cooldown_persist_failed: - # Local cooldown never reached the DB, so an empty row is not evidence it was cleared; keep local. - return local_state - self._summary_failure_cooldown_until = 0.0 - self._last_summary_error = None - return None - - remaining_seconds = float(state.get("remaining_seconds") or 0.0) + remaining_seconds = float(state.get("remaining_seconds") or 0.0) if state else 0.0 if remaining_seconds <= 0: if refresh: if local_state is not None and self._cooldown_persist_failed: + # Local cooldown never reached the DB, so an empty row is not evidence it was cleared; keep local. return local_state self._summary_failure_cooldown_until = 0.0 self._last_summary_error = None @@ -2351,20 +2340,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): session_id = getattr(self, "_session_id", "") if not session_db or not session_id: return - - recorder = getattr(session_db, "record_compression_failure_cooldown", None) - if recorder is None: - self._cooldown_persist_failed = True - return - try: - recorder(session_id, cooldown_until, error) - self._cooldown_persist_failed = False - except sqlite3.Error as exc: - self._cooldown_persist_failed = True - logger.debug("compression failure cooldown persist failed: %s", exc) - except Exception as exc: - self._cooldown_persist_failed = True - logger.debug("compression failure cooldown persist failed (non-sqlite): %s", exc) + # A store without the recorder or a failed write both leave the durable row unauthoritative. + self._cooldown_persist_failed = not self._durable_write( + "record_compression_failure_cooldown", "compression failure cooldown", + cooldown_until, error, + ) def record_timeout_failure(self, error: str, failure_kind: str = "timeout") -> None: """Record a consecutive timeout/stall failure via the shared ladder. @@ -2378,38 +2358,18 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _clear_compression_failure_cooldown(self) -> None: # Fence check BEFORE cooldown-clear: a late cancelled worker must not undo the host's timeout cooldown. - cancelled_check = getattr(self, "_compression_cancelled_check", None) - if callable(cancelled_check): - try: - if cancelled_check(): - logger.info( - "Skipping compression cooldown clear: host already " - "cancelled this compression attempt" - ) - return - except Exception: - logger.debug( - "compression cancellation check failed", exc_info=True - ) + if self._compression_cancelled(): + logger.info( + "Skipping compression cooldown clear: host already " + "cancelled this compression attempt" + ) + return self._summary_failure_cooldown_until = 0.0 self._last_summary_error = None self._consecutive_timeout_failures = 0 self._cooldown_persist_failed = False - session_db = getattr(self, "_session_db", None) - session_id = getattr(self, "_session_id", "") - if not session_db or not session_id: - return - - clearer = getattr(session_db, "clear_compression_failure_cooldown", None) - if clearer is None: - return - try: - clearer(session_id) - except sqlite3.Error as exc: - logger.debug("compression failure cooldown clear failed: %s", exc) - except Exception as exc: - logger.debug("compression failure cooldown clear failed (non-sqlite): %s", exc) + self._durable_write("clear_compression_failure_cooldown", "compression failure cooldown clear") def _compression_cancelled(self) -> bool: """Read the host-owned cooperative cancellation signal, if installed.""" @@ -2823,18 +2783,15 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _refresh_durable_guards(self) -> None: """Re-read durable cooldown + breaker state; called only when a gate is about to block.""" - try: - self.get_active_compression_failure_cooldown(refresh=True) - except Exception as exc: - logger.debug("compression cooldown refresh failed: %s", exc) - try: - self._load_fallback_compression_streak() - except Exception as exc: - logger.debug("compression fallback-streak refresh failed: %s", exc) - try: - self._load_ineffective_compression_count() - except Exception as exc: - logger.debug("compression ineffective-count refresh failed: %s", exc) + for label, refresh in ( + ("cooldown", lambda: self.get_active_compression_failure_cooldown(refresh=True)), + ("fallback-streak", self._load_fallback_compression_streak), + ("ineffective-count", self._load_ineffective_compression_count), + ): + try: + refresh() + except Exception as exc: + logger.debug("compression %s refresh failed: %s", label, exc) def _automatic_compression_blocked(self, *, ignore_cooldown: bool = False) -> bool: """Return whether automatic compaction is in cooldown or tripped. From 608e8560fe1d9ac468251cc6b351397c27c30a05 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:33:40 -0700 Subject: [PATCH 03/27] refactor(agent/context_compressor): split compress() into phase helpers (494 -> ~130 LOC) --- agent/context_compressor.py | 690 +++++++++++++++++------------------- 1 file changed, 332 insertions(+), 358 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index f7b74317b5..67b8d4b54e 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -5077,23 +5077,8 @@ This compaction should PRIORITISE preserving all information related to the focu has_user_turn_before=_summary_has_user_turn_before_scan, ) - def compress( - self, - messages: List[Dict[str, Any]], - current_tokens: Optional[int] = None, - focus_topic: Optional[str] = None, - force: bool = False, - memory_context: str = "", - bypass_cooldown: bool = False, - ) -> List[Dict[str, Any]]: - """Compress conversation messages by summarizing middle turns. - - Prunes tool results and blank echo rows (survives an abort), protects head and a - token-budget tail, summarizes the middle, then cleans orphaned tool pairs. - ``force`` clears the failure cooldown and bypasses the feasibility skip; - ``bypass_cooldown`` runs the summary LLM without clearing the cooldown (#100661). - """ - # Per-call summary failure state; callers read these after compress() returns. + def _begin_compress_attempt(self, current_tokens: Optional[int], force: bool) -> Dict[str, Any]: + """Reset per-call result state (callers read it after compress()) and open telemetry.""" self._last_summary_dropped_count = 0 self._last_summary_fallback_used = False self._last_feasibility_skip = False @@ -5107,219 +5092,130 @@ This compaction should PRIORITISE preserving all information related to the focu # reset would fall through to the destructive static fallback (#29559). Success clears them. telemetry = self._begin_compression_telemetry(current_tokens=current_tokens) telemetry["chunk_count"] = 0 - - # Manual /compress bypasses the failure cooldown. + # Manual /compress bypasses the failure cooldown and the structural no-op backoff (#93022). if force: self._clear_compression_failure_cooldown() - # Manual /compress also overrides the structural no-op backoff (#93022). self._structural_no_op_backoff_until = 0.0 - n_messages = len(messages) - # Only need head + 3 tail messages minimum (token budget decides the real tail size) - _min_for_compress = self._protect_head_size(messages) + 3 + 1 - if n_messages <= _min_for_compress: - # Structural no-op (#93022): transient backoff, not an ineffectiveness strike (which - # permanently disarmed auto-compaction on short sessions). - self._last_compression_savings_pct = 0.0 - telemetry["failure_class"] = "insufficient_messages" - self._record_structural_no_op( - f"only {n_messages} messages (need > {_min_for_compress})" - ) - return messages + return telemetry - display_tokens = current_tokens if current_tokens else self.last_prompt_tokens or estimate_messages_tokens_rough(messages) - - # Phase 1: Prune old tool results (cheap, no LLM call) - messages, pruned_count = self._prune_old_tool_results( - messages, protect_tail_count=self.protect_last_n, - protect_tail_tokens=self.tail_token_budget, - ) - if pruned_count and not self.quiet_mode: - logger.info("Pre-compression: pruned %d old tool result(s)", pruned_count) + def _structural_no_op_result( + self, telemetry: Dict[str, Any], failure_class: str, reason: str, + ) -> None: + """Nothing eligible to compress: transient backoff (#93022), never an ineffectiveness strike.""" + telemetry["failure_class"] = failure_class + self._last_compression_savings_pct = 0.0 + self._record_structural_no_op(reason) + def _drop_blank_echoes(self, messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """Remove blank platform echoes trailing the latest actionable user turn.""" latest_actionable_idx = self._find_last_user_message_idx(messages, 0) - blank_echo_indices = self._blank_echo_indices_after( - messages, latest_actionable_idx - ) + blank_echo_indices = self._blank_echo_indices_after(messages, latest_actionable_idx) if blank_echo_indices: - messages = [ - message - for idx, message in enumerate(messages) - if idx not in blank_echo_indices - ] - n_messages = len(messages) - latest_actionable_idx = self._find_last_user_message_idx(messages, 0) - - # Phase 2: Determine boundaries - compress_start = self._protect_head_size(messages) - compress_start = self._align_boundary_forward(messages, compress_start) + messages = [m for idx, m in enumerate(messages) if idx not in blank_echo_indices] + return messages + def _compress_window(self, messages: List[Dict[str, Any]]) -> tuple[int, int]: + """Return ``(compress_start, compress_end)`` for the summarizable middle.""" + compress_start = self._align_boundary_forward(messages, self._protect_head_size(messages)) compress_end = self._find_tail_cut_by_tokens(messages, compress_start) - # A role collision can merge the summary into the first tail row; keep an actionable user # event out of that slot by retaining an older assistant/tool bridge. + latest_actionable_idx = self._find_last_user_message_idx(messages, 0) if compress_end == latest_actionable_idx: bridge_idx = latest_actionable_idx - 1 if bridge_idx >= 0 and messages[bridge_idx].get("role") == "tool": - bridge_idx = self._align_boundary_backward( - messages, latest_actionable_idx - ) + bridge_idx = self._align_boundary_backward(messages, latest_actionable_idx) elif bridge_idx < 0 or messages[bridge_idx].get("role") != "assistant": bridge_idx = -1 if bridge_idx > compress_start: compress_end = bridge_idx + return compress_start, compress_end - if compress_start >= compress_end: - self._record_compression_regions( - head_messages=messages[:compress_start], - middle_messages=[], - tail_messages=messages[compress_end:], - ) - telemetry["failure_class"] = "no_compressible_window" - # Nothing eligible to compress: structural no-op (#93022) -> transient backoff, not an - # ineffectiveness strike. - self._last_compression_savings_pct = 0.0 - self._record_structural_no_op( - f"compress_start ({compress_start}) >= compress_end " - f"({compress_end}) - transcript fits within tail budget" - ) - return messages - - turns_to_summarize = messages[compress_start:compress_end] - # Lean mode demotes stale tail tool results before summary generation so stubs exist even if - # it aborts. - if getattr(self, "tail_mode", "lean") == "lean": - messages = self._demote_stale_tail_tools(messages, compress_end) - scan = self._scan_window_handoffs( - messages, compress_start, compress_end, turns_to_summarize + def _log_compression_start( + self, display_tokens: int, compress_start: int, compress_end: int, + n_turns: int, tail_msgs: int, + ) -> None: + logger.info( + "Context compression triggered (%d tokens >= %d threshold)", + display_tokens, self.threshold_tokens, ) - turns_to_summarize = scan.turns_to_summarize - summary_indices = scan.summary_indices - tail_start = scan.tail_start - _previous_summary_before_scan = scan.previous_summary_before - _summary_has_user_turn_before_scan = scan.has_user_turn_before - - self._record_compression_regions( - head_messages=messages[:compress_start], - middle_messages=turns_to_summarize, - tail_messages=messages[compress_end:], + logger.info( + "Model context limit: %d tokens (%.0f%% = %d)", + self.context_length, self.threshold_percent * 100, self.threshold_tokens, + ) + logger.info( + "Summarizing turns %d-%d (%d turns), protecting %d head + %d tail messages", + compress_start + 1, compress_end, n_turns, compress_start, tail_msgs, ) - telemetry["chunk_count"] = 1 if turns_to_summarize else 0 - - if not turns_to_summarize: - # Window is only handoff rows: structural no-op, skip the aux call (#59496), transient - # backoff. _previous_summary is KEPT — it came from this transcript. - telemetry["failure_class"] = "empty_post_handoff_window" - self._last_compression_savings_pct = 0.0 - self._record_structural_no_op( - f"window {compress_start}-{compress_end} holds only " - "already-summarized handoffs" - ) - return messages + def _feasibility_skip( + self, telemetry: Dict[str, Any], turns_to_summarize: List[Dict[str, Any]], + compress_start: int, compress_end: int, + ) -> bool: + """Pre-LLM skip after a real-usage ineffectiveness strike (reads the counter, never writes).""" + if self._ineffective_compression_count < 1: + return False + # Reuse the telemetry estimate so log and telemetry agree; None means the regions helper + # no-op'd (0 is valid). + middle_tokens = telemetry.get("middle_window_tokens") + if middle_tokens is None: + middle_tokens = estimate_messages_tokens_rough(turns_to_summarize) + if middle_tokens >= int(self.threshold_tokens * _FEASIBILITY_SKIP_MIDDLE_FRACTION): + return False + self._last_feasibility_skip = True + self._prellm_skip_count += 1 + telemetry["prellm_skip_count"] = self._prellm_skip_count if not self.quiet_mode: - logger.info( - "Context compression triggered (%d tokens >= %d threshold)", - display_tokens, - self.threshold_tokens, - ) - logger.info( - "Model context limit: %d tokens (%.0f%% = %d)", - self.context_length, - self.threshold_percent * 100, - self.threshold_tokens, - ) - tail_msgs = n_messages - tail_start - logger.info( - "Summarizing turns %d-%d (%d turns), protecting %d head + %d tail messages", - compress_start + 1, - compress_end, - len(turns_to_summarize), - compress_start, - tail_msgs, + logger.warning( + "Compression: middle section (%d tokens at indices " + "%d-%d) is below %.0f%% of threshold (%d tokens) — " + "skipping LLM summarization, proceeding with " + "deterministic message dropping. prellm_skip_count=%d", + middle_tokens, compress_start, compress_end, + _FEASIBILITY_SKIP_MIDDLE_FRACTION * 100, + self.threshold_tokens, self._prellm_skip_count, ) + return True - # Phase 3: Generate structured summary + def _abort_on_summary_failure( + self, telemetry: Dict[str, Any], n_skipped: int, previous_summary_before_scan: Optional[str], + ) -> bool: + """Abort (messages unchanged) on a terminal failure or when configured to; True when aborted. - # Pre-LLM feasibility skip after a real-usage ineffectiveness strike: READS the strike - # counter, never writes it (skips tracked in _prellm_skip_count). Bypassed by force=True. - feasibility_skip = False - if not force and self._ineffective_compression_count >= 1: - # Reuse the telemetry estimate so log and telemetry agree; None means the regions helper - # no-op'd (0 is valid). - middle_tokens = telemetry.get("middle_window_tokens") - if middle_tokens is None: - middle_tokens = estimate_messages_tokens_rough(turns_to_summarize) - if middle_tokens < int( - self.threshold_tokens * _FEASIBILITY_SKIP_MIDDLE_FRACTION - ): - feasibility_skip = True - self._last_feasibility_skip = True - self._prellm_skip_count += 1 - telemetry["prellm_skip_count"] = self._prellm_skip_count - if not self.quiet_mode: - logger.warning( - "Compression: middle section (%d tokens at indices " - "%d-%d) is below %.0f%% of threshold (%d tokens) — " - "skipping LLM summarization, proceeding with " - "deterministic message dropping. prellm_skip_count=%d", - middle_tokens, compress_start, compress_end, - _FEASIBILITY_SKIP_MIDDLE_FRACTION * 100, - self.threshold_tokens, self._prellm_skip_count, - ) + Access/quota, network, truncated and empty-content failures ALWAYS abort (#29559); + otherwise ``abort_on_summary_failure`` decides between abort and the static fallback. + """ + terminal_failure = next( + ( + (failure_class, message) + for flag, failure_class, message in _TERMINAL_SUMMARY_FAILURES + if getattr(self, flag) + ), + None, + ) + if terminal_failure is None and not self.abort_on_summary_failure: + return False + self._last_summary_dropped_count = 0 # nothing actually dropped + self._last_summary_fallback_used = False + self._last_compress_aborted = True + failure_class, message = terminal_failure or ( + "summary_generation_aborted", + "Summary generation failed — aborting compression " + "(compression.abort_on_summary_failure=true). " + "%d message(s) preserved unchanged. Conversation is " + "frozen until the next /compress or /new.", + ) + telemetry["failure_class"] = failure_class + # Roll back the self-heal rehydration so the aborted attempt is a true no-op (#57835). + self._previous_summary = previous_summary_before_scan + if not self.quiet_mode: + logger.warning(message, n_skipped) + return True - if feasibility_skip: - summary = None # No LLM call; Phase 4 inserts the deterministic fallback - else: - # Focus-topic derivation scans user turns; only pay when a summary is generated. - summary_focus_topic = focus_topic or self._derive_auto_focus_topic(messages) - try: - summary = self._generate_summary( - turns_to_summarize, - focus_topic=summary_focus_topic, - memory_context=memory_context, - bypass_cooldown=bypass_cooldown, - ) - except AuxiliaryExplicitCancellation: - # Cancellation is a true no-op: restore the self-heal scan's mutation before the - # exception escapes. - self._previous_summary = _previous_summary_before_scan - self._summary_has_user_turn = _summary_has_user_turn_before_scan - raise + _COMPRESSION_NOTE = "[Note: Some earlier conversation turns have been compacted into a handoff summary to preserve context space. The current session state may still reflect earlier work, so build on that summary and state rather than re-doing work. Your persistent memory (MEMORY.md, USER.md) remains fully authoritative regardless of compaction.]" - # abort_on_summary_failure: True aborts unchanged (_last_compress_aborted), False uses the - # static fallback. Access/quota, network, empty-content failures ALWAYS abort (#29559). - terminal_failure = None - if not summary and not feasibility_skip: - terminal_failure = next( - ( - (failure_class, message) - for flag, failure_class, message in _TERMINAL_SUMMARY_FAILURES - if getattr(self, flag) - ), - None, - ) - if terminal_failure is not None or ( - not summary and not feasibility_skip and self.abort_on_summary_failure - ): - n_skipped = compress_end - compress_start - self._last_summary_dropped_count = 0 # nothing actually dropped - self._last_summary_fallback_used = False - self._last_compress_aborted = True - failure_class, message = terminal_failure or ( - "summary_generation_aborted", - "Summary generation failed — aborting compression " - "(compression.abort_on_summary_failure=true). " - "%d message(s) preserved unchanged. Conversation is " - "frozen until the next /compress or /new.", - ) - telemetry["failure_class"] = failure_class - # Roll back the self-heal rehydration so the aborted attempt is a true no-op (#57835). - self._previous_summary = _previous_summary_before_scan - if not self.quiet_mode: - logger.warning(message, n_skipped) - return messages - - # Phase 4: Assemble compressed message list + def _assemble_head(self, messages: List[Dict[str, Any]], compress_start: int) -> List[Dict[str, Any]]: + """Protected head with the compaction note on the system prompt and stale handoffs stripped.""" compressed = [] for i in range(compress_start): # Head handoff already lives in _previous_summary: strip it (standalone dropped, merged @@ -5327,66 +5223,71 @@ This compaction should PRIORITISE preserving all information related to the focu msg = _fresh_compaction_message_copy(messages[i]) if i == 0 and msg.get("role") == "system": existing = msg.get("content") - _compression_note = "[Note: Some earlier conversation turns have been compacted into a handoff summary to preserve context space. The current session state may still reflect earlier work, so build on that summary and state rather than re-doing work. Your persistent memory (MEMORY.md, USER.md) remains fully authoritative regardless of compaction.]" - if _compression_note not in _content_text_for_contains(existing): + if self._COMPRESSION_NOTE not in _content_text_for_contains(existing): msg["content"] = _append_text_to_content( existing, - "\n\n" + _compression_note if isinstance(existing, str) and existing else _compression_note, + "\n\n" + self._COMPRESSION_NOTE if isinstance(existing, str) and existing else self._COMPRESSION_NOTE, ) stripped = self._strip_context_summary_handoff_message(msg) if stripped is not None: compressed.append(stripped) + return compressed - # Deterministic fallback so the model gets recoverable continuity anchors. - if not summary: - if not self.quiet_mode: - if feasibility_skip: - logger.info("Feasibility skip — inserting deterministic fallback context summary") - else: - logger.warning("Summary generation failed — inserting deterministic fallback context summary") - n_dropped = compress_end - compress_start - self._last_summary_dropped_count = n_dropped - self._last_summary_fallback_used = True - telemetry["fallback_used"] = True + def _fallback_summary_for_window( + self, telemetry: Dict[str, Any], turns_to_summarize: List[Dict[str, Any]], + n_dropped: int, feasibility_skip: bool, + ) -> str: + """Deterministic fallback so the model gets recoverable continuity anchors.""" + if not self.quiet_mode: if feasibility_skip: - # Feasibility skip is deliberate, not aux-model breakage — keep the telemetry class - # distinct. - telemetry["failure_class"] = telemetry.get("failure_class") or "feasibility_skip" + logger.info("Feasibility skip — inserting deterministic fallback context summary") else: - telemetry["failure_class"] = telemetry.get("failure_class") or "summary_generation_failed" - summary = self._build_static_fallback_summary( - turns_to_summarize, - # A stale error from an earlier failure must not be embedded in a feasibility-skip - # fallback. - reason=None if feasibility_skip else self._last_summary_error, - ) + logger.warning("Summary generation failed — inserting deterministic fallback context summary") + self._last_summary_dropped_count = n_dropped + self._last_summary_fallback_used = True + telemetry["fallback_used"] = True + # Feasibility skip is deliberate, not aux-model breakage — keep the telemetry class distinct. + telemetry["failure_class"] = telemetry.get("failure_class") or ( + "feasibility_skip" if feasibility_skip else "summary_generation_failed" + ) + return self._build_static_fallback_summary( + turns_to_summarize, + # A stale error from an earlier failure must not be embedded in a feasibility-skip fallback. + reason=None if feasibility_skip else self._last_summary_error, + ) + def _assemble_tail( + self, messages: List[Dict[str, Any]], compress_end: int, tail_start: int, + summary_indices: set, + ) -> List[Dict[str, Any]]: + """Protected tail with already-folded handoff rows dropped and merged handoffs unwrapped.""" tail_messages: List[Dict[str, Any]] = [] # Start at tail_start, not compress_end: the rehydration scan may have advanced it (#57835). - for i in range(max(compress_end, tail_start), n_messages): + for i in range(max(compress_end, tail_start), len(messages)): if i in summary_indices and i >= tail_start: - # Already folded into _previous_summary; don't re-emit. - continue - msg = _fresh_compaction_message_copy(messages[i]) - stripped = self._strip_context_summary_handoff_message(msg) + continue # already folded into _previous_summary; don't re-emit + stripped = self._strip_context_summary_handoff_message( + _fresh_compaction_message_copy(messages[i]) + ) if stripped is not None: tail_messages.append(stripped) + return tail_messages - _merge_summary_into_tail = False - # Roles read the assembled (post-strip) head/tail and are TEMPLATE-VISIBLE: Mistral-strict - # templates skip tool rows for alternation, so alternate against what the template counts. + @staticmethod + def _summary_placement( + compressed: List[Dict[str, Any]], tail_messages: List[Dict[str, Any]], compress_start: int, + ) -> tuple[str, bool, bool, Optional[int]]: + """Pick the summary row's role so template-visible alternation holds. + + Returns ``(summary_role, merge_into_tail, force_user_leading, first_tail_visible_idx)``. + Roles read the assembled (post-strip) head/tail and are TEMPLATE-VISIBLE: Mistral-strict + templates skip tool rows for alternation, so alternate against what the template counts. + """ last_head_role: Optional[str] = "user" if compressed: + # None = all-exempt head: the summary opens the visible sequence and must be "user". last_head_role = next( - ( - role - for role in ( - _template_visible_role(m) for m in reversed(compressed) - ) - if role is not None - ), - # All-exempt head: the summary opens the visible sequence and must be "user" - # (handled below). + (r for r in (_template_visible_role(m) for m in reversed(compressed)) if r is not None), None, ) first_tail_role = None @@ -5395,144 +5296,84 @@ This compaction should PRIORITISE preserving all information related to the focu first_tail_visible_idx, first_tail_role = next( ( (idx, role) - for idx, role in ( - (idx, _template_visible_role(m)) - for idx, m in enumerate(tail_messages) - ) + for idx, role in ((idx, _template_visible_role(m)) for idx, m in enumerate(tail_messages)) if role is not None ), (None, None), ) # System-only head: the summary is the first visible message and Anthropic requires - # role=user (#52160). - _force_user_leading = compress_start == 0 or last_head_role == "system" - # Zero-user-turn guard (#58753): if no user row with non-empty TEXT survives, the summary - # must be role="user" or OpenAI-compatible backends reject. Image-only rows don't count. - if not _force_user_leading: - def _is_nonempty_user_turn(message: Dict[str, Any]) -> bool: - return message.get("role") == "user" and bool( - _content_text_for_contains(message.get("content")).strip() - ) - - _user_survives = any( - _is_nonempty_user_turn(message) for message in compressed - ) or any( - _is_nonempty_user_turn(message) for message in tail_messages - ) - if not _user_survives: - _force_user_leading = True - # Alternate against head first, then tail; None (all-exempt head) means the summary must be - # "user". - if ( - last_head_role is None - or last_head_role in {"assistant", "tool"} - or _force_user_leading - ): + # role=user (#52160). Zero-user-turn guard (#58753): if no user row with non-empty TEXT + # survives, the summary must be role="user" or OpenAI-compatible backends reject. + # Image-only rows don't count. + force_user_leading = compress_start == 0 or last_head_role == "system" or not any( + m.get("role") == "user" and bool(_content_text_for_contains(m.get("content")).strip()) + for m in (*compressed, *tail_messages) + ) + # Alternate against head first, then tail; None (all-exempt head) means "user". + if last_head_role is None or last_head_role in {"assistant", "tool"} or force_user_leading: summary_role = "user" else: summary_role = "assistant" + merge_into_tail = False # Flip on a tail collision only if that doesn't collide with the head. if first_tail_role is not None and summary_role == first_tail_role: flipped = "assistant" if summary_role == "user" else "user" - # All-exempt head pins "user"; flipping would open the visible sequence with - # "assistant". - if ( - flipped != last_head_role - and last_head_role is not None - and not _force_user_leading - ): + # All-exempt head pins "user"; flipping would open the visible sequence with "assistant". + if flipped != last_head_role and last_head_role is not None and not force_user_leading: summary_role = flipped else: # Neither role alternates: merge the summary into the first tail message instead. - _merge_summary_into_tail = bool(tail_messages) + merge_into_tail = bool(tail_messages) + return summary_role, merge_into_tail, force_user_leading, first_tail_visible_idx - # End marker stops weak models treating the quoted summary as fresh input (#11475) or - # regurgitating it (#33256). - if not _merge_summary_into_tail: - summary = summary + "\n\n" + _SUMMARY_END_MARKER - - if not _merge_summary_into_tail: - compressed.append({ - "role": summary_role, - "content": summary, - COMPRESSED_SUMMARY_METADATA_KEY: True, - COMPRESSED_SUMMARY_HAS_USER_TURN_KEY: bool( - self._summary_has_user_turn - ), - }) - - # Default carrier is tail[0]: an exempt row absorbs the summary invisibly. The forced repair - # path needs a non-empty role=user row, so it targets the template-visible row. - _merge_target_idx = 0 - if _force_user_leading and first_tail_visible_idx is not None: - _merge_target_idx = first_tail_visible_idx - for tail_idx, msg in enumerate(tail_messages): - # Tag carried-forward tail rows so archive_and_compact treats their originals as - # superseded duplicates (#86366). - if isinstance(msg, dict): - msg[_COMPACTION_TAIL_MARKER] = True - if _merge_summary_into_tail and tail_idx == _merge_target_idx: - old_content = msg.get("content", "") - if _force_user_leading and summary_role == "user": - # Anthropic/Bedrock: summary must lead the first visible message; the real - # request follows the end marker. - prefix = summary + "\n\n" + _SUMMARY_END_MARKER + "\n\n" - msg["content"] = _append_text_to_content( - old_content, - prefix, - prepend=True, - ) - else: - # Old tail content is kept as delimited reference BEFORE the summary; the end - # marker goes last. - suffix = ( - "\n\n" + _MERGED_SUMMARY_DELIMITER + "\n\n" - + summary + "\n\n" - + _SUMMARY_END_MARKER - ) - msg["content"] = _append_text_to_content( - _append_text_to_content(old_content, suffix, prepend=False), - _MERGED_PRIOR_CONTEXT_HEADER + "\n", - prepend=True, - ) - # Frontends use this to detect a summary-prefixed message. - msg[COMPRESSED_SUMMARY_METADATA_KEY] = True - msg[COMPRESSED_SUMMARY_HAS_USER_TURN_KEY] = bool( - self._summary_has_user_turn - ) - # Rewritten content: drop the stale api_content sidecar so replay can't resend pre- - # merge bytes. - drop_stale_api_content(msg) - _merge_summary_into_tail = False - compressed.append(msg) + def _merge_summary_into_tail_row( + self, msg: Dict[str, Any], summary: str, summary_role: str, force_user_leading: bool, + ) -> None: + """Fold the summary into a carried tail row (in place) when no standalone role alternates.""" + old_content = msg.get("content", "") + if force_user_leading and summary_role == "user": + # Anthropic/Bedrock: summary must lead the first visible message; the real request + # follows the end marker. + msg["content"] = _append_text_to_content( + old_content, summary + "\n\n" + _SUMMARY_END_MARKER + "\n\n", prepend=True, + ) + else: + # Old tail content is kept as delimited reference BEFORE the summary; the end marker + # goes last. + suffix = "\n\n" + _MERGED_SUMMARY_DELIMITER + "\n\n" + summary + "\n\n" + _SUMMARY_END_MARKER + msg["content"] = _append_text_to_content( + _append_text_to_content(old_content, suffix, prepend=False), + _MERGED_PRIOR_CONTEXT_HEADER + "\n", + prepend=True, + ) + # Frontends use this to detect a summary-prefixed message. + msg[COMPRESSED_SUMMARY_METADATA_KEY] = True + msg[COMPRESSED_SUMMARY_HAS_USER_TURN_KEY] = bool(self._summary_has_user_turn) + # Rewritten content: drop the stale api_content sidecar so replay can't resend pre-merge bytes. + drop_stale_api_content(msg) + def _finalize_compressed( + self, compressed: List[Dict[str, Any]], messages: List[Dict[str, Any]], n_messages: int, + ) -> List[Dict[str, Any]]: + """Post-assembly cleanup: orphan pairs, media, savings, markers, replay prune, mem trim.""" self.compression_count += 1 - compressed = self._sanitize_tool_pairs(compressed) - # Replace historical image payloads with placeholders; multi-MB base64 blobs otherwise # exceed body limits. compressed = _strip_historical_media(compressed) - new_estimate = estimate_messages_tokens_rough(compressed) - # Like-for-like savings: current_tokens includes system prompt/tool schemas, new_estimate is # messages-only; comparing them fakes ~96% savings and kills the anti-thrashing guard. + # Message-only savings are diagnostic; the verdict belongs to the next provider prompt count. + new_estimate = estimate_messages_tokens_rough(compressed) pre_estimate = estimate_messages_tokens_rough(messages) saved_estimate = pre_estimate - new_estimate savings_pct = (saved_estimate / pre_estimate * 100) if pre_estimate > 0 else 0 self._last_compression_savings_pct = savings_pct - - # Message-only savings are diagnostic; the anti-thrashing verdict belongs to the next - # provider prompt count. - if not self.quiet_mode: logger.info( "Compressed: %d -> %d messages (~%d tokens saved, %.0f%%)", - n_messages, - len(compressed), - saved_estimate, - savings_pct, + n_messages, len(compressed), saved_estimate, savings_pct, ) logger.info("Compression #%d complete", self.compression_count) @@ -5549,18 +5390,13 @@ This compaction should PRIORITISE preserving all information related to the focu self._last_compression_made_progress = True # Compaction frees the biggest allocation: hand pages back to the OS (glibc/config-gated, - # rate-limited, #70782). + # rate-limited, #70782). debug, not warning: compression must never fail because of a trim. try: from hermes_cli.mem_trim import trim_memory trim_memory(reason="post-compression") except Exception as exc: - # debug, not warning: compression must never fail because of a trim. - logger.debug( - "post-compression memory trim failed: %s: %s", - type(exc).__name__, - exc, - ) + logger.debug("post-compression memory trim failed: %s: %s", type(exc).__name__, exc) # Batch marker holds MORE history than the rolling summary: reset micro state so it can't # supersede/defrag content it lacks; the next micro pass rehydrates from the batch marker. @@ -5569,9 +5405,147 @@ This compaction should PRIORITISE preserving all information related to the focu self._micro_compact_consecutive_failures = 0 self._micro_compact_last_failure_cursor = -1 self._proactive_prune_rearm_tokens = 0 - return compressed + def compress( + self, + messages: List[Dict[str, Any]], + current_tokens: Optional[int] = None, + focus_topic: Optional[str] = None, + force: bool = False, + memory_context: str = "", + bypass_cooldown: bool = False, + ) -> List[Dict[str, Any]]: + """Compress conversation messages by summarizing middle turns. + + Prunes tool results and blank echo rows (survives an abort), protects head and a + token-budget tail, summarizes the middle, then cleans orphaned tool pairs. + ``force`` clears the failure cooldown and bypasses the feasibility skip; + ``bypass_cooldown`` runs the summary LLM without clearing the cooldown (#100661). + """ + telemetry = self._begin_compress_attempt(current_tokens, force) + n_messages = len(messages) + # Only need head + 3 tail messages minimum (token budget decides the real tail size) + _min_for_compress = self._protect_head_size(messages) + 3 + 1 + if n_messages <= _min_for_compress: + self._structural_no_op_result( + telemetry, "insufficient_messages", + f"only {n_messages} messages (need > {_min_for_compress})", + ) + return messages + + display_tokens = current_tokens if current_tokens else self.last_prompt_tokens or estimate_messages_tokens_rough(messages) + + # Phase 1: Prune old tool results (cheap, no LLM call) + messages, pruned_count = self._prune_old_tool_results( + messages, protect_tail_count=self.protect_last_n, + protect_tail_tokens=self.tail_token_budget, + ) + if pruned_count and not self.quiet_mode: + logger.info("Pre-compression: pruned %d old tool result(s)", pruned_count) + messages = self._drop_blank_echoes(messages) + n_messages = len(messages) + + # Phase 2: Determine boundaries + compress_start, compress_end = self._compress_window(messages) + if compress_start >= compress_end: + self._record_compression_regions( + head_messages=messages[:compress_start], + middle_messages=[], + tail_messages=messages[compress_end:], + ) + self._structural_no_op_result( + telemetry, "no_compressible_window", + f"compress_start ({compress_start}) >= compress_end " + f"({compress_end}) - transcript fits within tail budget", + ) + return messages + + turns_to_summarize = messages[compress_start:compress_end] + # Lean mode demotes stale tail tool results before summary generation so stubs exist even if + # it aborts. + if getattr(self, "tail_mode", "lean") == "lean": + messages = self._demote_stale_tail_tools(messages, compress_end) + scan = self._scan_window_handoffs(messages, compress_start, compress_end, turns_to_summarize) + turns_to_summarize = scan.turns_to_summarize + + self._record_compression_regions( + head_messages=messages[:compress_start], + middle_messages=turns_to_summarize, + tail_messages=messages[compress_end:], + ) + telemetry["chunk_count"] = 1 if turns_to_summarize else 0 + if not turns_to_summarize: + # Window is only handoff rows (#59496): skip the aux call; _previous_summary is KEPT — + # it came from this transcript. + self._structural_no_op_result( + telemetry, "empty_post_handoff_window", + f"window {compress_start}-{compress_end} holds only already-summarized handoffs", + ) + return messages + if not self.quiet_mode: + self._log_compression_start( + display_tokens, compress_start, compress_end, + len(turns_to_summarize), n_messages - scan.tail_start, + ) + + # Phase 3: Generate structured summary (or skip the LLM when the middle is too small to matter) + feasibility_skip = not force and self._feasibility_skip( + telemetry, turns_to_summarize, compress_start, compress_end, + ) + if feasibility_skip: + summary = None # No LLM call; Phase 4 inserts the deterministic fallback + else: + # Focus-topic derivation scans user turns; only pay when a summary is generated. + try: + summary = self._generate_summary( + turns_to_summarize, + focus_topic=focus_topic or self._derive_auto_focus_topic(messages), + memory_context=memory_context, + bypass_cooldown=bypass_cooldown, + ) + except AuxiliaryExplicitCancellation: + # Cancellation is a true no-op: restore the self-heal scan's mutation before the + # exception escapes. + self._previous_summary = scan.previous_summary_before + self._summary_has_user_turn = scan.has_user_turn_before + raise + if not summary and self._abort_on_summary_failure( + telemetry, compress_end - compress_start, scan.previous_summary_before, + ): + return messages + + # Phase 4: Assemble compressed message list + compressed = self._assemble_head(messages, compress_start) + if not summary: + summary = self._fallback_summary_for_window( + telemetry, turns_to_summarize, compress_end - compress_start, feasibility_skip, + ) + tail_messages = self._assemble_tail(messages, compress_end, scan.tail_start, scan.summary_indices) + summary_role, merge_into_tail, force_user_leading, first_tail_visible_idx = ( + self._summary_placement(compressed, tail_messages, compress_start) + ) + if not merge_into_tail: + # End marker stops weak models treating the quoted summary as fresh input (#11475) or + # regurgitating it (#33256). + compressed.append({ + "role": summary_role, + "content": summary + "\n\n" + _SUMMARY_END_MARKER, + COMPRESSED_SUMMARY_METADATA_KEY: True, + COMPRESSED_SUMMARY_HAS_USER_TURN_KEY: bool(self._summary_has_user_turn), + }) + # Default carrier is tail[0]: an exempt row absorbs the summary invisibly. The forced repair + # path needs a non-empty role=user row, so it targets the template-visible row. + merge_target_idx = first_tail_visible_idx if force_user_leading and first_tail_visible_idx is not None else 0 + for tail_idx, msg in enumerate(tail_messages): + # Tag carried-forward tail rows so archive_and_compact treats their originals as + # superseded duplicates (#86366). + if isinstance(msg, dict): + msg[_COMPACTION_TAIL_MARKER] = True + if merge_into_tail and tail_idx == merge_target_idx: + self._merge_summary_into_tail_row(msg, summary, summary_role, force_user_leading) + compressed.append(msg) + return self._finalize_compressed(compressed, messages, n_messages) def is_compaction_summary_message(message: Any) -> bool: """Return True when *message* is a context-compaction handoff summary. From 83168ff4e401d9a872a279461ed399c337a321da Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:45:36 -0700 Subject: [PATCH 04/27] refactor(agent/context_compressor): split _prune_old_tool_results into boundary/dedupe/demote/pressure helpers --- agent/context_compressor.py | 394 ++++++++++++++++++------------------ 1 file changed, 193 insertions(+), 201 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 67b8d4b54e..0a55c93254 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1153,6 +1153,25 @@ def _extract_tool_call_id(tool_call: Any) -> str: return str(getattr(tool_call, "id", "") or "") +def _tool_calls_by_id(messages: List[Dict[str, Any]]) -> Dict[str, tuple]: + """Map ``tool_call_id -> (tool_name, raw_arguments)`` over every assistant tool call.""" + out: Dict[str, tuple] = {} + for msg in messages: + if msg.get("role") != "assistant": + continue + for tc in msg.get("tool_calls") or []: + if isinstance(tc, dict): + fn = tc.get("function", {}) + out[tc.get("id", "")] = (fn.get("name", "unknown"), fn.get("arguments", "")) + else: + fn = getattr(tc, "function", None) + out[getattr(tc, "id", "") or ""] = ( + getattr(fn, "name", "unknown") if fn else "unknown", + getattr(fn, "arguments", "") if fn else "", + ) + return out + + def _collect_path_mentions(text: str, relevant_files: list[str], *, limit: int = 12) -> None: for match in _PATH_MENTION_RE.findall(text): _dedupe_append(relevant_files, match.rstrip(".,:;"), limit=limit) @@ -2878,6 +2897,169 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return False + def _prune_boundary( + self, result: List[Dict[str, Any]], protect_tail_count: int, protect_tail_tokens: int | None, + ) -> int: + """First index of the protected tail; token budget (when given) beats the count floor.""" + if protect_tail_tokens is None or protect_tail_tokens <= 0: + return len(result) - protect_tail_count + # Token-budget walk; cap the message-count floor like tail-cut so a bulky recent run stays prunable. + accumulated = 0 + boundary = len(result) + min_protect = min(protect_tail_count, len(result), _MAX_TAIL_MESSAGE_FLOOR) + # Charge thinking on the newest turn only (parity with tail-cut and the estimator). + _newest_asst_idx = _last_assistant_index(result) + _charge_all_thinking = self._stale_thinking_on_wire() + for i in range(len(result) - 1, -1, -1): + msg_tokens = _estimate_msg_budget_tokens( + result[i], charge_stale_thinking=(_charge_all_thinking or i == _newest_asst_idx), + ) + if accumulated + msg_tokens > protect_tail_tokens and (len(result) - i) >= min_protect: + boundary = i + break + accumulated += msg_tokens + boundary = i + # Apply the floor in count-space: `max` in index-space would invert (smaller index = MORE protected). + return len(result) - max(len(result) - boundary, min_protect) + + @staticmethod + def _dedupe_tool_results(result: List[Dict[str, Any]]) -> int: + """Pass 1: keep the newest copy of identical tool results, back-reference older ones.""" + pruned = 0 + content_hashes: set = set() + for i in range(len(result) - 1, -1, -1): + msg = result[i] + content = msg.get("content") or "" + # Non-string/multimodal-envelope shapes can't be hashed by text. + if msg.get("role") != "tool" or not isinstance(content, str) or len(content) < _PRUNE_MIN_CHARS: + continue + h = hashlib.md5(content.encode("utf-8", errors="replace")).hexdigest()[:12] + if h in content_hashes: + result[i] = {**msg, "content": "[Duplicate tool output — same content as a more recent call]"} + pruned += 1 + else: + content_hashes.add(h) + return pruned + + @staticmethod + def _truncate_tool_call_args_at(result: List[Dict[str, Any]], idx: int) -> bool: + """Shrink large tool_call argument payloads at ``idx`` (inside the parsed JSON, so it stays valid).""" + msg = result[idx] + if msg.get("role") != "assistant" or not msg.get("tool_calls"): + return False + new_tcs = [] + modified = False + for tc in msg["tool_calls"]: + if isinstance(tc, dict): + args = tc.get("function", {}).get("arguments", "") + if len(args) > 500: + new_args = _truncate_tool_call_args_json(args) + if new_args != args: + tc = {**tc, "function": {**tc["function"], "arguments": new_args}} + modified = True + new_tcs.append(tc) + if modified: + result[idx] = {**msg, "tool_calls": new_tcs} + return modified + + @staticmethod + def _demote_tool_result_at( + result: List[Dict[str, Any]], idx: int, call_id_to_tool: Dict[str, tuple[str, str]], + min_prune_chars: int, protected_skills: Optional[set[str]] = None, + ) -> bool: + """Replace the tool result at ``idx`` with a 1-line summary; True if modified. + + ``protected_skills`` (lower-cased) spares matching skill_view bodies; pass None for the + pressure pass, which overrides the guard. + """ + msg = result[idx] + if msg.get("role") != "tool": + return False + content = msg.get("content", "") + if isinstance(content, list) or (isinstance(content, dict) and content.get("_multimodal")): + # Shared strip policy with pass 3.5 (also drops the stale api_content sidecar). + new_msg = _strip_images_from_tool_msg(msg) + if new_msg is None: + return False + result[idx] = new_msg + return True + if not isinstance(content, str) or not content or content == _PRUNED_TOOL_PLACEHOLDER: + return False + if content.startswith(("[Duplicate tool output", "[screenshot removed")): + return False + if content.startswith("[") and " chars)" in content and len(content) < 400: + return False + if len(content) <= min_prune_chars: + return False + tool_name, tool_args = call_id_to_tool.get(msg.get("tool_call_id", ""), ("unknown", "")) + if protected_skills and tool_name == "skill_view": + try: + _args = json.loads(tool_args) if tool_args else {} + except (json.JSONDecodeError, TypeError): + _args = {} + _skill = _args.get("name", "") if isinstance(_args, dict) else "" + if isinstance(_skill, str) and _skill.lower() in protected_skills: + return False + result[idx] = {**msg, "content": _summarize_tool_result(tool_name, tool_args, content)} + return True + + def _pressure_demote_tail( + self, result: List[Dict[str, Any]], prune_boundary: int, protect_tail_tokens: int, + call_id_to_tool: Dict[str, tuple[str, str]], min_prune_chars: int, + ) -> int: + """Pass 4: demote inside the protected tail when it alone exceeds the soft budget (#61932). + + Keeps a short recent floor verbatim; overrides the skill guard (else the dead-end recurs). + Returns the number of tool results demoted (arg truncations are logged but not counted). + """ + soft_ceiling = int(protect_tail_tokens * 1.5) + demote_end = len(result) - min(_PRESSURE_KEEP_RECENT_MESSAGES, len(result)) + start = max(0, prune_boundary) + + def _protected_region_tokens() -> int: + return sum(_estimate_msg_budget_tokens(result[i]) for i in range(start, len(result))) + + demoted = 0 + pressure_hits = 0 + + def _shrink_at(i: int) -> None: + # Each helper no-ops on the other role, so both may run unconditionally. + nonlocal demoted, pressure_hits + if self._demote_tool_result_at(result, i, call_id_to_tool, min_prune_chars): + demoted += 1 + pressure_hits += 1 + if self._truncate_tool_call_args_at(result, i): + pressure_hits += 1 + + if demote_end <= prune_boundary or _protected_region_tokens() <= soft_ceiling: + return 0 + for i in range(start, demote_end): + _shrink_at(i) + if _protected_region_tokens() <= soft_ceiling: + break + # If the recent floor is still dominated by huge tool bodies, demote all but the newest. + if _protected_region_tokens() > soft_ceiling: + last_tool_idx = next((i for i in range(len(result) - 1, -1, -1) if result[i].get("role") == "tool"), None) + for i in range(start, len(result)): + if i != last_tool_idx: + _shrink_at(i) + # Last resort: the newest body alone may exceed the soft budget; summarize it. + if ( + last_tool_idx is not None + and last_tool_idx >= prune_boundary + and _protected_region_tokens() > soft_ceiling + ) and self._demote_tool_result_at(result, last_tool_idx, call_id_to_tool, min_prune_chars): + demoted += 1 + pressure_hits += 1 + if pressure_hits and not self.quiet_mode: + logger.info( + "Pre-compression pressure demotion: reclaimed protected-tail " + "tool output (%d change(s); protected region now ~%s tokens, " + "soft ceiling %s)", + pressure_hits, f"{_protected_region_tokens():,}", f"{soft_ceiling:,}", + ) + return demoted + def _prune_old_tool_results( self, messages: List[Dict[str, Any]], protect_tail_count: int, protect_tail_tokens: int | None = None, @@ -2890,216 +3072,26 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): """ if not messages: return messages, 0 - result = [m.copy() for m in messages] - pruned = 0 - - call_id_to_tool: Dict[str, tuple] = {} - for msg in result: - if msg.get("role") == "assistant": - for tc in msg.get("tool_calls") or []: - if isinstance(tc, dict): - cid = tc.get("id", "") - fn = tc.get("function", {}) - call_id_to_tool[cid] = (fn.get("name", "unknown"), fn.get("arguments", "")) - else: - cid = getattr(tc, "id", "") or "" - fn = getattr(tc, "function", None) - name = getattr(fn, "name", "unknown") if fn else "unknown" - args_str = getattr(fn, "arguments", "") if fn else "" - call_id_to_tool[cid] = (name, args_str) - - if protect_tail_tokens is not None and protect_tail_tokens > 0: - # Token-budget walk; cap the message-count floor like tail-cut so a bulky recent run stays prunable. - accumulated = 0 - boundary = len(result) - min_protect = min( - protect_tail_count, - len(result), - _MAX_TAIL_MESSAGE_FLOOR, - ) - # Charge thinking on the newest turn only (parity with tail-cut and the estimator). - _newest_asst_idx = _last_assistant_index(result) - _charge_all_thinking = self._stale_thinking_on_wire() - for i in range(len(result) - 1, -1, -1): - msg = result[i] - msg_tokens = _estimate_msg_budget_tokens( - msg, - charge_stale_thinking=( - _charge_all_thinking or i == _newest_asst_idx - ), - ) - if accumulated + msg_tokens > protect_tail_tokens and (len(result) - i) >= min_protect: - boundary = i - break - accumulated += msg_tokens - boundary = i - # Apply the floor in count-space: `max` in index-space would invert (smaller index = MORE protected). - budget_protect_count = len(result) - boundary - protected_count = max(budget_protect_count, min_protect) - prune_boundary = len(result) - protected_count - else: - prune_boundary = len(result) - protect_tail_count - - # Pass 1: dedup identical tool results; keep the newest copy, back-reference older ones. - content_hashes: dict = {} # hash -> (index, tool_call_id) - for i in range(len(result) - 1, -1, -1): - msg = result[i] - if msg.get("role") != "tool": - continue - content = msg.get("content") or "" - if isinstance(content, list): - continue - if not isinstance(content, str): - # Non-string/multimodal-envelope shapes can't be hashed by text. - continue - if len(content) < _PRUNE_MIN_CHARS: - continue - h = hashlib.md5(content.encode("utf-8", errors="replace")).hexdigest()[:12] - if h in content_hashes: - result[i] = {**msg, "content": "[Duplicate tool output — same content as a more recent call]"} - pruned += 1 - else: - content_hashes[h] = (i, msg.get("tool_call_id", "?")) - + call_id_to_tool = _tool_calls_by_id(result) + prune_boundary = self._prune_boundary(result, protect_tail_count, protect_tail_tokens) + pruned = self._dedupe_tool_results(result) # Just-loaded / tail-referenced skills keep full skill_view bodies through the ordinary passes. protected_skills = _collect_protected_skill_names(result, prune_boundary) - - def _demote_tool_result_at(idx: int, *, spare_protected_skills: bool = True) -> bool: - """Replace the tool result at ``idx`` with a 1-line summary; True if modified.""" - nonlocal pruned - msg = result[idx] - if msg.get("role") != "tool": - return False - content = msg.get("content", "") - if isinstance(content, list) or ( - isinstance(content, dict) and content.get("_multimodal") - ): - # Shared strip policy with pass 3.5 (also drops the stale api_content sidecar). - new_msg = _strip_images_from_tool_msg(msg) - if new_msg is None: - return False - result[idx] = new_msg - pruned += 1 - return True - if not isinstance(content, str): - return False - if not content or content == _PRUNED_TOOL_PLACEHOLDER: - return False - if content.startswith("[Duplicate tool output"): - return False - if content.startswith("[") and " chars)" in content and len(content) < 400: - return False - if content.startswith("[screenshot removed"): - return False - if len(content) <= min_prune_chars: - return False - call_id = msg.get("tool_call_id", "") - tool_name, tool_args = call_id_to_tool.get(call_id, ("unknown", "")) - if spare_protected_skills and tool_name == "skill_view" and protected_skills: - # Protected skills survive here; pass-4 pressure demotion overrides this. - try: - _args = json.loads(tool_args) if tool_args else {} - except (json.JSONDecodeError, TypeError): - _args = {} - _skill = _args.get("name", "") if isinstance(_args, dict) else "" - if isinstance(_skill, str) and _skill.lower() in protected_skills: - return False - summary = _summarize_tool_result(tool_name, tool_args, content) - result[idx] = {**msg, "content": summary} - pruned += 1 - return True - - def _truncate_tool_call_args_at(idx: int) -> bool: - """Shrink large tool_call argument payloads at ``idx``.""" - msg = result[idx] - if msg.get("role") != "assistant" or not msg.get("tool_calls"): - return False - new_tcs = [] - modified = False - for tc in msg["tool_calls"]: - if isinstance(tc, dict): - args = tc.get("function", {}).get("arguments", "") - if len(args) > 500: - new_args = _truncate_tool_call_args_json(args) - if new_args != args: - tc = {**tc, "function": {**tc["function"], "arguments": new_args}} - modified = True - new_tcs.append(tc) - if modified: - result[idx] = {**msg, "tool_calls": new_tcs} - return modified - - # Pass 2: summarize old tool results. + # Pass 2: summarize old tool results. Pass 3: shrink large tool_call arguments INSIDE the + # parsed JSON so the result stays valid; otherwise providers 400 on every turn until the + # call leaves the window. for i in range(max(0, prune_boundary)): - _demote_tool_result_at(i) - - # Pass 3: shrink large tool_call arguments INSIDE the parsed JSON so the result stays valid - # JSON; otherwise providers 400 on every turn until the call leaves the window. + pruned += self._demote_tool_result_at(result, i, call_id_to_tool, min_prune_chars, protected_skills) for i in range(max(0, prune_boundary)): - _truncate_tool_call_args_at(i) - + self._truncate_tool_call_args_at(result, i) # Pass 3.5: retire image payloads inside the protected tail; re-sent embeds otherwise make # compression look ineffective and trip anti-thrash. Newest frames stay live. pruned += _retire_stale_tool_result_images(result) - - # Pass 4: pressure demotion inside the protected tail when it alone exceeds the soft budget, - # keeping a short recent floor verbatim (#61932). if protect_tail_tokens is not None and protect_tail_tokens > 0 and result: - soft_ceiling = int(protect_tail_tokens * 1.5) - keep_recent = min(_PRESSURE_KEEP_RECENT_MESSAGES, len(result)) - demote_end = len(result) - keep_recent - - def _protected_region_tokens() -> int: - start = max(0, prune_boundary) - return sum( - _estimate_msg_budget_tokens(result[i]) - for i in range(start, len(result)) - ) - - if demote_end > prune_boundary and _protected_region_tokens() > soft_ceiling: - pressure_hits = 0 - for i in range(max(0, prune_boundary), demote_end): - # Pressure passes override the skill guard, else the #61932 dead-end recurs. - if _demote_tool_result_at(i, spare_protected_skills=False): - pressure_hits += 1 - if _truncate_tool_call_args_at(i): - pressure_hits += 1 - if _protected_region_tokens() <= soft_ceiling: - break - # If the recent floor is still dominated by huge tool bodies, demote all but the newest. - if _protected_region_tokens() > soft_ceiling: - last_tool_idx = None - for i in range(len(result) - 1, -1, -1): - if result[i].get("role") == "tool": - last_tool_idx = i - break - for i in range(max(0, prune_boundary), len(result)): - if last_tool_idx is not None and i == last_tool_idx: - continue - # _demote_tool_result_at / _truncate_tool_call_args_at each no-op on the - # other role, so both may run unconditionally. - if _demote_tool_result_at(i, spare_protected_skills=False): - pressure_hits += 1 - if _truncate_tool_call_args_at(i): - pressure_hits += 1 - # Last resort: the newest body alone may exceed the soft budget; summarize it. - if ( - last_tool_idx is not None - and last_tool_idx >= prune_boundary - and _protected_region_tokens() > soft_ceiling - ) and _demote_tool_result_at(last_tool_idx, spare_protected_skills=False): - pressure_hits += 1 - if pressure_hits and not self.quiet_mode: - logger.info( - "Pre-compression pressure demotion: reclaimed protected-tail " - "tool output (%d change(s); protected region now ~%s tokens, " - "soft ceiling %s)", - pressure_hits, - f"{_protected_region_tokens():,}", - f"{soft_ceiling:,}", - ) - + pruned += self._pressure_demote_tail( + result, prune_boundary, protect_tail_tokens, call_id_to_tool, min_prune_chars, + ) return result, pruned def prune_tool_results_only( From f00024ecc8d71d124c338c2a5c510611090c60a7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:48:29 -0700 Subject: [PATCH 05/27] refactor(agent/context_compressor): lift static-fallback helpers to module level, extract anchor collection --- agent/context_compressor.py | 166 +++++++++++++++++------------------- 1 file changed, 77 insertions(+), 89 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 0a55c93254..757386e469 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1177,6 +1177,42 @@ def _collect_path_mentions(text: str, relevant_files: list[str], *, limit: int = _dedupe_append(relevant_files, match.rstrip(".,:;"), limit=limit) +def _collect_paths_from_jsonish(obj: Any, relevant_files: list[str]) -> None: + """Harvest path-like values (known keys + inline mentions) from parsed tool arguments.""" + if isinstance(obj, dict): + for key, val in obj.items(): + if key in {"path", "workdir", "file_path", "output_path"} and isinstance(val, str): + _dedupe_append(relevant_files, val, limit=12) + _collect_paths_from_jsonish(val, relevant_files) + elif isinstance(obj, list): + for val in obj: + _collect_paths_from_jsonish(val, relevant_files) + elif isinstance(obj, str): + _collect_path_mentions(obj, relevant_files) + + +def _compact_fallback_turn(value: Any) -> str: + """One-line, redacted, length-capped rendering of a turn's content for the static fallback.""" + text = _redact_compaction_text(_content_text_for_contains(value)) + text = re.sub(r"\bgh[pousr]_[A-Za-z0-9_]{8,}\b", "[REDACTED]", text) + text = re.sub(r"\s+", " ", text).strip() + if len(text) > _FALLBACK_TURN_MAX_CHARS: + text = text[: _FALLBACK_TURN_MAX_CHARS - 15].rstrip() + " ...[truncated]" + return re.sub(r"\bgh[pousr]_[A-Za-z0-9_.-]+", "[REDACTED]", text) + + +def _bullets(items: list[str], limit: int = 8) -> str: + """Markdown bullets of the first ``limit`` distinct non-blank items, or ``None.``.""" + unique: list[str] = [] + for item in items: + item = item.strip() + if item and item not in unique: + unique.append(item) + if len(unique) >= limit: + break + return "\n".join(f"- {item}" for item in unique) if unique else "None." + + def _content_length_for_budget(raw_content: Any) -> int: """Return the effective char-length of a message's content for token budgeting. @@ -3240,16 +3276,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return "\n\n".join(parts) - def _build_static_fallback_summary( - self, - turns_to_summarize: List[Dict[str, Any]], - reason: str | None = None, - ) -> str: - """Build a deterministic handoff when the LLM summarizer is unavailable. - - Keeps locally extractable anchors (recent user asks, actions, files/commands, errors) - in the normal summary structure so downstream prompts recover gracefully. - """ + def _fallback_anchors(self, turns_to_summarize: List[Dict[str, Any]]) -> Dict[str, list[str]]: + """Locally extractable anchors: user asks, actions, files, blockers, last dropped turns.""" user_asks: list[str] = [] assistant_actions: list[str] = [] tool_actions: list[str] = [] @@ -3257,34 +3285,6 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): blockers: list[str] = [] last_dropped_turns: list[str] = [] - def _compact_fallback_turn(value: Any) -> str: - text = _redact_compaction_text(_content_text_for_contains(value)) - text = re.sub(r"\bgh[pousr]_[A-Za-z0-9_]{8,}\b", "[REDACTED]", text) - text = re.sub(r"\s+", " ", text).strip() - if len(text) > _FALLBACK_TURN_MAX_CHARS: - text = text[: _FALLBACK_TURN_MAX_CHARS - 15].rstrip() + " ...[truncated]" - return re.sub(r"\bgh[pousr]_[A-Za-z0-9_.-]+", "[REDACTED]", text) - - def _remember_dropped_turn(label: str, text: str, *, limit: int = 8) -> None: - text = text.strip() - if not text: - return - last_dropped_turns.append(f"{label}: {text}") - if len(last_dropped_turns) > limit: - del last_dropped_turns[0] - - def _collect_paths_from_jsonish(obj: Any) -> None: - if isinstance(obj, dict): - for key, val in obj.items(): - if key in {"path", "workdir", "file_path", "output_path"} and isinstance(val, str): - _dedupe_append(relevant_files, val, limit=12) - _collect_paths_from_jsonish(val) - elif isinstance(obj, list): - for val in obj: - _collect_paths_from_jsonish(val) - elif isinstance(obj, str): - _collect_path_mentions(obj, relevant_files) - call_id_to_tool: dict[str, tuple[str, str]] = {} for msg in turns_to_summarize: if msg.get("role") == "assistant" and msg.get("tool_calls"): @@ -3299,27 +3299,26 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): parsed = json.loads(args) except Exception: parsed = args - _collect_paths_from_jsonish(parsed) + _collect_paths_from_jsonish(parsed, relevant_files) for msg in turns_to_summarize: role = msg.get("role", "unknown") text = _compact_fallback_turn(msg.get("content")) _collect_path_mentions(text, relevant_files) - synthetic_user = ( - role == "user" and self._is_synthetic_compression_user_turn(msg) - ) + synthetic_user = role == "user" and self._is_synthetic_compression_user_turn(msg) + tool_names = [ + _extract_tool_call_name_and_args(tc)[0] + for tc in (msg.get("tool_calls") or []) + ] if role == "assistant" else [] turn_text = text - turn_tool_names: list[str] = [] - if role == "assistant" and msg.get("tool_calls"): - for tc in msg.get("tool_calls") or []: - name, _args = _extract_tool_call_name_and_args(tc) - turn_tool_names.append(name) - if turn_tool_names: - prefix = "tool calls: " + ", ".join(turn_tool_names[:6]) - turn_text = f"{prefix}; {turn_text}" if turn_text else prefix + if tool_names: + prefix = "tool calls: " + ", ".join(tool_names[:6]) + turn_text = f"{prefix}; {turn_text}" if turn_text else prefix turn_label = "INTERNAL CONTEXT" if synthetic_user else str(role).upper() - _remember_dropped_turn(turn_label, turn_text) + if turn_text.strip(): + last_dropped_turns.append(f"{turn_label}: {turn_text.strip()}") + del last_dropped_turns[:-8] if len(text) > 600: text = text[:420].rstrip() + " ... " + text[-160:].lstrip() @@ -3327,46 +3326,36 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if role == "user" and text and not synthetic_user: user_asks.append(text) elif role == "assistant": - tool_names: list[str] = [] - for tc in msg.get("tool_calls") or []: - name, _args = _extract_tool_call_name_and_args(tc) - tool_names.append(name) if tool_names: - assistant_actions.append( - "Called tool(s): " + ", ".join(tool_names[:6]) - ) + assistant_actions.append("Called tool(s): " + ", ".join(tool_names[:6])) elif text: assistant_actions.append(text) elif role == "tool": - call_id = str(msg.get("tool_call_id") or "") - tool_name, tool_args = call_id_to_tool.get(call_id, ("unknown", "")) - tool_actions.append( - _summarize_tool_result(tool_name, tool_args, text or "") - ) - if re.search( - r"\b(error|failed|exception|traceback|timeout|timed out|fatal)\b", - text, - re.I, - ): + tool_name, tool_args = call_id_to_tool.get(str(msg.get("tool_call_id") or ""), ("unknown", "")) + tool_actions.append(_summarize_tool_result(tool_name, tool_args, text or "")) + if re.search(r"\b(error|failed|exception|traceback|timeout|timed out|fatal)\b", text, re.I): blockers.append(text[:500]) + return { + "user_asks": user_asks, + "completed": [f"{idx}. {item}" for idx, item in enumerate((assistant_actions + tool_actions)[:12], start=1)], + "relevant_files": relevant_files, + "blockers": blockers, + "last_dropped_turns": last_dropped_turns, + } - def _bullets(items: list[str], limit: int = 8) -> str: - unique: list[str] = [] - seen: set[str] = set() - for item in items: - item = item.strip() - if not item or item in seen: - continue - seen.add(item) - unique.append(item) - if len(unique) >= limit: - break - return "\n".join(f"- {item}" for item in unique) if unique else "None." - - completed: list[str] = [] - for idx, item in enumerate((assistant_actions + tool_actions)[:12], start=1): - completed.append(f"{idx}. {item}") + def _build_static_fallback_summary( + self, + turns_to_summarize: List[Dict[str, Any]], + reason: str | None = None, + ) -> str: + """Build a deterministic handoff when the LLM summarizer is unavailable. + Keeps locally extractable anchors (recent user asks, actions, files/commands, errors) + in the normal summary structure so downstream prompts recover gracefully. + """ + anchors = self._fallback_anchors(turns_to_summarize) + user_asks = anchors["user_asks"] + completed = anchors["completed"] active_task = ( f"User asked: {user_asks[-1]!r}" if user_asks @@ -3406,7 +3395,7 @@ Recovered from a deterministic fallback because the LLM context summarizer was u Unknown from deterministic fallback. Inspect current repository/session state if needed. ## Blocked -{_bullets(blockers, limit=5)} +{_bullets(anchors["blockers"], limit=5)} ## Key Decisions None recoverable from deterministic fallback. @@ -3415,10 +3404,10 @@ None recoverable from deterministic fallback. None recoverable from deterministic fallback. ## Relevant Files -{_bullets(relevant_files, limit=12)} +{_bullets(anchors["relevant_files"], limit=12)} ## Last Dropped Turns -{_bullets(last_dropped_turns, limit=8)} +{_bullets(anchors["last_dropped_turns"], limit=8)} ## Critical Context Summary generation was unavailable, so this is a best-effort deterministic fallback for {len(turns_to_summarize)} compacted message(s).{reason_text}""" @@ -3430,8 +3419,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb summary = summary[: _FALLBACK_SUMMARY_MAX_CHARS - 42].rstrip() + "\n...[fallback summary truncated]" # Re-inject AFTER the size cap: markers live at the end, where truncation cuts. summary = _reinject_pruned_skill_markers(summary, _pruned_names) - summary = self._augment_summary_lean(summary, turns_to_summarize) - return summary + return self._augment_summary_lean(summary, turns_to_summarize) def _demote_stale_tail_tools( self, messages: List[Dict[str, Any]], tail_start: int, From 6549a8bfb6d46d91d2265d328150b2690b0bcd15 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:51:31 -0700 Subject: [PATCH 06/27] refactor(agent/context_compressor): extract _call_summary_llm and a summary-failure classifier dataclass --- agent/context_compressor.py | 297 ++++++++++++++++++------------------ 1 file changed, 151 insertions(+), 146 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 757386e469..9a9e06208b 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -636,6 +636,70 @@ class _HandoffScan: previous_summary_before: Optional[str] has_user_turn_before: Optional[bool] +def _short_error_text(e: Exception, limit: int = 220) -> str: + """Error text (or class name) capped for durable cooldown rows and telemetry.""" + text = str(e).strip() or e.__class__.__name__ + if len(text) > limit: + text = text[: limit - 3].rstrip() + "..." + return text + + +@dataclass +class _SummaryFailureKind: + """Transient-failure classes of a summary call (several may hold at once).""" + + model_not_found: bool + timeout: bool + json_decode: bool + streaming_closed: bool + empty_content: bool + truncated: bool + + def fallback_reason(self) -> str: + """Reason string for the one-shot main-model retry log line, most specific first.""" + for flagged, reason in ( + (self.json_decode, "returned invalid JSON"), + (self.truncated, "returned a truncated summary (output token cap)"), + (self.empty_content, "returned empty content"), + (self.model_not_found, "unavailable"), + (self.streaming_closed, "closed stream prematurely"), + (self.timeout, "timed out"), + ): + if flagged: + return reason + return "failed" + + +def _classify_summary_failure(e: Exception) -> _SummaryFailureKind: + """Classify a summary-call exception by status code / message shape.""" + status = getattr(e, "status_code", None) or getattr(getattr(e, "response", None), "status_code", None) + err = str(e).lower() + return _SummaryFailureKind( + # Permanent-looking error on a distinct summary model: fall back to main instead of cooldown. + model_not_found=( + status in {404, 503} + or "model_not_found" in err + or "does not exist" in err + or "no available channel" in err + ), + timeout=status in {408, 429, 502, 504} or "timeout" in err or "timed out" in err, + # Malformed/non-JSON bodies (HTML 502 as application/json) surface as JSONDecodeError or + # APIResponseValidationError "expecting value"; treat as transient. + json_decode=isinstance(e, json.JSONDecodeError) or "expecting value" in err, + # httpx premature-close errors are transient; treat like a timeout, not a 60s cooldown. + streaming_closed=_is_connection_error(e), + # HTTP 200 with empty body from a degraded provider, plus the sibling "no usable response" + # shapes from _validate_llm_response. + empty_content=isinstance(e, RuntimeError) and ( + "empty content" in err + or "llm returned none response" in err + or "llm returned invalid response" in err + ), + # Truncated summary: one main-model retry, then ABORT preserving the session. + truncated=isinstance(e, RuntimeError) and _TRUNCATED_SUMMARY_MARKER in err, + ) + + # Summary failures that abort compress() regardless of abort_on_summary_failure, in precedence # order: (flag attribute, telemetry failure_class, user-facing warning with %d preserved messages). _TERMINAL_SUMMARY_FAILURES = ( @@ -3559,10 +3623,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb "Falling back to main model '%s' for compression.", self.summary_model, reason, e, self.model, ) - _err_text = str(e).strip() or e.__class__.__name__ - if len(_err_text) > 220: - _err_text = _err_text[:217].rstrip() + "..." - self._last_aux_model_failure_error = _err_text + self._last_aux_model_failure_error = _short_error_text(e) self._last_aux_model_failure_model = self.summary_model telemetry = getattr(self, "_active_compression_telemetry", None) if isinstance(telemetry, dict): @@ -3571,6 +3632,74 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb self.summary_model = "" # empty = use main model self._clear_compression_failure_cooldown() # no cooldown — retry immediately + def _call_summary_llm(self, prompt: str, prompt_started_at: float) -> str: + """Issue the single aux summary call; return validated content text. + + Raises RuntimeError for empty content or a length-truncated (PARTIAL) summary so the + failure routes through main-model fallback + cooldown instead of wiping the compacted turns. + """ + call_kwargs: Dict[str, Any] = { + "task": "compression", + "main_runtime": { + "model": self.model, + "provider": self.provider, + "base_url": self.base_url, + "api_key": self.api_key, + "api_mode": self.api_mode, + }, + "messages": [{"role": "user", "content": prompt}], + # NO max_tokens: Anthropic/NIM wires forward it and a hard cap truncates summaries + # (thinking models burn it on reasoning). Timeout comes from call_llm config. + } + if self.summary_model: + call_kwargs["model"] = self.summary_model + # call_llm writes the route it actually selected; never pre-resolve a second, stale pair. + _aux_route: Dict[str, str] = {} + call_kwargs["route_info"] = _aux_route + # Pinned route (stall fallback) overrides task routing so the retry leaves the stalled backend. + call_kwargs.update(_pinned_summary_call_kwargs()) + _aux_call_start = time.monotonic() + _latency_info: Dict[str, int] = { + "prompt_build_ms": max(0, int((_aux_call_start - prompt_started_at) * 1000)) + } + call_kwargs["latency_info"] = _latency_info + try: + # Compression is atomic: shield the summary call from gateway interrupts. Re-entrant. + with aux_interrupt_protection(): + response = call_llm(**call_kwargs) + finally: + route_known = bool(_aux_route.get("provider") and _aux_route.get("model")) + _aux_model = _aux_route.get("model") or self.summary_model or self.model or "" + self._record_aux_compression_call( + prompt_messages=call_kwargs["messages"], + # max_tokens is intentionally absent; .get() keeps the telemetry hook from breaking the call. + max_tokens=call_kwargs.get("max_tokens"), + duration_ms=int((time.monotonic() - _aux_call_start) * 1000), + aux_provider=_aux_route.get("provider") or self.provider or "", + aux_model=_aux_model, + effective_aux_context=self.context_length if route_known and _aux_model == self.model else None, + phase_timings=_latency_info, + ) + if self._compression_cancelled(): + raise AuxiliaryExplicitCancellation() + # Reasoning-field fallback (DeepSeek/Qwen/Kimi put the summary in reasoning_content); capped. + content = extract_content_or_reasoning(response, max_reasoning_chars=8000) + if not content.strip(): + raise RuntimeError( + "Context compression LLM returned empty content " + f"(provider={self.provider or 'auto'} " + f"model={self.summary_model or self.model})" + ) + if _response_finish_reason(response) == "length": + raise RuntimeError( + "Context compression summary was truncated " + f"({_TRUNCATED_SUMMARY_MARKER}): generation hit the output " + "token cap and the summary is incomplete " + f"(provider={self.provider or 'auto'} " + f"model={self.summary_model or self.model})" + ) + return content + def _generate_summary( self, turns_to_summarize: List[Dict[str, Any]], @@ -3585,12 +3714,11 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb prompt_started_at = time.monotonic() if self._compression_cancelled(): raise AuxiliaryExplicitCancellation() - now = prompt_started_at # bypass_cooldown: provider-proven overflow gets ONE real attempt while armed. - if now < self._summary_failure_cooldown_until and not bypass_cooldown: + if prompt_started_at < self._summary_failure_cooldown_until and not bypass_cooldown: logger.debug( "Skipping context summary during cooldown (%.0fs remaining)", - self._summary_failure_cooldown_until - now, + self._summary_failure_cooldown_until - prompt_started_at, ) return None @@ -3610,10 +3738,8 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb _pruned_skill_names.append(_name) del _pruned_skill_names[_MAX_PRUNED_SKILL_MARKERS:] # Lean mode even-samples oversized input (one bounded request, never a second). - if getattr(self, "tail_mode", "lean") == "lean": - content_to_summarize = self._sample_summary_input(content_to_summarize) - else: - content_to_summarize = self._bound_summary_input(content_to_summarize) + bound = self._sample_summary_input if getattr(self, "tail_mode", "lean") == "lean" else self._bound_summary_input + content_to_summarize = bound(content_to_summarize) has_user_turn = getattr(self, "_summary_has_user_turn", None) if has_user_turn is None: has_user_turn = self._transcript_has_real_user_turn(turns_to_summarize) @@ -3626,84 +3752,10 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb ) try: - call_kwargs = { - "task": "compression", - "main_runtime": { - "model": self.model, - "provider": self.provider, - "base_url": self.base_url, - "api_key": self.api_key, - "api_mode": self.api_mode, - }, - "messages": [{"role": "user", "content": prompt}], - # NO max_tokens: Anthropic/NIM wires forward it and a hard cap truncates summaries - # (thinking models burn it on reasoning). Timeout comes from call_llm config. - } - if self.summary_model: - call_kwargs["model"] = self.summary_model - # call_llm writes the route it actually selected; never pre-resolve a second, stale pair. - _aux_route: Dict[str, str] = {} - call_kwargs["route_info"] = _aux_route - # Pinned route (stall fallback) overrides task routing so the retry leaves the stalled backend. - _pinned_route = _pinned_summary_call_kwargs() - if _pinned_route: - call_kwargs.update(_pinned_route) - # Compression is atomic: shield the summary call from gateway interrupts. Re-entrant. - _aux_call_start = time.monotonic() - _latency_info: Dict[str, int] = { - "prompt_build_ms": max(0, int((_aux_call_start - prompt_started_at) * 1000)) - } - call_kwargs["latency_info"] = _latency_info - try: - with aux_interrupt_protection(): - response = call_llm(**call_kwargs) - finally: - route_known = bool(_aux_route.get("provider") and _aux_route.get("model")) - _aux_provider = _aux_route.get("provider") or self.provider or "" - _aux_model = _aux_route.get("model") or self.summary_model or self.model or "" - _aux_context = ( - self.context_length - if route_known and _aux_model == self.model - else None - ) - self._record_aux_compression_call( - prompt_messages=call_kwargs["messages"], - # max_tokens is intentionally absent; .get() keeps the telemetry hook from breaking the call. - max_tokens=call_kwargs.get("max_tokens"), - duration_ms=int((time.monotonic() - _aux_call_start) * 1000), - aux_provider=_aux_provider, - aux_model=_aux_model, - effective_aux_context=_aux_context, - phase_timings=_latency_info, - ) - if self._compression_cancelled(): - raise AuxiliaryExplicitCancellation() - # Reasoning-field fallback (DeepSeek/Qwen/Kimi put the summary in reasoning_content); capped. - content = extract_content_or_reasoning( - response, max_reasoning_chars=8000 - ) - # Some proxies return HTTP 200 with empty content; treat as failure so it routes - # through main-model fallback + cooldown instead of wiping the compacted turns. - if not content.strip(): - raise RuntimeError( - "Context compression LLM returned empty content " - f"(provider={self.provider or 'auto'} " - f"model={self.summary_model or self.model})" - ) - # finish_reason "length" means PARTIAL text; never persist it as a checkpoint. - if _response_finish_reason(response) == "length": - raise RuntimeError( - "Context compression summary was truncated " - f"({_TRUNCATED_SUMMARY_MARKER}): generation hit the output " - "token cap and the summary is incomplete " - f"(provider={self.provider or 'auto'} " - f"model={self.summary_model or self.model})" - ) + content = self._call_summary_llm(prompt, prompt_started_at) # Strip blocks: they would be stored, injected, and compounded on every iterative update. from agent.agent_runtime_helpers import strip_think_blocks - stripped = strip_think_blocks(None, content).strip() - if stripped: - content = stripped + content = strip_think_blocks(None, content).strip() or content # The summarizer may echo secrets verbatim; redact the output too. summary = _redact_compaction_text(content.strip()) # Restore any [SKILL_PRUNED] marker the summarizer paraphrased away. @@ -3715,10 +3767,8 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb self._clear_compression_failure_cooldown() self._summary_model_fallen_back = False self._last_summary_error = None - self._last_summary_auth_failure = False - self._last_summary_network_failure = False - self._last_summary_empty_content_failure = False - self._last_summary_truncated_failure = False + for flag, _class, _msg in _TERMINAL_SUMMARY_FAILURES: + setattr(self, flag, False) return self._with_summary_prefix(summary) except Exception as e: return self._on_summary_failure(e, turns_to_summarize, focus_topic, memory_context) @@ -4002,44 +4052,16 @@ This compaction should PRIORITISE preserving all information related to the focu "for %d seconds.", _SUMMARY_FAILURE_COOLDOWN_SECONDS) return None - # Permanent-looking error on a distinct summary model: fall back to main instead of cooldown. - _status = getattr(e, "status_code", None) or getattr(getattr(e, "response", None), "status_code", None) - _err_str = str(e).lower() - _is_model_not_found = ( - _status in {404, 503} - or "model_not_found" in _err_str - or "does not exist" in _err_str - or "no available channel" in _err_str - ) - _is_timeout = ( - _status in {408, 429, 502, 504} - or "timeout" in _err_str - or "timed out" in _err_str - ) - # Malformed/non-JSON bodies (HTML 502 as application/json) surface as JSONDecodeError or - # APIResponseValidationError "expecting value"; treat as transient. - _is_json_decode = ( - isinstance(e, json.JSONDecodeError) - or "expecting value" in _err_str - ) - # httpx premature-close errors are transient; treat like a timeout, not a 60s cooldown. - _is_streaming_closed = _is_connection_error(e) - # HTTP 200 with empty body from a degraded provider. - _is_empty_content = isinstance(e, RuntimeError) and ( - "empty content" in _err_str - # Sibling "no usable response" shapes from _validate_llm_response — same class. - or "llm returned none response" in _err_str - or "llm returned invalid response" in _err_str - ) - # Truncated summary: one main-model retry, then ABORT preserving the session. - _is_truncated_summary = ( - isinstance(e, RuntimeError) - and _TRUNCATED_SUMMARY_MARKER in _err_str - ) + kind = _classify_summary_failure(e) + _is_model_not_found = kind.model_not_found + _is_timeout = kind.timeout + _is_json_decode = kind.json_decode + _is_streaming_closed = kind.streaming_closed + _is_empty_content = kind.empty_content + _is_truncated_summary = kind.truncated # Auth/permission/quota failures are not retryable: flag so compress() preserves the # session. A distinct summary_model still gets the one-shot main-model fallback. - _is_access_or_quota_error = _is_summary_access_or_quota_error(e) - if _is_access_or_quota_error: + if _is_summary_access_or_quota_error(e): # Field name kept for caller compatibility; now covers the whole access/quota class. self._last_summary_auth_failure = True if _is_json_decode and not _is_model_not_found and not _is_timeout: @@ -4061,22 +4083,7 @@ This compaction should PRIORITISE preserving all information related to the focu and self.summary_model != self.model and not getattr(self, "_summary_model_fallen_back", False) ): - _reason = next( - ( - reason - for flagged, reason in ( - (_is_json_decode, "returned invalid JSON"), - (_is_truncated_summary, "returned a truncated summary (output token cap)"), - (_is_empty_content, "returned empty content"), - (_is_model_not_found, "unavailable"), - (_is_streaming_closed, "closed stream prematurely"), - (_is_timeout, "timed out"), - ) - if flagged - ), - "failed", - ) - self._fallback_to_main_for_compression(e, _reason) + self._fallback_to_main_for_compression(e, kind.fallback_reason()) return self._generate_summary( turns_to_summarize, focus_topic=focus_topic, @@ -4091,9 +4098,7 @@ This compaction should PRIORITISE preserving all information related to the focu _transient_cooldown = 30 else: _transient_cooldown = 60 - err_text = str(e).strip() or e.__class__.__name__ - if len(err_text) > 220: - err_text = err_text[:217].rstrip() + "..." + err_text = _short_error_text(e) self._record_compression_failure_cooldown(_transient_cooldown, err_text) self._last_summary_error = err_text # Terminal network/empty-content failure after any fallback: flag so compress() ABORTS From 6fe54ec6d7689b24d0d5fe43fe4ad086e00a7cb4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:00:02 -0700 Subject: [PATCH 07/27] refactor(agent/context_compressor): AST-identical reflow of short multi-line statements and headers --- agent/context_compressor.py | 394 ++++++++---------------------------- 1 file changed, 83 insertions(+), 311 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 9a9e06208b..f3c9a92258 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -59,14 +59,7 @@ _SUMMARY_ROUTE_PIN: contextvars.ContextVar[Optional[Dict[str, Any]]] = ( ) # ``timeout`` is included so a fallback entry keeps its own deadline. -_PINNED_ROUTE_FIELDS: tuple[str, ...] = ( - "provider", - "model", - "base_url", - "api_key", - "api_mode", - "timeout", -) +_PINNED_ROUTE_FIELDS: tuple[str, ...] = ("provider", "model", "base_url", "api_key", "api_mode", "timeout") @contextlib.contextmanager @@ -99,11 +92,7 @@ def _pinned_summary_call_kwargs() -> Dict[str, Any]: route = take_pinned_summary_route() if not route: return {} - return { - field: route[field] - for field in _PINNED_ROUTE_FIELDS - if route.get(field) not in (None, "") - } + return {field: route[field] for field in _PINNED_ROUTE_FIELDS if route.get(field) not in (None, "")} _SUMMARY_PERMANENT_QUOTA_MARKERS: tuple[str, ...] = ( @@ -116,10 +105,7 @@ _SUMMARY_PERMANENT_QUOTA_MARKERS: tuple[str, ...] = ( "out of extra usage", ) -_SUMMARY_MISSING_CREDENTIAL_MARKERS: tuple[str, ...] = ( - "no api key was found", - "no api key found", -) +_SUMMARY_MISSING_CREDENTIAL_MARKERS: tuple[str, ...] = ("no api key was found", "no api key found") _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS: tuple[str, ...] = ( "session hygiene compression timed out", @@ -134,9 +120,7 @@ def _is_hygiene_preagent_only_cooldown(error: object) -> bool: failure and must never block the in-agent compressor. """ text = str(error or "").strip().casefold() - return any( - marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS - ) + return any(marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS) def _response_finish_reason(response: Any) -> str: @@ -351,11 +335,7 @@ def _prune_stale_reasoning_replay(messages: List[Dict[str, Any]]) -> int: items = msg.get(key) if not isinstance(items, list) or not items: continue - kept = [ - item - for item in items - if isinstance(item, dict) and item.get("type") == "compaction" - ] + kept = [item for item in items if isinstance(item, dict) and item.get("type") == "compaction"] if len(kept) == len(items): continue # nothing stale in this sidecar if kept: @@ -392,10 +372,7 @@ def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool: # compressor marker. Tool messages are handled only by the stub/keep-recent pass. if msg.get("role") == "tool": return False - if ( - msg.get("role") in ("user", "assistant") - and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY) - ): + if msg.get("role") in ("user", "assistant") and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY): return False head = content[:280] return ( @@ -490,10 +467,7 @@ def salvage_grown_transcript( if estimate_messages_tokens_rough(out) >= budget: _salvage_reduce_todo_snapshot(out) - if not any( - isinstance(message, dict) and message.get("role") == "user" - for message in out - ): + if not any(isinstance(message, dict) and message.get("role") == "user" for message in out): return None if estimate_messages_tokens_rough(out) < budget: return out @@ -753,9 +727,7 @@ def _next_timeout_cooldown(compressor: Any) -> int: Module-level (not a method) so callers that bind a single real method onto a stub still exercise the ladder. """ - compressor._consecutive_timeout_failures = ( - getattr(compressor, "_consecutive_timeout_failures", 0) + 1 - ) + compressor._consecutive_timeout_failures = getattr(compressor, "_consecutive_timeout_failures", 0) + 1 return _TIMEOUT_COOLDOWN_LADDER[ min(compressor._consecutive_timeout_failures, len(_TIMEOUT_COOLDOWN_LADDER)) - 1 ] @@ -888,10 +860,7 @@ def _reinject_pruned_skill_markers(summary: str, skill_names: list[str]) -> str: """ if not skill_names: return summary - missing = [ - name for name in skill_names - if _skill_pruned_marker(name) not in summary - ] + missing = [name for name in skill_names if _skill_pruned_marker(name) not in summary] if not missing: return summary lines = [_skill_pruned_marker(name) for name in missing] @@ -925,10 +894,7 @@ _LEAN_TAIL_DEMOTE_MIN_CHARS = 1_500 def _lean_recovery_stub(tool_name: str, content_len: int, session_id: str) -> str: """One-line replacement for a demoted tail tool result.""" - hint = ( - f" Recover with session_search(query=..., session_id='{session_id}')" - if session_id else "" - ) + hint = f" Recover with session_search(query=..., session_id='{session_id}')" if session_id else "" return ( f"[{tool_name or 'tool'} output demoted at compaction — {content_len:,} " f"chars preserved in session history.{hint}]" @@ -1055,9 +1021,7 @@ def _build_anchor_index(turns: List[Dict[str, Any]]) -> str: if not counts: continue ranked = sorted(counts, key=lambda v: (-counts[v], -last_seen[v]))[:cap] - line = f"{label}: " + ", ".join( - f"{v}(x{counts[v]})" if counts[v] > 1 else v for v in ranked - ) + line = f"{label}: " + ", ".join(f"{v}(x{counts[v]})" if counts[v] > 1 else v for v in ranked) if used + len(line) > _LEAN_ANCHOR_BUDGET_CHARS: break sections.append(line) @@ -1077,9 +1041,7 @@ def _build_anchor_index(turns: List[Dict[str, Any]]) -> str: _SKILL_PRUNE_RECENT_WINDOW = 10 -def _skill_view_call_sites( - messages: List[Dict[str, Any]], -) -> list[tuple[int, str]]: +def _skill_view_call_sites(messages: List[Dict[str, Any]]) -> list[tuple[int, str]]: """Yield ``(message_index, skill_name)`` for every skill_view tool call.""" sites: list[tuple[int, str]] = [] for i, msg in enumerate(messages): @@ -1107,9 +1069,7 @@ def _skill_view_call_sites( return sites -def _collect_protected_skill_names( - messages: List[Dict[str, Any]], prune_boundary: int, -) -> set[str]: +def _collect_protected_skill_names(messages: List[Dict[str, Any]], prune_boundary: int) -> set[str]: """Skill names (lower-cased) whose skill_view bodies must survive Phase-1 demotion. Recently loaded, loaded inside the protected tail, or named by a tail user message. @@ -1130,9 +1090,7 @@ def _collect_protected_skill_names( protected: set[str] = set() for idx, skill in _skill_view_call_sites(messages): key = skill.lower() - if idx >= recent_start or idx >= tail_start or any( - key in text for text in tail_user_texts - ): + if idx >= recent_start or idx >= tail_start or any(key in text for text in tail_user_texts): protected.add(key) return protected @@ -1175,9 +1133,7 @@ _PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+") # MEDIA directives must not reach the summarizer or they get re-emitted as active. _MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+") -_HISTORICAL_TASK_SECTION_RE = re.compile( - rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)" -) +_HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)") def _redact_compaction_text(text: Any) -> str: @@ -1186,11 +1142,7 @@ def _redact_compaction_text(text: Any) -> str: ``force=True`` overrides ``security.redact_secrets: false``; URL credentials are redacted too, since summaries persist and re-enter every later prompt. """ - return redact_sensitive_text( - text or "", - force=True, - redact_url_credentials=True, - ) + return redact_sensitive_text(text or "", force=True, redact_url_credentials=True) def _dedupe_append(items: list[str], value: str, *, limit: int) -> None: @@ -1318,31 +1270,18 @@ def _serialized_length_for_budget(value: Any) -> int: # Replay/metadata fields invisible to content/tool_calls accounting but shipped # on the wire. ``reasoning_details`` is handled by _reasoning_details_text_chars. -_REPLAY_BUDGET_KEYS = ( - "reasoning", - "reasoning_content", - "codex_reasoning_items", - "codex_message_items", -) +_REPLAY_BUDGET_KEYS = "reasoning", "reasoning_content", "codex_reasoning_items", "codex_message_items" # Keys replayed on EVERY retained assistant turn: Codex items ride every request and message # items are required for prefix-cache continuity. Generic thinking keys ship for the newest turn # only elsewhere (Anthropic strips older, Bedrock never replays, strict chat-completions reject # or pad the field); charging them everywhere overcut the tail. -_ALWAYS_REPLAYED_BUDGET_KEYS = ( - "codex_reasoning_items", - "codex_message_items", -) -_NEWEST_TURN_ONLY_BUDGET_KEYS = ( - "reasoning", - "reasoning_content", -) +_ALWAYS_REPLAYED_BUDGET_KEYS = "codex_reasoning_items", "codex_message_items" +_NEWEST_TURN_ONLY_BUDGET_KEYS = "reasoning", "reasoning_content" # Safe to strip from stale assistant turns: only the current turn's replay needs # them, and the compaction boundary already invalidated the prompt-cache prefix. -_STALE_REPLAY_PRUNE_KEYS = ( - "codex_reasoning_items", -) +_STALE_REPLAY_PRUNE_KEYS = "codex_reasoning_items", def _reasoning_details_text_chars(value: Any) -> int: @@ -1402,10 +1341,7 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) - # Charge only thinking TEXT, never the signed/base64 envelope; skip when the # same text already rides in reasoning/reasoning_content. if not (msg.get("reasoning") or msg.get("reasoning_content")): - tokens += ( - _reasoning_details_text_chars(msg.get("reasoning_details")) - // _CHARS_PER_TOKEN - ) + tokens += _reasoning_details_text_chars(msg.get("reasoning_details")) // _CHARS_PER_TOKEN return tokens @@ -1845,11 +1781,7 @@ def _summarize_tool_result_unguarded(tool_name: str, tool_args: str, tool_conten return f"[{tool_name}]{first_arg} ({content_len:,} chars result)" -def resolve_model_threshold( - model: str, - model_thresholds: dict[str, float] | None, - default: float, -) -> float: +def resolve_model_threshold(model: str, model_thresholds: dict[str, float] | None, default: float) -> float: """Resolve the effective compression threshold for a given model. Longest matching ``model_thresholds`` substring key wins; otherwise ``default``. @@ -1974,10 +1906,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): telemetry["aux_model"] = aux_model if effective_aux_context is not None: telemetry["effective_aux_context"] = _safe_int(effective_aux_context) - if ( - telemetry["effective_aux_context"] is not None - and telemetry["aux_prompt_tokens"] is not None - ): + if telemetry["effective_aux_context"] is not None and telemetry["aux_prompt_tokens"] is not None: telemetry["fit_margin"] = ( telemetry["effective_aux_context"] - telemetry["aux_prompt_tokens"] @@ -2044,9 +1973,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # Re-apply the raise-only floor so percent and tokens derive from the same window. _base = getattr(self, "_base_threshold_percent", None) if _base is not None: - self.threshold_percent = self._effective_threshold_percent( - value, _base, - ) + self.threshold_percent = self._effective_threshold_percent(value, _base) self._threshold_tokens = None self._tail_token_budget = None self._max_summary_tokens = None @@ -2087,9 +2014,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): @property def max_summary_tokens(self) -> int: if self._max_summary_tokens is None: - self._max_summary_tokens = min( - int(self.context_length * 0.05), _SUMMARY_TOKENS_CEILING, - ) + self._max_summary_tokens = min(int(self.context_length * 0.05), _SUMMARY_TOKENS_CEILING) return self._max_summary_tokens @max_summary_tokens.setter @@ -2301,9 +2226,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if deadline == self._anti_thrash_recovery_deadline: return self._anti_thrash_recovery_deadline = deadline - self._durable_write( - "set_compression_recovery_deadline", "compression recovery deadline", deadline, - ) + self._durable_write("set_compression_recovery_deadline", "compression recovery deadline", deadline) def _record_ineffective_compression_verdict(self, count: int) -> None: """Set the anti-thrash strike counter; persists only on change.""" @@ -2318,9 +2241,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): Nothing eligible existed, so nothing was "ineffective"; striking would permanently disarm auto-compaction on short sessions. The backoff still stops per-turn re-scans. """ - self._structural_no_op_backoff_until = ( - time.monotonic() + self._STRUCTURAL_NO_OP_BACKOFF_SECONDS - ) + self._structural_no_op_backoff_until = time.monotonic() + self._STRUCTURAL_NO_OP_BACKOFF_SECONDS if not self.quiet_mode: logger.warning( "Compression skipped (%s): retrying in %.0fs " @@ -2334,9 +2255,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): Does not arm real-usage verification or touch the fallback streak (nothing was committed). """ - self._record_ineffective_compression_verdict( - self._ineffective_compression_count + 1 - ) + self._record_ineffective_compression_verdict(self._ineffective_compression_count + 1) if not self.quiet_mode: logger.warning( "Compaction rejected before commit (would grow the " @@ -2375,11 +2294,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._fallback_compression_streak = 0 self._persist_fallback_compression_streak() - def get_active_compression_failure_cooldown( - self, - *, - refresh: bool = False, - ) -> Optional[Dict[str, Any]]: + def get_active_compression_failure_cooldown(self, *, refresh: bool = False) -> Optional[Dict[str, Any]]: """Return the live compression-failure cooldown for the bound session.""" if refresh: # Rollback must distinguish an authoritative empty row from a failed read; the return value can't. @@ -2441,11 +2356,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): "error": self._last_summary_error, } - def _record_compression_failure_cooldown( - self, - cooldown_seconds: float, - error: Optional[str], - ) -> None: + def _record_compression_failure_cooldown(self, cooldown_seconds: float, error: Optional[str]) -> None: now_mono = time.monotonic() new_mono = now_mono + float(cooldown_seconds) # Never shorten a longer live deadline; record the latest error text only. @@ -2525,16 +2436,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self.api_mode = api_mode self.context_length = context_length # Re-resolve from the raw config value so a switch away from an overridden model falls back correctly. - _config_pct = getattr( - self, "_config_threshold_percent", self.threshold_percent, - ) - _new_base = resolve_model_threshold( - model, self.model_thresholds, _config_pct, - ) + _config_pct = getattr(self, "_config_threshold_percent", self.threshold_percent) + _new_base = resolve_model_threshold(model, self.model_thresholds, _config_pct) self._base_threshold_percent = _new_base - self.threshold_percent = self._effective_threshold_percent( - context_length, _new_base, - ) + self.threshold_percent = self._effective_threshold_percent(context_length, _new_base) # max_tokens=None means "unspecified": keep the existing output reservation. if max_tokens is not None: self.max_tokens = self._coerce_max_tokens(max_tokens) @@ -2545,9 +2450,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # Reset to None so the property recomputes via the mode-aware path (not the legacy formula). self._tail_token_budget = None _ = self.tail_token_budget # eager recompute, same timing as before - self.max_summary_tokens = min( - int(context_length * 0.05), _SUMMARY_TOKENS_CEILING, - ) + self.max_summary_tokens = min(int(context_length * 0.05), _SUMMARY_TOKENS_CEILING) # Calibration state is only valid for the model that produced it: carried across a switch to a # smaller window it would let should_defer_preflight_to_real_usage() suppress a compaction the @@ -2606,9 +2509,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self.threshold_tokens = _effective_cap @staticmethod - def _effective_threshold_percent( - context_length: int, threshold_percent: float, - ) -> float: + def _effective_threshold_percent(context_length: int, threshold_percent: float) -> float: """Raise-only small-context threshold floor: models under 512K trigger at >= 75%.""" if context_length and context_length < _SMALL_CTX_WINDOW_LIMIT: return max(threshold_percent, _SMALL_CTX_THRESHOLD_PERCENT) @@ -2679,9 +2580,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): ) self.threshold_percent = self._base_threshold_percent # Effective trigger = min(ratio threshold, cap); re-applied in update_model(). - self.threshold_tokens_cap = self._coerce_threshold_tokens_cap( - threshold_tokens_cap, - ) + self.threshold_tokens_cap = self._coerce_threshold_tokens_cap(threshold_tokens_cap) self.protect_first_n = protect_first_n self.protect_last_n = protect_last_n # Proactive prune runs independently of the full-compression trigger. 0 = disabled. @@ -2692,9 +2591,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): _PRUNE_MIN_CHARS, int(proactive_prune_min_result_chars or 8000) ) # Every commit breaks the prompt-cache prefix; require a meaningful reclaim batch so fires are episodic. - self.proactive_prune_min_reclaim_tokens = max( - 0, int(proactive_prune_min_reclaim_tokens or 0) - ) + self.proactive_prune_min_reclaim_tokens = max(0, int(proactive_prune_min_reclaim_tokens or 0)) # A committed prune is a cache boundary: rearm only after the prompt regrows the reclaimed tokens. self._proactive_prune_rearm_tokens: int = 0 self.min_tail_user_messages = min_tail_user_messages @@ -2783,9 +2680,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it. if self._verify_compaction_cleared_threshold: if self.last_prompt_tokens >= self.threshold_tokens: - self._record_ineffective_compression_verdict( - self._ineffective_compression_count + 1, - ) + self._record_ineffective_compression_verdict(self._ineffective_compression_count + 1) if not self.quiet_mode: logger.warning( "Compaction did not clear the threshold: %d real " @@ -2865,9 +2760,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): decision, _reason = self.should_compress_info(prompt_tokens) return decision - def should_compress_info( - self, prompt_tokens: int = None - ) -> "tuple[bool, str | None]": + def should_compress_info(self, prompt_tokens: int = None) -> "tuple[bool, str | None]": """Return ``(should_compress, reason)``. ``reason`` is None unless compression is needed but blocked: ``"cooldown:"`` or @@ -2888,15 +2781,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): _cooldown_remaining = self._summary_failure_cooldown_until - time.monotonic() if _cooldown_remaining > 0: return f"cooldown:{_cooldown_remaining:.0f}" - _structural_remaining = ( - self._structural_no_op_backoff_until - time.monotonic() - ) + _structural_remaining = self._structural_no_op_backoff_until - time.monotonic() if _structural_remaining > 0: return f"structural_backoff:{_structural_remaining:.0f}" - if ( - self._ineffective_compression_count >= 2 - or self._fallback_compression_streak >= 2 - ): + if self._ineffective_compression_count >= 2 or self._fallback_compression_streak >= 2: return "ineffective" return None @@ -2936,9 +2824,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): ) return True # Structural no-op backoff is transient (in-memory, no strikes); auto-compaction resumes when it lapses. - _structural_remaining = ( - self._structural_no_op_backoff_until - time.monotonic() - ) + _structural_remaining = self._structural_no_op_backoff_until - time.monotonic() if _structural_remaining > 0: if not self.quiet_mode: logger.debug( @@ -2960,9 +2846,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._anti_thrash_recovery_deadline - _now > self._ANTI_THRASH_RECOVERY_SECONDS ): - self._set_anti_thrash_recovery_deadline( - _now + self._ANTI_THRASH_RECOVERY_SECONDS - ) + self._set_anti_thrash_recovery_deadline(_now + self._ANTI_THRASH_RECOVERY_SECONDS) elif _now >= self._anti_thrash_recovery_deadline: self._set_anti_thrash_recovery_deadline(0.0) if self._ineffective_compression_count >= 2: @@ -3215,11 +3099,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # Capability gate first: a store without archive_and_compact makes every prune a no-op. session_db = getattr(self, "_session_db", None) session_id = getattr(self, "_session_id", "") - if ( - session_db - and session_id - and not callable(getattr(session_db, "archive_and_compact", None)) - ): + if session_db and session_id and not callable(getattr(session_db, "archive_and_compact", None)): return messages, 0 pruned_msgs, pruned_count = self._prune_old_tool_results( messages, @@ -3236,11 +3116,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if reclaimed < self.proactive_prune_min_reclaim_tokens: return messages, 0 # Require a full trigger-sized regrowth before the next cache-breaking rewrite. - runway = max( - reclaimed, - self.proactive_prune_tokens, - self.proactive_prune_min_reclaim_tokens, - ) + runway = max(reclaimed, self.proactive_prune_tokens, self.proactive_prune_min_reclaim_tokens) next_rearm_tokens = after + runway if session_db and session_id: try: @@ -3420,11 +3296,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): anchors = self._fallback_anchors(turns_to_summarize) user_asks = anchors["user_asks"] completed = anchors["completed"] - active_task = ( - f"User asked: {user_asks[-1]!r}" - if user_asks - else _NO_USER_TASK_SENTINEL - ) + active_task = f"User asked: {user_asks[-1]!r}" if user_asks else _NO_USER_TASK_SENTINEL previous_summary_note = "" if self._previous_summary: previous_summary = redact_sensitive_text(self._previous_summary.strip()) @@ -3524,9 +3396,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb continue if content.startswith("[") and " chars)" in content and len(content) < 400: continue # already a summary stub - stub = _lean_recovery_stub( - msg.get("tool_name") or "", len(content), session_id, - ) + stub = _lean_recovery_stub(msg.get("tool_name") or "", len(content), session_id) replaced = {**msg, "content": stub} drop_stale_api_content(replaced) result[i] = replaced @@ -3535,20 +3405,14 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb logger.info("Lean tail: demoted %d stale tool result(s)", demoted) return result - def _augment_summary_lean( - self, summary: str, turns_to_summarize: List[Dict[str, Any]], - ) -> str: + def _augment_summary_lean(self, summary: str, turns_to_summarize: List[Dict[str, Any]]) -> str: """Append deterministic lean-mode sections to a summary; no-op in legacy mode.""" if getattr(self, "tail_mode", "lean") != "lean": return summary if _LEAN_ANCHOR_HEADING not in summary: - summary += _redact_compaction_text( - _build_anchor_index(turns_to_summarize) - ) + summary += _redact_compaction_text(_build_anchor_index(turns_to_summarize)) if _LEAN_USER_MESSAGES_HEADING not in summary: - summary += _redact_compaction_text( - _build_verbatim_user_section(turns_to_summarize) - ) + summary += _redact_compaction_text(_build_verbatim_user_section(turns_to_summarize)) if _LEAN_RECOVERY_HEADING not in summary: summary += _build_recovery_footer( getattr(self, "_session_id", "") or "", @@ -3786,10 +3650,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb Focus guidance is appended last so it takes precedence. """ _sanitized_memory_context = sanitize_memory_context(memory_context) - _serialized_memory_context = json.dumps( - _sanitized_memory_context, - ensure_ascii=False, - ) + _serialized_memory_context = json.dumps(_sanitized_memory_context, ensure_ascii=False) _serialized_memory_context = ( _serialized_memory_context.replace("&", "\\u0026") .replace("<", "\\u003c") @@ -3882,12 +3743,8 @@ Describe agent/tool work only as completed actions, state, or historical work.]" "[Runtime, configuration, and technical constraints only. Do not " "invent user preferences.]" ) - _resolved_questions_instructions = ( - "[Write exactly: None. No user-authored questions exist.]" - ) - _pending_asks_instructions = ( - "[Write exactly: None. No user-authored requests exist.]" - ) + _resolved_questions_instructions = "[Write exactly: None. No user-authored questions exist.]" + _pending_asks_instructions = "[Write exactly: None. No user-authored requests exist.]" _summarizer_preamble = ( "You are a summarization agent creating a context checkpoint. " @@ -3992,9 +3849,7 @@ Write only the summary body. Do not include any preamble or prefix.""" if self._previous_summary: # Iterative update. Bound the previous summary too: a rehydrated handoff can be huge. - _bounded_previous_summary = self._bound_summary_input( - self._previous_summary - ) + _bounded_previous_summary = self._bound_summary_input(self._previous_summary) prompt = f"""{_summarizer_preamble} You are updating a context compaction summary. A previous compaction produced the summary below. New conversation turns have occurred since then and need to be incorporated. @@ -4226,10 +4081,7 @@ This compaction should PRIORITISE preserving all information related to the focu """Reject user attribution when the source transcript has no user.""" if has_user_turn: return - match = re.search( - rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n(.*?)(?=\n##\s|\Z)", - summary, - ) + match = re.search(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n(.*?)(?=\n##\s|\Z)", summary) task_snapshot = match.group(1).strip() if match else "" # The "User asked:" scan can false-positive on quoted tool output; acceptable, since # the RuntimeError only costs one retry on the existing fallback path. @@ -4289,9 +4141,7 @@ This compaction should PRIORITISE preserving all information related to the focu return not cls._is_blank_user_turn(message) @classmethod - def _blank_echo_indices_after( - cls, messages: List[Dict[str, Any]], user_idx: int - ) -> set[int]: + def _blank_echo_indices_after(cls, messages: List[Dict[str, Any]], user_idx: int) -> set[int]: """Return contiguous blank echoes after a user event; removable only if an assistant follows.""" indices: set[int] = set() if user_idx < 0: @@ -4305,10 +4155,7 @@ This compaction should PRIORITISE preserving all information related to the focu return indices if messages[idx].get("role") == "assistant" else set() @classmethod - def _derive_auto_focus_topic( - cls, - messages: List[Dict[str, Any]], - ) -> Optional[str]: + def _derive_auto_focus_topic(cls, messages: List[Dict[str, Any]]) -> Optional[str]: """Infer a compact focus hint from the most recent real user turns.""" candidates: list[str] = [] for idx in range(len(messages) - 1, -1, -1): @@ -4341,10 +4188,7 @@ This compaction should PRIORITISE preserving all information related to the focu return focus @classmethod - def _latest_user_task_snapshot( - cls, - messages: List[Dict[str, Any]], - ) -> Optional[str]: + def _latest_user_task_snapshot(cls, messages: List[Dict[str, Any]]) -> Optional[str]: """Return a deterministic task-snapshot line from the newest real user turn. The summarizer must not invent the active-task anchor from a prompt example or a @@ -4372,11 +4216,7 @@ This compaction should PRIORITISE preserving all information related to the focu return None @classmethod - def _ground_historical_task_snapshot( - cls, - summary: str, - messages: List[Dict[str, Any]], - ) -> str: + def _ground_historical_task_snapshot(cls, summary: str, messages: List[Dict[str, Any]]) -> str: """Force the task snapshot section to match a real user turn when possible.""" snapshot = cls._latest_user_task_snapshot(messages) if not snapshot: @@ -4387,9 +4227,7 @@ This compaction should PRIORITISE preserving all information related to the focu # this regex on the next compaction (deleting every following section). replacement = f"{HISTORICAL_TASK_HEADING}\n{snapshot}\n\n" if _HISTORICAL_TASK_SECTION_RE.search(body): - grounded = _HISTORICAL_TASK_SECTION_RE.sub( - lambda _m: replacement, body, count=1 - ) + grounded = _HISTORICAL_TASK_SECTION_RE.sub(lambda _m: replacement, body, count=1) return grounded.strip() return f"{replacement}{body}".strip() @@ -4409,10 +4247,7 @@ This compaction should PRIORITISE preserving all information related to the focu for idx in range(start, end): content = messages[idx].get("content") if cls._is_context_summary_message(messages[idx]): - summaries.append(( - idx, - cls._strip_summary_prefix(_content_text_for_contains(content)), - )) + summaries.append((idx, cls._strip_summary_prefix(_content_text_for_contains(content)))) return summaries @classmethod @@ -4429,10 +4264,7 @@ This compaction should PRIORITISE preserving all information related to the focu return None, "" @classmethod - def _strip_context_summary_handoff_message( - cls, - message: Dict[str, Any], - ) -> Optional[Dict[str, Any]]: + def _strip_context_summary_handoff_message(cls, message: Dict[str, Any]) -> Optional[Dict[str, Any]]: """Drop stale handoff data while preserving merged prior-tail content. Returns a copy for non-handoff rows, the unwrapped prior-tail content for merged @@ -4615,10 +4447,7 @@ This compaction should PRIORITISE preserving all information related to the focu idx += 1 return idx - def _restart_handoff_probe_bounds( - self, - messages: List[Dict[str, Any]], - ) -> tuple[int, int]: + def _restart_handoff_probe_bounds(self, messages: List[Dict[str, Any]]) -> tuple[int, int]: """Return the bounded transcript region that can indicate restart decay.""" if not messages or self.protect_first_n <= 0: return 0, 0 @@ -4630,10 +4459,7 @@ This compaction should PRIORITISE preserving all information related to the focu + _RESTART_HANDOFF_PROBE_EXTRA_MESSAGES, ) - def _effective_protect_first_n( - self, - messages: Optional[List[Dict[str, Any]]] = None, - ) -> int: + def _effective_protect_first_n(self, messages: Optional[List[Dict[str, Any]]] = None) -> int: """``protect_first_n`` decayed to 0 once the session has been compressed. Otherwise early turns fossilize across compactions. After a restart the decayed @@ -4643,9 +4469,7 @@ This compaction should PRIORITISE preserving all information related to the focu return 0 if messages and self.protect_first_n > 0: # Probe only the early resumed-handoff shape; summary-like tail content must not decay protection. - first_non_system, restart_probe_end = self._restart_handoff_probe_bounds( - messages - ) + first_non_system, restart_probe_end = self._restart_handoff_probe_bounds(messages) if any( self._is_context_summary_message(msg) for msg in messages[first_non_system:restart_probe_end] @@ -4681,25 +4505,18 @@ This compaction should PRIORITISE preserving all information related to the focu return idx - def _find_last_user_message_idx( - self, messages: List[Dict[str, Any]], head_end: int - ) -> int: + def _find_last_user_message_idx(self, messages: List[Dict[str, Any]], head_end: int) -> int: """Return the latest actionable user turn at or after *head_end*, or -1. Compaction handoffs and blank platform echoes never displace the real request. """ for i in range(len(messages) - 1, head_end - 1, -1): msg = messages[i] - if ( - self._is_actionable_user_turn(msg) - and not self._is_synthetic_compression_user_turn(msg) - ): + if self._is_actionable_user_turn(msg) and not self._is_synthetic_compression_user_turn(msg): return i return -1 - def _find_last_assistant_message_idx( - self, messages: List[Dict[str, Any]], head_end: int - ) -> int: + def _find_last_assistant_message_idx(self, messages: List[Dict[str, Any]], head_end: int) -> int: """Return the last text-bearing, non-summary assistant reply at or after *head_end*, or -1. Falls back to the last non-summary assistant of any kind when none has text. @@ -4813,10 +4630,7 @@ This compaction should PRIORITISE preserving all information related to the focu user_indices = [] for i in range(len(messages) - 1, head_end - 1, -1): msg = messages[i] - if ( - self._is_actionable_user_turn(msg) - and not self._is_synthetic_compression_user_turn(msg) - ): + if self._is_actionable_user_turn(msg) and not self._is_synthetic_compression_user_turn(msg): user_indices.append(i) if len(user_indices) == 0: @@ -4830,11 +4644,7 @@ This compaction should PRIORITISE preserving all information related to the focu cut_idx = target_idx return max(cut_idx, head_end + 1) - def _find_turn_pair_end( - self, - messages: List[Dict[str, Any]], - user_idx: int, - ) -> int: + def _find_turn_pair_end(self, messages: List[Dict[str, Any]], user_idx: int) -> int: """Return the index after the turn-pair (user -> assistant -> tools) at *user_idx*. Returns ``user_idx + 1`` when there is no reply yet. @@ -4886,10 +4696,7 @@ This compaction should PRIORITISE preserving all information related to the focu min_tail_floor = max(3, min(self.protect_last_n, _MAX_TAIL_MESSAGE_FLOOR)) # Keep >= 2 non-head messages summarizable so a tiny middle still saves messages. compressible_tail_cap = max(3, available_tail - 2) - min_tail = ( - min(min_tail_floor, compressible_tail_cap, available_tail) - if available_tail > 1 else 0 - ) + min_tail = min(min_tail_floor, compressible_tail_cap, available_tail) if available_tail > 1 else 0 soft_ceiling = int(token_budget * 1.5) # Only the newest assistant turn's thinking ships (#73624), except echo-back providers @@ -4988,11 +4795,7 @@ This compaction should PRIORITISE preserving all information related to the focu summary_idx = None summary_body = None tail_start = compress_end - summary_hits = self._find_context_summaries( - messages, - summary_search_start, - summary_search_end, - ) + summary_hits = self._find_context_summaries(messages, summary_search_start, summary_search_end) real_user_present = self._transcript_has_real_user_turn(messages) if summary_hits: summary_idx = summary_hits[-1][0] @@ -5002,9 +4805,7 @@ This compaction should PRIORITISE preserving all information related to the focu if summary_bodies: self._previous_summary = "\n\n".join(summary_bodies) # Zero-user provenance (#64650) rides on the newest handoff hit. - provenance = messages[summary_idx].get( - COMPRESSED_SUMMARY_HAS_USER_TURN_KEY - ) + provenance = messages[summary_idx].get(COMPRESSED_SUMMARY_HAS_USER_TURN_KEY) if real_user_present: self._summary_has_user_turn = True elif isinstance(provenance, bool): @@ -5012,18 +4813,14 @@ This compaction should PRIORITISE preserving all information related to the focu elif self._summary_has_user_turn is None: # Legacy handoffs lack provenance: assume a user turn unless the exact no-user # sentinel is present. - self._summary_has_user_turn = not ( - summary_body and _NO_USER_TASK_SENTINEL in summary_body - ) + self._summary_has_user_turn = not (summary_body and _NO_USER_TASK_SENTINEL in summary_body) summary_indices = {idx for idx, _ in summary_hits} # Summary rows are excluded from summarizer input, but a merged handoff carries genuine # prior-tail user content — unwrap it into the window (#47274). def _window_row(idx: int, msg: Dict[str, Any]): if idx not in summary_indices: return msg - stripped = self._strip_context_summary_handoff_message( - _fresh_compaction_message_copy(msg) - ) + stripped = self._strip_context_summary_handoff_message(_fresh_compaction_message_copy(msg)) return stripped # None for standalone handoffs → dropped pre_summary_turns = [ row for idx, msg in enumerate( @@ -5032,9 +4829,7 @@ This compaction should PRIORITISE preserving all information related to the focu ) if (row := _window_row(idx, msg)) is not None ] - turns_to_summarize = ( - pre_summary_turns + messages[summary_idx + 1:compress_end] - ) + turns_to_summarize = pre_summary_turns + messages[summary_idx + 1:compress_end] # The newest hit may itself be a merged handoff — recover its prior-tail content too. _newest_stripped = self._strip_context_summary_handoff_message( _fresh_compaction_message_copy(messages[summary_idx]) @@ -5083,9 +4878,7 @@ This compaction should PRIORITISE preserving all information related to the focu self._structural_no_op_backoff_until = 0.0 return telemetry - def _structural_no_op_result( - self, telemetry: Dict[str, Any], failure_class: str, reason: str, - ) -> None: + def _structural_no_op_result(self, telemetry: Dict[str, Any], failure_class: str, reason: str) -> None: """Nothing eligible to compress: transient backoff (#93022), never an ineffectiveness strike.""" telemetry["failure_class"] = failure_class self._last_compression_savings_pct = 0.0 @@ -5568,13 +5361,7 @@ def _handoff_only_content(content: Any) -> Any: # Ordinary merge: summary suffix starts in the delimiter part; later parts may carry live media # — never retain. for item in content: - text = ( - item - if isinstance(item, str) - else item.get("text") - if isinstance(item, dict) - else None - ) + text = item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None if not isinstance(text, str) or _MERGED_SUMMARY_DELIMITER not in text: continue suffix = text.split(_MERGED_SUMMARY_DELIMITER, 1)[1].lstrip() @@ -5592,13 +5379,7 @@ def _handoff_only_content(content: Any) -> Any: # Force-user-leading: keep parts through the end marker, truncated before the live ask. projected: list[Any] = [] for item in content: - text = ( - item - if isinstance(item, str) - else item.get("text") - if isinstance(item, dict) - else None - ) + text = item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None if isinstance(text, str) and _SUMMARY_END_MARKER in text: prefix = text.split(_SUMMARY_END_MARKER, 1)[0] + _SUMMARY_END_MARKER if isinstance(item, dict): @@ -5613,9 +5394,7 @@ def _handoff_only_content(content: Any) -> Any: return projected -def split_user_originated_turn( - message: Any, -) -> tuple[Optional[Dict[str, Any]], Optional[Dict[str, Any]]]: +def split_user_originated_turn(message: Any) -> tuple[Optional[Dict[str, Any]], Optional[Dict[str, Any]]]: """Split a user row into ``(handoff_only, live_view)``; either may be None; fresh dicts.""" if not isinstance(message, dict) or message.get("role") != "user": return None, None @@ -5740,15 +5519,10 @@ def _handoff_carries_live_user_content(message: Any) -> bool: """ if not isinstance(message, dict): return False - return ( - ContextCompressor._strip_context_summary_handoff_message(message) - is not None - ) + return ContextCompressor._strip_context_summary_handoff_message(message) is not None -def reference_handoff_would_drive_next_model_call( - messages: Optional[List[Dict[str, Any]]], -) -> bool: +def reference_handoff_would_drive_next_model_call(messages: Optional[List[Dict[str, Any]]]) -> bool: """Return True when the next model call would be driven only by a handoff (#80622). Mid tool-loop compression is allowed: trailing tool rows mean an in-flight exchange. @@ -5794,9 +5568,7 @@ def reference_handoff_would_drive_next_model_call( and not ContextCompressor._is_synthetic_compression_user_turn(message) ): return False - if is_compaction_summary_message(message) and _handoff_carries_live_user_content( - message - ): + if is_compaction_summary_message(message) and _handoff_carries_live_user_content(message): return False return True From 34f96ff29c3ac7d7a3730c1b9b51596e28774221 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:01:32 -0700 Subject: [PATCH 08/27] refactor(agent/context_compressor): reflow short multi-line call statements --- agent/context_compressor.py | 44 +++++++++++-------------------------- 1 file changed, 13 insertions(+), 31 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index f3c9a92258..587a7a6957 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -108,8 +108,7 @@ _SUMMARY_PERMANENT_QUOTA_MARKERS: tuple[str, ...] = ( _SUMMARY_MISSING_CREDENTIAL_MARKERS: tuple[str, ...] = ("no api key was found", "no api key found") _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS: tuple[str, ...] = ( - "session hygiene compression timed out", - "hygiene compression deferred: turn-hold budget expired", + "session hygiene compression timed out", "hygiene compression deferred: turn-hold budget expired", ) @@ -796,8 +795,7 @@ def _skill_pruned_marker(skill_name: str) -> str: # Anchored on the shared prefix so marker wording changes stay in sync. _SKILL_PRUNED_MARKER_RE = re.compile( - re.escape(SKILL_PRUNED_MARKER_PREFIX) - + r"[^\]]*?reload with skill_view\(name='([^']+)'\)" + re.escape(SKILL_PRUNED_MARKER_PREFIX) + r"[^\]]*?reload with skill_view\(name='([^']+)'\)", ) @@ -2000,8 +1998,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if getattr(self, "tail_mode", "lean") == "lean": # Lean mode: tail is a small clamped recency window; the summary carries continuity. self._tail_token_budget = max( - LEAN_TAIL_FLOOR_TOKENS, - min(LEAN_TAIL_CAP_TOKENS, int(self.context_length * 0.025)), + LEAN_TAIL_FLOOR_TOKENS, min(LEAN_TAIL_CAP_TOKENS, int(self.context_length * 0.025)), ) else: self._tail_token_budget = int(self.threshold_tokens * self.summary_target_ratio) @@ -2372,8 +2369,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return # A store without the recorder or a failed write both leave the durable row unauthoritative. self._cooldown_persist_failed = not self._durable_write( - "record_compression_failure_cooldown", "compression failure cooldown", - cooldown_until, error, + "record_compression_failure_cooldown", "compression failure cooldown", cooldown_until, error, ) def record_timeout_failure(self, error: str, failure_kind: str = "timeout") -> None: @@ -2390,8 +2386,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # Fence check BEFORE cooldown-clear: a late cancelled worker must not undo the host's timeout cooldown. if self._compression_cancelled(): logger.info( - "Skipping compression cooldown clear: host already " - "cancelled this compression attempt" + "Skipping compression cooldown clear: host already cancelled this compression attempt", ) return self._summary_failure_cooldown_until = 0.0 @@ -2819,8 +2814,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if _cooldown_remaining > 0 and not ignore_cooldown: if not self.quiet_mode: logger.debug( - "Compression deferred — summary LLM in cooldown for %.0fs more", - _cooldown_remaining, + "Compression deferred — summary LLM in cooldown for %.0fs more", _cooldown_remaining, ) return True # Structural no-op backoff is transient (in-memory, no strikes); auto-compaction resumes when it lapses. @@ -3415,8 +3409,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb summary += _redact_compaction_text(_build_verbatim_user_section(turns_to_summarize)) if _LEAN_RECOVERY_HEADING not in summary: summary += _build_recovery_footer( - getattr(self, "_session_id", "") or "", - len(turns_to_summarize), + getattr(self, "_session_id", "") or "", len(turns_to_summarize), ) return summary @@ -3608,11 +3601,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb if has_user_turn is None: has_user_turn = self._transcript_has_real_user_turn(turns_to_summarize) prompt = self._build_summary_prompt( - content_to_summarize, - summary_budget, - focus_topic, - memory_context, - has_user_turn, + content_to_summarize, summary_budget, focus_topic, memory_context, has_user_turn, ) try: @@ -3898,8 +3887,7 @@ This compaction should PRIORITISE preserving all information related to the focu # RuntimeErrors are transient and must get the main-model retry below first. if isinstance(e, RuntimeError) and "no llm provider configured" in str(e).lower(): self._record_compression_failure_cooldown( - _SUMMARY_FAILURE_COOLDOWN_SECONDS, - "no auxiliary LLM provider configured", + _SUMMARY_FAILURE_COOLDOWN_SECONDS, "no auxiliary LLM provider configured", ) self._last_summary_error = "no auxiliary LLM provider configured" logger.warning("Context compression: no provider available for " @@ -4453,10 +4441,7 @@ This compaction should PRIORITISE preserving all information related to the focu return 0, 0 first_non_system = 1 if messages[0].get("role") == "system" else 0 return first_non_system, min( - len(messages), - first_non_system - + self.protect_first_n - + _RESTART_HANDOFF_PROBE_EXTRA_MESSAGES, + len(messages), first_non_system + self.protect_first_n + _RESTART_HANDOFF_PROBE_EXTRA_MESSAGES, ) def _effective_protect_first_n(self, messages: Optional[List[Dict[str, Any]]] = None) -> int: @@ -4714,8 +4699,7 @@ This compaction should PRIORITISE preserving all information related to the focu cut = n # start from beyond the end for i in range(n - 1, head_end - 1, -1): msg_tokens = _estimate_msg_budget_tokens( - messages[i], - charge_stale_thinking=(_charge_all_thinking or i == _newest_asst_idx), + messages[i], charge_stale_thinking=(_charge_all_thinking or i == _newest_asst_idx), ) if accumulated + msg_tokens > ceiling and (n - i) >= min_tail: return (i if cut_at_break else cut), accumulated @@ -5162,8 +5146,7 @@ This compaction should PRIORITISE preserving all information related to the focu _pruned_replay = _prune_stale_reasoning_replay(compressed) if _pruned_replay and not self.quiet_mode: logger.info( - "Pruned stale replay items from %d assistant message(s) during compaction", - _pruned_replay, + "Pruned stale replay items from %d assistant message(s) during compaction", _pruned_replay, ) self._last_compression_made_progress = True @@ -5216,8 +5199,7 @@ This compaction should PRIORITISE preserving all information related to the focu # Phase 1: Prune old tool results (cheap, no LLM call) messages, pruned_count = self._prune_old_tool_results( - messages, protect_tail_count=self.protect_last_n, - protect_tail_tokens=self.tail_token_budget, + messages, protect_tail_count=self.protect_last_n, protect_tail_tokens=self.tail_token_budget, ) if pruned_count and not self.quiet_mode: logger.info("Pre-compression: pruned %d old tool result(s)", pruned_count) From 9a89ce554833808d518c442766f7faf5318dd186 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:08:33 -0700 Subject: [PATCH 09/27] refactor(agent/context_compressor): pack short literal collections and call arguments (AST-identical) --- agent/context_compressor.py | 133 +++++++++++------------------------- 1 file changed, 39 insertions(+), 94 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 587a7a6957..b7f648cd32 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -96,13 +96,8 @@ def _pinned_summary_call_kwargs() -> Dict[str, Any]: _SUMMARY_PERMANENT_QUOTA_MARKERS: tuple[str, ...] = ( - "insufficient_quota", - "quota exceeded", - "quota_exceeded", - "out of funds", - "out of credits", - "out of credit", - "out of extra usage", + "insufficient_quota", "quota exceeded", "quota_exceeded", "out of funds", "out of credits", + "out of credit", "out of extra usage", ) _SUMMARY_MISSING_CREDENTIAL_MARKERS: tuple[str, ...] = ("no api key was found", "no api key found") @@ -752,10 +747,8 @@ _PRUNE_MIN_CHARS = 200 # Sentinel ``user_response`` values from timeout / no-user clarify callbacks; # must never be quoted as a user answer. _CLARIFY_NON_RESPONSE_PREFIXES = ( - "The user did not provide a response", - "[user did not respond", - "[clarify prompt could not be delivered", - "[oneshot mode:", + "The user did not provide a response", "[user did not respond", + "[clarify prompt could not be delivered", "[oneshot mode:", ) @@ -905,10 +898,8 @@ def _synthetic_user_row(content: str) -> bool: return True stripped = content.lstrip() _synthetic_prefixes = ( - "[System:", "[CONTEXT", "[PRIOR CONTEXT", "[IMPORTANT: Background", - "[Your active task list", "[Planning state preserved", - "[ASYNC DELEGATION", "[OUT-OF-BAND", - "Cronjob Response:", + "[System:", "[CONTEXT", "[PRIOR CONTEXT", "[IMPORTANT: Background", "[Your active task list", + "[Planning state preserved", "[ASYNC DELEGATION", "[OUT-OF-BAND", "Cronjob Response:", ) return stripped.startswith(_synthetic_prefixes) @@ -1831,38 +1822,19 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): session_id = session_id or seed.get("session_id") trigger_source = trigger_source or seed.get("trigger_source") telemetry: Dict[str, Any] = { - "event": "compression_attempt", - "attempt_id": attempt_id or uuid.uuid4().hex, - "session_id": session_id or "", - "trigger_source": trigger_source or "unknown", - "main_provider": self.provider or "", - "main_model": self.model or "", + "event": "compression_attempt", "attempt_id": attempt_id or uuid.uuid4().hex, + "session_id": session_id or "", "trigger_source": trigger_source or "unknown", + "main_provider": self.provider or "", "main_model": self.model or "", "main_context_limit": _safe_int(self.context_length), "current_estimated_tokens": _safe_int(current_tokens), - "effective_threshold": _safe_int(self.threshold_tokens), - "protected_head_tokens": None, - "protected_tail_tokens": None, - "middle_window_tokens": None, - "prellm_skip_count": 0, - "aux_prompt_tokens": None, - "aux_output_reservation": None, - "aux_provider": "", - "aux_model": "", - "effective_aux_context": None, - "fit_margin": None, - "chunking": False, - "chunk_count": 0, - "total_duration_ms": None, - "aux_call_duration_ms": None, - "queue_wait_ms": None, - "prompt_build_ms": None, - "time_to_first_progress_ms": None, - "summary_generation_ms": None, - "commit_ms": None, - "fallback_used": False, - "commit_status": "unknown", - "split_status": "unknown", - "failure_class": None, + "effective_threshold": _safe_int(self.threshold_tokens), "protected_head_tokens": None, + "protected_tail_tokens": None, "middle_window_tokens": None, "prellm_skip_count": 0, + "aux_prompt_tokens": None, "aux_output_reservation": None, "aux_provider": "", "aux_model": "", + "effective_aux_context": None, "fit_margin": None, "chunking": False, "chunk_count": 0, + "total_duration_ms": None, "aux_call_duration_ms": None, "queue_wait_ms": None, + "prompt_build_ms": None, "time_to_first_progress_ms": None, "summary_generation_ms": None, + "commit_ms": None, "fallback_used": False, "commit_status": "unknown", + "split_status": "unknown", "failure_class": None, } self._active_compression_telemetry = telemetry self._last_compression_telemetry = telemetry @@ -1945,11 +1917,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): """Resolve and cache the model's context length on first access.""" if self._resolved_context_length is None: self._resolved_context_length = get_model_context_length( - self.model, - base_url=self.base_url, - api_key=self.api_key, - config_context_length=self._config_context_length, - provider=self.provider, + self.model, base_url=self.base_url, api_key=self.api_key, + config_context_length=self._config_context_length, provider=self.provider, ) # Raise-only small-context floor; must run after context_length resolves and before threshold_tokens derives. self.threshold_percent = self._effective_threshold_percent( @@ -2180,9 +2149,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _load_proactive_prune_rearm_tokens(self) -> None: """Restore the cache-boundary runway for a resumed durable session.""" self._load_durable( - "_proactive_prune_rearm_tokens", - "get_session_model_config_value", "proactive prune runway", int, 0, - PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY, 0, + "_proactive_prune_rearm_tokens", "get_session_model_config_value", "proactive prune runway", + int, 0, PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY, 0, ) def _clear_durable_proactive_prune_rearm(self) -> None: @@ -2241,9 +2209,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._structural_no_op_backoff_until = time.monotonic() + self._STRUCTURAL_NO_OP_BACKOFF_SECONDS if not self.quiet_mode: logger.warning( - "Compression skipped (%s): retrying in %.0fs " - "(structural no-op backoff)", - reason, + "Compression skipped (%s): retrying in %.0fs (structural no-op backoff)", reason, self._STRUCTURAL_NO_OP_BACKOFF_SECONDS, ) @@ -2274,8 +2240,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # A pre-LLM feasibility skip is not a summary-quality verdict: it must neither extend nor reset the streak. if not self.quiet_mode: logger.info( - "Compaction completed via pre-LLM feasibility skip; " - "fallback_compression_streak unchanged (%d)", + "Compaction completed via pre-LLM feasibility skip; fallback_compression_streak unchanged (%d)", self._fallback_compression_streak, ) return @@ -2283,8 +2248,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._fallback_compression_streak += 1 if not self.quiet_mode: logger.warning( - "Compaction completed with a deterministic fallback summary. " - "fallback_compression_streak=%d", + "Compaction completed with a deterministic fallback summary. fallback_compression_streak=%d", self._fallback_compression_streak, ) elif self._fallback_compression_streak: @@ -2822,9 +2786,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if _structural_remaining > 0: if not self.quiet_mode: logger.debug( - "Compression deferred — structural no-op backoff for " - "%.0fs more", - _structural_remaining, + "Compression deferred — structural no-op backoff for %.0fs more", _structural_remaining, ) return True # Anti-thrash back-off must not be permanent: after _ANTI_THRASH_RECOVERY_SECONDS blocked, allow ONE @@ -3096,9 +3058,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): if session_db and session_id and not callable(getattr(session_db, "archive_and_compact", None)): return messages, 0 pruned_msgs, pruned_count = self._prune_old_tool_results( - messages, - protect_tail_count=self.protect_last_n, - protect_tail_tokens=None, + messages, protect_tail_count=self.protect_last_n, protect_tail_tokens=None, min_prune_chars=self.proactive_prune_min_result_chars, ) if not pruned_count: @@ -3123,8 +3083,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): ) except Exception as exc: logger.warning( - "Proactive tool-result prune DB commit failed; keeping the " - "original transcript: %s", + "Proactive tool-result prune DB commit failed; keeping the original transcript: %s", exc, ) return messages, 0 @@ -3476,8 +3435,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb """ self._summary_model_fallen_back = True logger.warning( - "Summary model '%s' %s (%s). " - "Falling back to main model '%s' for compression.", + "Summary model '%s' %s (%s). Falling back to main model '%s' for compression.", self.summary_model, reason, e, self.model, ) self._last_aux_model_failure_error = _short_error_text(e) @@ -3953,9 +3911,7 @@ This compaction should PRIORITISE preserving all information related to the focu elif _is_empty_content: self._last_summary_empty_content_failure = True logger.warning( - "Failed to generate context summary: %s. " - "Further summary attempts paused for %d seconds.", - e, + "Failed to generate context summary: %s. Further summary attempts paused for %d seconds.", e, _transient_cooldown, ) return None @@ -4655,10 +4611,8 @@ This compaction should PRIORITISE preserving all information related to the focu from agent.message_sanitization import stale_thinking_reaches_wire return stale_thinking_reaches_wire( - getattr(self, "api_mode", "") or "", - getattr(self, "provider", "") or "", - getattr(self, "model", "") or "", - getattr(self, "base_url", "") or "", + getattr(self, "api_mode", "") or "", getattr(self, "provider", "") or "", + getattr(self, "model", "") or "", getattr(self, "base_url", "") or "", ) except Exception: return False @@ -4834,9 +4788,7 @@ This compaction should PRIORITISE preserving all information related to the focu else: self._summary_has_user_turn = real_user_present return _HandoffScan( - turns_to_summarize=turns_to_summarize, - summary_indices=summary_indices, - tail_start=tail_start, + turns_to_summarize=turns_to_summarize, summary_indices=summary_indices, tail_start=tail_start, previous_summary_before=_previous_summary_before_scan, has_user_turn_before=_summary_has_user_turn_before_scan, ) @@ -5105,8 +5057,7 @@ This compaction should PRIORITISE preserving all information related to the focu suffix = "\n\n" + _MERGED_SUMMARY_DELIMITER + "\n\n" + summary + "\n\n" + _SUMMARY_END_MARKER msg["content"] = _append_text_to_content( _append_text_to_content(old_content, suffix, prepend=False), - _MERGED_PRIOR_CONTEXT_HEADER + "\n", - prepend=True, + _MERGED_PRIOR_CONTEXT_HEADER + "\n", prepend=True, ) # Frontends use this to detect a summary-prefixed message. msg[COMPRESSED_SUMMARY_METADATA_KEY] = True @@ -5210,8 +5161,7 @@ This compaction should PRIORITISE preserving all information related to the focu compress_start, compress_end = self._compress_window(messages) if compress_start >= compress_end: self._record_compression_regions( - head_messages=messages[:compress_start], - middle_messages=[], + head_messages=messages[:compress_start], middle_messages=[], tail_messages=messages[compress_end:], ) self._structural_no_op_result( @@ -5230,8 +5180,7 @@ This compaction should PRIORITISE preserving all information related to the focu turns_to_summarize = scan.turns_to_summarize self._record_compression_regions( - head_messages=messages[:compress_start], - middle_messages=turns_to_summarize, + head_messages=messages[:compress_start], middle_messages=turns_to_summarize, tail_messages=messages[compress_end:], ) telemetry["chunk_count"] = 1 if turns_to_summarize else 0 @@ -5259,10 +5208,8 @@ This compaction should PRIORITISE preserving all information related to the focu # Focus-topic derivation scans user turns; only pay when a summary is generated. try: summary = self._generate_summary( - turns_to_summarize, - focus_topic=focus_topic or self._derive_auto_focus_topic(messages), - memory_context=memory_context, - bypass_cooldown=bypass_cooldown, + turns_to_summarize, focus_topic=focus_topic or self._derive_auto_focus_topic(messages), + memory_context=memory_context, bypass_cooldown=bypass_cooldown, ) except AuxiliaryExplicitCancellation: # Cancellation is a true no-op: restore the self-heal scan's mutation before the @@ -5385,10 +5332,8 @@ def split_user_originated_turn(message: Any) -> tuple[Optional[Dict[str, Any]], handoff: Optional[Dict[str, Any]] = None if is_summary: handoff = { - "role": "user", - "content": _handoff_only_content(message.get("content")), - COMPRESSED_SUMMARY_METADATA_KEY: True, - "display_kind": "hidden", + "role": "user", "content": _handoff_only_content(message.get("content")), + COMPRESSED_SUMMARY_METADATA_KEY: True, "display_kind": "hidden", } if COMPRESSED_SUMMARY_HAS_USER_TURN_KEY in message: handoff[COMPRESSED_SUMMARY_HAS_USER_TURN_KEY] = bool( From e468577b1376982a7c12fb404ffd4d00d8e68745 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:13:23 -0700 Subject: [PATCH 10/27] refactor(agent/context_compressor): shared content-part text helpers for handoff projection/unwrapping --- agent/context_compressor.py | 137 +++++++++++++----------------------- 1 file changed, 50 insertions(+), 87 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index b7f648cd32..005ffeecce 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1346,6 +1346,16 @@ def _last_assistant_index(messages: "List[Dict[str, Any]]") -> int: return -1 +def _part_text(item: Any) -> Optional[str]: + """Text of a content part: the string itself, a dict's ``text``, else None.""" + return item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None + + +def _with_part_text(item: Any, text: str) -> Any: + """Copy of a content part carrying ``text`` (string parts become the text itself).""" + return {**item, "text": text} if isinstance(item, dict) else text + + def _content_text_for_contains(content: Any) -> str: """Return a best-effort text view of message content (for substring checks only).""" if content is None: @@ -4218,16 +4228,11 @@ This compaction should PRIORITISE preserving all information related to the focu return message content = message.get("content") - is_summary = ( - cls._is_context_summary_content(content) - or cls._has_compressed_summary_metadata(message) - ) - if not is_summary: + if not cls._is_context_summary_message(message): return message.copy() def _unwrapped(new_content: Any) -> Dict[str, Any]: - unwrapped = message.copy() - unwrapped["content"] = new_content + unwrapped = {**message, "content": new_content} unwrapped.pop(COMPRESSED_SUMMARY_METADATA_KEY, None) return unwrapped @@ -4249,63 +4254,40 @@ This compaction should PRIORITISE preserving all information related to the focu prior_blocks: list[Any] = [] found_delimiter = False for item in content: - if isinstance(item, str): - if _MERGED_SUMMARY_DELIMITER in item: - before = item.split(_MERGED_SUMMARY_DELIMITER, 1)[0] - if before.strip(): - prior_blocks.append(before) - found_delimiter = True - break - prior_blocks.append(item) - continue - if isinstance(item, dict): - text = item.get("text") - if isinstance(text, str) and _MERGED_SUMMARY_DELIMITER in text: - before = text.split(_MERGED_SUMMARY_DELIMITER, 1)[0] - if before.strip(): - prior_blocks.append({**item, "text": before}) - found_delimiter = True - break - prior_blocks.append(item.copy()) - continue - prior_blocks.append(item) + text = _part_text(item) + if isinstance(text, str) and _MERGED_SUMMARY_DELIMITER in text: + before = text.split(_MERGED_SUMMARY_DELIMITER, 1)[0] + if before.strip(): + prior_blocks.append(_with_part_text(item, before)) + found_delimiter = True + break + prior_blocks.append(item.copy() if isinstance(item, dict) else item) if not found_delimiter: - legacy_blocks: list[Any] = [] - found_marker = False + # Legacy end-marker form: live content follows the marker inside/after one part. for index, item in enumerate(content): - text = item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None + text = _part_text(item) if not isinstance(text, str) or _SUMMARY_END_MARKER not in text: continue remainder = text.split(_SUMMARY_END_MARKER, 1)[1].lstrip() - if remainder: - legacy_blocks.append({**item, "text": remainder} if isinstance(item, dict) else remainder) - for later in content[index + 1:]: - legacy_blocks.append(later.copy() if isinstance(later, dict) else later) - found_marker = True - break - if found_marker and legacy_blocks: - return _unwrapped(legacy_blocks) + legacy_blocks = [_with_part_text(item, remainder)] if remainder else [] + legacy_blocks += [later.copy() if isinstance(later, dict) else later for later in content[index + 1:]] + return _unwrapped(legacy_blocks) if legacy_blocks else None + return None - if found_delimiter: - # Strip the PRIOR CONTEXT header from the first block that carries it. - for index, item in enumerate(prior_blocks): - text = item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None - if not isinstance(text, str): - continue - if not text.lstrip().startswith(_MERGED_PRIOR_CONTEXT_HEADER): - continue - leading = text.lstrip()[len(_MERGED_PRIOR_CONTEXT_HEADER):].lstrip() - if not leading: - prior_blocks.pop(index) - elif isinstance(item, str): - prior_blocks[index] = leading - else: - prior_blocks[index] = {**item, "text": leading} - break - - if prior_blocks: - return _unwrapped(prior_blocks) + # Strip the PRIOR CONTEXT header from the first block that carries it. + for index, item in enumerate(prior_blocks): + text = _part_text(item) + if not isinstance(text, str) or not text.lstrip().startswith(_MERGED_PRIOR_CONTEXT_HEADER): + continue + leading = text.lstrip()[len(_MERGED_PRIOR_CONTEXT_HEADER):].lstrip() + if leading: + prior_blocks[index] = _with_part_text(item, leading) + else: + prior_blocks.pop(index) + break + if prior_blocks: + return _unwrapped(prior_blocks) return None @@ -5272,51 +5254,32 @@ SUMMARY_CARRIER_DURABLE_DISPLAY_METADATA_KEYS = ("reactions",) def _handoff_only_content(content: Any) -> Any: """Project summary-bearing content to the synthetic handoff alone; never keeps live media.""" + def _through_end_marker(text: str) -> str: + marker_idx = text.find(_SUMMARY_END_MARKER) + return text[: marker_idx + len(_SUMMARY_END_MARKER)] if marker_idx >= 0 else text + if isinstance(content, str): if _MERGED_SUMMARY_DELIMITER in content: - suffix = content.split(_MERGED_SUMMARY_DELIMITER, 1)[1].lstrip() - marker_idx = suffix.find(_SUMMARY_END_MARKER) - if marker_idx >= 0: - return suffix[: marker_idx + len(_SUMMARY_END_MARKER)] - return suffix - marker_idx = content.find(_SUMMARY_END_MARKER) - if marker_idx >= 0: - return content[: marker_idx + len(_SUMMARY_END_MARKER)] - return content - + content = content.split(_MERGED_SUMMARY_DELIMITER, 1)[1].lstrip() + return _through_end_marker(content) if not isinstance(content, list): return content # Ordinary merge: summary suffix starts in the delimiter part; later parts may carry live media # — never retain. for item in content: - text = item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None + text = _part_text(item) if not isinstance(text, str) or _MERGED_SUMMARY_DELIMITER not in text: continue - suffix = text.split(_MERGED_SUMMARY_DELIMITER, 1)[1].lstrip() - marker_idx = suffix.find(_SUMMARY_END_MARKER) - if marker_idx >= 0: - suffix = suffix[: marker_idx + len(_SUMMARY_END_MARKER)] - if not suffix: - return [] - if isinstance(item, dict): - copied = item.copy() - copied["text"] = suffix - return [copied] - return [suffix] + suffix = _through_end_marker(text.split(_MERGED_SUMMARY_DELIMITER, 1)[1].lstrip()) + return [_with_part_text(item, suffix)] if suffix else [] # Force-user-leading: keep parts through the end marker, truncated before the live ask. projected: list[Any] = [] for item in content: - text = item if isinstance(item, str) else item.get("text") if isinstance(item, dict) else None + text = _part_text(item) if isinstance(text, str) and _SUMMARY_END_MARKER in text: - prefix = text.split(_SUMMARY_END_MARKER, 1)[0] + _SUMMARY_END_MARKER - if isinstance(item, dict): - copied = item.copy() - copied["text"] = prefix - projected.append(copied) - else: - projected.append(prefix) + projected.append(_with_part_text(item, text.split(_SUMMARY_END_MARKER, 1)[0] + _SUMMARY_END_MARKER)) return projected if isinstance(text, str): projected.append(item.copy() if isinstance(item, dict) else item) From a7e616ee5f634d2e4b7036d9c0fe069019dd42ec Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:17:16 -0700 Subject: [PATCH 11/27] refactor(agent/context_compressor): flatten _scan_window_handoffs around an early return --- agent/context_compressor.py | 112 ++++++++++++++++-------------------- 1 file changed, 49 insertions(+), 63 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 005ffeecce..f4cdd7ea8b 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -4703,77 +4703,63 @@ This compaction should PRIORITISE preserving all information related to the focu prior-tail content) and ``tail_start`` advances past a handoff beyond the window. The pre-scan state is captured so an aborted attempt can roll the mutation back (#57835). """ - # Snapshot so an aborted attempt can roll back the self-heal scan's mutation of - # _previous_summary (#57835). - _previous_summary_before_scan = self._previous_summary - _summary_has_user_turn_before_scan = getattr(self, "_summary_has_user_turn", None) + scan = _HandoffScan( + turns_to_summarize=turns_to_summarize, summary_indices=set(), tail_start=compress_end, + # Snapshot so an aborted attempt can roll back the self-heal mutation (#57835). + previous_summary_before=self._previous_summary, + has_user_turn_before=getattr(self, "_summary_has_user_turn", None), + ) # Always scan the full transcript for handoffs: a narrow scan could hide a same-session # handoff and wrongly trigger the cross-session discard (#57835, #83248). summary_search_start = 1 if messages and messages[0].get("role") == "system" else 0 - summary_search_end = len(messages) - summary_indices: set[int] = set() - summary_idx = None - summary_body = None - tail_start = compress_end - summary_hits = self._find_context_summaries(messages, summary_search_start, summary_search_end) + summary_hits = self._find_context_summaries(messages, summary_search_start, len(messages)) real_user_present = self._transcript_has_real_user_turn(messages) - if summary_hits: - summary_idx = summary_hits[-1][0] - summary_body = summary_hits[-1][1] - if not self._previous_summary: - summary_bodies = [body for _, body in summary_hits if body] - if summary_bodies: - self._previous_summary = "\n\n".join(summary_bodies) - # Zero-user provenance (#64650) rides on the newest handoff hit. - provenance = messages[summary_idx].get(COMPRESSED_SUMMARY_HAS_USER_TURN_KEY) - if real_user_present: - self._summary_has_user_turn = True - elif isinstance(provenance, bool): - self._summary_has_user_turn = provenance - elif self._summary_has_user_turn is None: - # Legacy handoffs lack provenance: assume a user turn unless the exact no-user - # sentinel is present. - self._summary_has_user_turn = not (summary_body and _NO_USER_TASK_SENTINEL in summary_body) - summary_indices = {idx for idx, _ in summary_hits} - # Summary rows are excluded from summarizer input, but a merged handoff carries genuine - # prior-tail user content — unwrap it into the window (#47274). - def _window_row(idx: int, msg: Dict[str, Any]): - if idx not in summary_indices: - return msg - stripped = self._strip_context_summary_handoff_message(_fresh_compaction_message_copy(msg)) - return stripped # None for standalone handoffs → dropped - pre_summary_turns = [ - row for idx, msg in enumerate( - messages[compress_start:summary_idx], - start=compress_start, - ) - if (row := _window_row(idx, msg)) is not None - ] - turns_to_summarize = pre_summary_turns + messages[summary_idx + 1:compress_end] - # The newest hit may itself be a merged handoff — recover its prior-tail content too. - _newest_stripped = self._strip_context_summary_handoff_message( - _fresh_compaction_message_copy(messages[summary_idx]) - ) - if _newest_stripped is not None: - turns_to_summarize = ( - pre_summary_turns - + [_newest_stripped] - + messages[summary_idx + 1:compress_end] - ) - if summary_idx >= compress_end: - tail_start = summary_idx + 1 - elif self._previous_summary: + if not summary_hits: # No handoff anywhere but _previous_summary is set: it came from another session — # discard. Never decide this from a compress_end-bounded miss (#83248). - self._previous_summary = None + if self._previous_summary: + self._previous_summary = None self._summary_has_user_turn = real_user_present - else: - self._summary_has_user_turn = real_user_present - return _HandoffScan( - turns_to_summarize=turns_to_summarize, summary_indices=summary_indices, tail_start=tail_start, - previous_summary_before=_previous_summary_before_scan, - has_user_turn_before=_summary_has_user_turn_before_scan, + return scan + + summary_idx, summary_body = summary_hits[-1] + if not self._previous_summary: + summary_bodies = [body for _, body in summary_hits if body] + if summary_bodies: + self._previous_summary = "\n\n".join(summary_bodies) + # Zero-user provenance (#64650) rides on the newest handoff hit. + provenance = messages[summary_idx].get(COMPRESSED_SUMMARY_HAS_USER_TURN_KEY) + if real_user_present: + self._summary_has_user_turn = True + elif isinstance(provenance, bool): + self._summary_has_user_turn = provenance + elif self._summary_has_user_turn is None: + # Legacy handoffs lack provenance: assume a user turn unless the exact no-user + # sentinel is present. + self._summary_has_user_turn = not (summary_body and _NO_USER_TASK_SENTINEL in summary_body) + scan.summary_indices = {idx for idx, _ in summary_hits} + + # Summary rows are excluded from summarizer input, but a merged handoff carries genuine + # prior-tail user content — unwrap it into the window (#47274); standalone ones drop (None). + # The newest hit (summary_idx) may itself be a merged handoff — recover its prior tail too. + def _window_row(idx: int, msg: Dict[str, Any]): + if idx not in scan.summary_indices: + return msg + return self._strip_context_summary_handoff_message(_fresh_compaction_message_copy(msg)) + + pre_summary_turns = [ + row for idx, msg in enumerate(messages[compress_start:summary_idx], start=compress_start) + if (row := _window_row(idx, msg)) is not None + ] + newest_stripped = _window_row(summary_idx, messages[summary_idx]) + scan.turns_to_summarize = ( + pre_summary_turns + + ([newest_stripped] if newest_stripped is not None else []) + + messages[summary_idx + 1:compress_end] ) + if summary_idx >= compress_end: + scan.tail_start = summary_idx + 1 + return scan def _begin_compress_attempt(self, current_tokens: Optional[int], force: bool) -> Dict[str, Any]: """Reset per-call result state (callers read it after compress()) and open telemetry.""" From 5f716407b9f4acfcd9257eb8808d53438954e797 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:21:58 -0700 Subject: [PATCH 12/27] refactor(agent/context_compressor): move per-section summarizer instructions into a keyed table (text byte-identical) --- agent/context_compressor.py | 145 +++++++++++++++++++----------------- 1 file changed, 75 insertions(+), 70 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index f4cdd7ea8b..7ecf987197 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1797,6 +1797,75 @@ def resolve_model_threshold(model: str, model_thresholds: dict[str, float] | Non return default +# Per-section summarizer instructions, keyed by "the transcript has a real user turn". Wording +# is deliberately plain: Azure/OpenAI content filters have flagged stronger "injection" / +# "do not respond" framing. Prompt text is byte-pinned — restructure code around it only. +_SECTION_INSTRUCTIONS: Dict[bool, Dict[str, str]] = { + True: { + "language": ( + "Write the summary in the same language the user was using in the " + "conversation — do not translate or switch to English. " + ), + "historical_task": """[THE SINGLE MOST IMPORTANT FIELD. Capture the user's most recent unfulfilled +input verbatim — the exact words they used. This includes: +- Explicit task assignments ("") +- Questions awaiting an answer ("") +- Decisions awaiting input ("