From 6e5413844e77ffdf0b8a300b946bdbac934a72f6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 26 Aug 2026 06:43:58 -0700 Subject: [PATCH] =?UTF-8?q?feat(compression):=20lean=20tail=20retention=20?= =?UTF-8?q?is=20the=20default=20=E2=80=94=20compaction=20keeps=2010-25K=20?= =?UTF-8?q?verbatim,=20not=20100-240K?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The legacy tail budget scales as threshold×target_ratio, which was designed around 128K windows at a 50% trigger (~13K tail). On modern big-window models with raised thresholds it silently hoards: a 1M-window session at threshold 0.85 keeps a 170K-token verbatim tail (255K soft ceiling) out of EVERY compaction, so a 540K manual /compress lands at ~290K and every subsequent turn re-ships the hoard. Nobody chooses this; it is an artifact of the formula outside its design envelope. Lean mode (#87326, compaction-v2) was built for exactly this and its recall was validated in the before/after eval (evals/compaction/results/): clamped 2.5%-of-window tail (10K floor / 25K cap), continuity carried by the upgraded summary (digests, anchor index, verbatim user messages, session_search recovery pointers). This flips the DEFAULT to lean; explicit 'tail_mode: legacy' in config keeps the old behavior exactly. Also fixes a latent bug the flip exposed: update_model() re-assigned the LEGACY formula directly when recomputing budgets, silently reverting a lean compressor to the hoard on every mid-session model switch. The recompute now routes through the mode-aware tail_token_budget property (regression test included). Surfaces: context_compressor.py defaults + getattr fallbacks, agent_init parse default, DEFAULT_CONFIG, gateway _CACHE_BUSTING_CONFIG_KEYS gains compression.tail_mode (mode changes now evict cached gateway agents like target_ratio changes do), user + developer docs. Tests: 3 new default contracts, legacy tests pinned explicitly, feasibility-skip scenario pinned to legacy (under lean its payloads correctly become compressible). E2E counterfactual (real imports, 1M window @ 0.85): main default: legacy, tail 170,000 (ceiling 255,000) head default: lean, tail 25,000 (ceiling 37,500) head legacy: 170,000 (opt-out intact) update_model to 400K: 10,000 (lean preserved across switch) --- agent/agent_init.py | 14 +-- agent/context_compressor.py | 20 +++-- gateway/run.py | 1 + hermes_cli/config_defaults.py | 8 +- ...t_compression_small_ctx_threshold_floor.py | 14 +++ tests/agent/test_context_compressor.py | 85 ++++++++++++++++++- .../context-compression-and-caching.md | 4 +- website/docs/user-guide/configuration.md | 2 +- 8 files changed, 125 insertions(+), 23 deletions(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index 8763af043b..7ff0d8a872 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -2146,11 +2146,15 @@ def init_agent( compression_enabled = str(_compression_cfg.get("enabled", True)).lower() in {"true", "1", "yes"} compression_target_ratio = float(_compression_cfg.get("target_ratio", 0.20)) compression_protect_last = int(_compression_cfg.get("protect_last_n", 20)) - # Tail retention mode (compression.tail_mode). "legacy" (default) keeps - # the 0.20*window verbatim tail; "lean" switches to the clamped - # 2.5%/10K-25K tail with recovery-pointer machinery (#87326). Unknown - # values fall back to legacy inside the compressor. - compression_tail_mode = str(_compression_cfg.get("tail_mode", "legacy")).strip().lower() + # Tail retention mode (compression.tail_mode). "lean" (default) keeps a + # clamped 2.5%/10K-25K verbatim tail with recovery-pointer machinery — + # continuity rides the upgraded summary (digests, anchor index, verbatim + # user messages, session_search pointers; recall-eval'd, see + # evals/compaction/results/). "legacy" restores the pre-#87326 + # 0.20*threshold verbatim tail, which on big-window/raised-threshold + # setups hoards 100-240K tokens per compaction. Unknown values fall back + # to lean inside the compressor. + compression_tail_mode = str(_compression_cfg.get("tail_mode", "lean")).strip().lower() # Minimum REAL (actionable) user messages guaranteed to survive in the # uncompressed tail (compression.min_tail_user_messages). Default 1 # preserves current behavior exactly — the existing single-user tail diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 93dd7ce61d..1086542a52 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -2323,7 +2323,7 @@ class ContextCompressor(ContextEngine): @property def tail_token_budget(self) -> int: if self._tail_token_budget is None: - if getattr(self, "tail_mode", "legacy") == "lean": + if getattr(self, "tail_mode", "lean") == "lean": # Lean mode (#compaction-v2): the verbatim tail is a small # recency window, not a context hoard — the upgraded summary # (verbatim user messages, constraints section, recovery @@ -2919,8 +2919,12 @@ class ContextCompressor(ContextEngine): self._apply_threshold_tokens_cap() # Recalculate token budgets for the new context length so the # compressor stays calibrated after a model switch (e.g. 200K → 32K). - target_tokens = int(self.threshold_tokens * self.summary_target_ratio) - self.tail_token_budget = target_tokens + # Reset to None and let the tail_token_budget property recompute + # through the MODE-AWARE path: assigning the legacy formula here + # directly silently reverted lean mode to the 0.20×threshold hoard + # on every mid-session model switch. + self._tail_token_budget = None + _ = self.tail_token_budget # eager recompute, same timing as before self.max_summary_tokens = min( int(context_length * 0.05), _SUMMARY_TOKENS_CEILING, ) @@ -3120,7 +3124,7 @@ class ContextCompressor(ContextEngine): proactive_prune_min_result_chars: int = 8000, proactive_prune_min_reclaim_tokens: int = 4096, min_tail_user_messages: int = 1, - tail_mode: str = "legacy", + tail_mode: str = "lean", ): self.model = model self.base_url = base_url @@ -3130,7 +3134,7 @@ class ContextCompressor(ContextEngine): # Lean tail mode (#compaction-v2): "lean" = small clamped recency # tail + verbatim-user-message summary section + recovery pointers; # "legacy" = 0.20*window tail (shipping behavior). - self.tail_mode = tail_mode if tail_mode in ("legacy", "lean") else "legacy" + self.tail_mode = tail_mode if tail_mode in ("legacy", "lean") else "lean" # Per-model threshold overrides (longest substring match wins). # Stored as a plain dict; resolved in _resolve_threshold(), then the # small-context floor is applied on top. @@ -4574,7 +4578,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb verbatim user messages and the recovery pointer never depend on the summarizer's cooperation. No-op in legacy mode. """ - if getattr(self, "tail_mode", "legacy") != "lean": + if getattr(self, "tail_mode", "lean") != "lean": return summary if _LEAN_ANCHOR_HEADING not in summary: summary += _redact_compaction_text( @@ -7355,7 +7359,7 @@ This compaction should PRIORITISE preserving all information related to the focu # Lean mode: snapshot pristine tool contents BEFORE Phase-1 pruning so # the chunk digests summarize what actually happened, not the pruned # stubs (#compaction-v2). Bounded per entry to keep memory sane. - if getattr(self, "tail_mode", "legacy") == "lean": + if getattr(self, "tail_mode", "lean") == "lean": self._lean_pristine_tools = { str(m.get("tool_call_id") or ""): (m.get("content") or "")[:80_000] for m in messages @@ -7433,7 +7437,7 @@ This compaction should PRIORITISE preserving all information related to the focu # budget binds without the tool-group alignment floor hoarding old # output (#compaction-v2). Runs before summary generation so the # recovery stubs are already in place if the summary aborts. - if getattr(self, "tail_mode", "legacy") == "lean": + if getattr(self, "tail_mode", "lean") == "lean": messages = self._demote_stale_tail_tools(messages, compress_end) # Snapshot the rehydration state so an aborted attempt below can roll # it back. The self-heal scan mutates ``_previous_summary`` (populating diff --git a/gateway/run.py b/gateway/run.py index b219097a03..915eee4d49 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -26628,6 +26628,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ("compression", "codex_gpt55_autoraise"), ("compression", "codex_app_server_auto"), ("compression", "target_ratio"), + ("compression", "tail_mode"), ("compression", "protect_last_n"), ("compression", "proactive_prune_tokens"), ("compression", "proactive_prune_min_result_chars"), diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 4a03422609..d3c36844a4 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -784,9 +784,8 @@ DEFAULT_CONFIG = { # threshold and this token count. Clamped to # the model's context length at apply-time. "target_ratio": 0.20, # fraction of threshold to preserve as recent tail - "tail_mode": "legacy", # tail retention policy (#87326): - # "legacy" — 0.20×window verbatim tail (default) - # "lean" — clamped 2.5%-of-window tail + "tail_mode": "lean", # tail retention policy (#87326): + # "lean" — clamped 2.5%-of-window tail (default) # (10K floor / 25K cap) plus chunked # digests, a mechanical anchor index, # verbatim user messages, and @@ -795,6 +794,9 @@ DEFAULT_CONFIG = { # tokens after compaction; costs a few # extra summarizer calls at the # compaction boundary. + # "legacy" — pre-#87326 0.20×threshold verbatim + # tail (100-240K tokens on big-window + # or raised-threshold setups). "protect_last_n": 20, # minimum recent messages to keep uncompressed "min_tail_user_messages": 1, # REAL (actionable) user messages guaranteed to # survive in the uncompressed tail. 1 = existing diff --git a/tests/agent/test_compression_small_ctx_threshold_floor.py b/tests/agent/test_compression_small_ctx_threshold_floor.py index 01a523b330..11797c5545 100644 --- a/tests/agent/test_compression_small_ctx_threshold_floor.py +++ b/tests/agent/test_compression_small_ctx_threshold_floor.py @@ -138,7 +138,21 @@ class TestSummaryBudgetEnvelope: class TestTailBudgetProportionality: def test_tail_budget_is_target_ratio_of_threshold(self): + # Legacy-mode contract: the threshold-proportional formula. The + # default is lean (clamped 10K-25K) since the tail-default flip, so + # this pins the LEGACY path explicitly. comp = _make(128_000) + comp.tail_mode = "legacy" + comp._tail_token_budget = None # force mode-aware recompute assert comp.tail_token_budget == int(comp.threshold_tokens * comp.summary_target_ratio) # Sanity: tail protection stays a modest slice of the window (<= 20%). assert comp.tail_token_budget <= comp.context_length * 0.20 + + def test_default_lean_tail_is_clamped(self): + # Default-mode contract after the flip: lean clamp, never the + # threshold-proportional hoard. + from agent.context_compressor import LEAN_TAIL_CAP_TOKENS, LEAN_TAIL_FLOOR_TOKENS + + comp = _make(128_000) + assert comp.tail_mode == "lean" + assert LEAN_TAIL_FLOOR_TOKENS <= comp.tail_token_budget <= LEAN_TAIL_CAP_TOKENS diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index d977eb9445..ede4840a6c 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -697,7 +697,10 @@ class TestNonStringContent: mock_response.choices[0].message = "plain summary text" with patch("agent.context_compressor.get_model_context_length", return_value=100000): - c = ContextCompressor(model="test", quiet_mode=True) + # Pin legacy: this test asserts the raw coerced string terminates + # the summary, which lean mode's verbatim-user-quote appendix + # intentionally follows. Coercion is mode-independent. + c = ContextCompressor(model="test", quiet_mode=True, tail_mode="legacy") messages = [ {"role": "user", "content": "do something"}, @@ -2043,7 +2046,9 @@ class TestUpdateModelBudgets: """tail_token_budget must change after switching to a different context length.""" from unittest.mock import patch with patch("agent.context_compressor.get_model_context_length", return_value=200_000): - comp = ContextCompressor("model-a", threshold_percent=0.50, quiet_mode=True) + comp = ContextCompressor( + "model-a", threshold_percent=0.50, quiet_mode=True, tail_mode="legacy", + ) old_tail = comp.tail_token_budget old_max_summary = comp.max_summary_tokens @@ -2056,11 +2061,75 @@ class TestUpdateModelBudgets: """Budgets should be proportional to context_length after update.""" from unittest.mock import patch with patch("agent.context_compressor.get_model_context_length", return_value=100_000): - comp = ContextCompressor("model-a", threshold_percent=0.50, quiet_mode=True) + comp = ContextCompressor( + "model-a", threshold_percent=0.50, quiet_mode=True, tail_mode="legacy", + ) comp.update_model("model-b", context_length=10_000) assert comp.tail_token_budget == int(comp.threshold_tokens * comp.summary_target_ratio) assert comp.max_summary_tokens == min(int(10_000 * 0.05), 4000) + def test_default_mode_is_lean(self): + """#tail-default-flip: an unconfigured compressor uses the lean tail. + + Behavior contract, not a snapshot: the default-constructed budget must + equal the lean clamp for the window, NOT the legacy threshold formula + (which on a 1M window would be ~100-170K tokens). + """ + from unittest.mock import patch + + from agent.context_compressor import ( + LEAN_TAIL_CAP_TOKENS, + LEAN_TAIL_FLOOR_TOKENS, + ) + + with patch("agent.context_compressor.get_model_context_length", return_value=1_000_000): + comp = ContextCompressor("model-big", threshold_percent=0.85, quiet_mode=True) + assert comp.tail_mode == "lean" + expected = max( + LEAN_TAIL_FLOOR_TOKENS, + min(LEAN_TAIL_CAP_TOKENS, int(comp.context_length * 0.025)), + ) + assert comp.tail_token_budget == expected + # The legacy hoard for this config would be far larger — prove the + # default no longer produces it. + assert comp.tail_token_budget < int(comp.threshold_tokens * comp.summary_target_ratio) + + def test_update_model_preserves_lean_mode(self): + """update_model() must recompute the tail through the MODE-AWARE path. + + Regression for the latent bug exposed by the default flip: the old + recompute assigned the legacy threshold formula directly, silently + reverting a lean compressor to the legacy hoard on every mid-session + model switch. + """ + from unittest.mock import patch + + from agent.context_compressor import ( + LEAN_TAIL_CAP_TOKENS, + LEAN_TAIL_FLOOR_TOKENS, + ) + + with patch("agent.context_compressor.get_model_context_length", return_value=1_000_000): + comp = ContextCompressor("model-a", threshold_percent=0.85, quiet_mode=True) + comp.update_model("model-b", context_length=400_000) + expected = max( + LEAN_TAIL_FLOOR_TOKENS, + min(LEAN_TAIL_CAP_TOKENS, int(400_000 * 0.025)), + ) + assert comp.tail_token_budget == expected + assert comp.tail_token_budget < int(comp.threshold_tokens * comp.summary_target_ratio) + + def test_explicit_legacy_still_honored(self): + """tail_mode: legacy in config keeps the pre-flip behavior exactly.""" + from unittest.mock import patch + + with patch("agent.context_compressor.get_model_context_length", return_value=1_000_000): + comp = ContextCompressor( + "model-a", threshold_percent=0.85, quiet_mode=True, tail_mode="legacy", + ) + assert comp.tail_mode == "legacy" + assert comp.tail_token_budget == int(comp.threshold_tokens * comp.summary_target_ratio) + class TestUpdateModelResetsCalibration: """#23767: update_model() must clear stale cross-call calibration state. @@ -3461,7 +3530,15 @@ class TestPreLlmFeasibilityCheck: """The target scenario from #60451: a tool-heavy transcript whose protected tail already holds most of the tokens, leaving a tiny middle window. The skip must fire and _generate_summary must not - be called.""" + be called. + + Pinned to legacy tail sizing: the scenario REQUIRES the big + payloads to sit inside the protected tail (legacy budget ≈ 17K on + this fixture). Under the lean default (10K clamp) the same + payloads fall into the compressible middle, so compression + correctly proceeds — that is desired behavior, not a skip case. + """ + compressor.tail_mode = "legacy" compressor._ineffective_compression_count = 1 msgs = [{"role": "system", "content": "system prompt"}] # Small middle: a few lightweight early exchanges. diff --git a/website/docs/developer-guide/context-compression-and-caching.md b/website/docs/developer-guide/context-compression-and-caching.md index 18a70d9792..146b53b58a 100644 --- a/website/docs/developer-guide/context-compression-and-caching.md +++ b/website/docs/developer-guide/context-compression-and-caching.md @@ -86,7 +86,7 @@ compression: # "glm-5.2": 0.40 # longest key wins). See "Per-model threshold # "claude-sonnet": 0.35 # overrides" below. target_ratio: 0.20 # How much of threshold to keep as tail (default: 0.20) - tail_mode: legacy # Tail retention policy: legacy | lean (default: legacy) + tail_mode: lean # Tail retention policy: lean | legacy (default: lean) protect_last_n: 20 # Minimum protected tail messages (default: 20) min_tail_user_messages: 1 # Real user messages guaranteed in the tail (default: 1) codex_gpt55_autoraise: true # gpt-5.5 on Codex OAuth: raise trigger to 85% (default: true) @@ -111,7 +111,7 @@ auxiliary: | `threshold` | `0.50` | 0.0-1.0 | Compression triggers when prompt tokens ≥ `threshold × context_length` | | `model_thresholds` | `{}` | map | Per-model overrides of `threshold`. Keys are substring-matched against the model name (longest match wins). The small-context floor still applies on top (see below) | | `target_ratio` | `0.20` | 0.10-0.80 | Controls tail protection token budget: `threshold_tokens × target_ratio` (legacy mode only — `lean` uses its own clamp) | -| `tail_mode` | `legacy` | `legacy`, `lean` | Tail retention policy. `legacy` keeps a `target_ratio`-sized verbatim tail (~100K+ tokens on big-window models). `lean` keeps a clamped tail of `2.5% × context window` (10K floor, 25K cap) and instead carries continuity in the summary: chunked identifier-preserving digests of the compacted region, a mechanically extracted anchor index (PR numbers, SHAs, paths, error strings — regex, never paraphrased), every real user message quoted verbatim (newest-first budget), and a `session_search` recovery pointer so the agent can re-access anything summarized away. Result on 500K-token real sessions: ~49K retained vs ~162K, with higher recall when paired with recovery (see `evals/compaction/results/`). Costs a few extra summarizer calls at the compaction boundary. Old tool results inside the lean tail are demoted to one-line stubs carrying a recovery pointer | +| `tail_mode` | `lean` | `lean`, `legacy` | Tail retention policy. `legacy` keeps a `target_ratio`-sized verbatim tail (~100K+ tokens on big-window models). `lean` keeps a clamped tail of `2.5% × context window` (10K floor, 25K cap) and instead carries continuity in the summary: chunked identifier-preserving digests of the compacted region, a mechanically extracted anchor index (PR numbers, SHAs, paths, error strings — regex, never paraphrased), every real user message quoted verbatim (newest-first budget), and a `session_search` recovery pointer so the agent can re-access anything summarized away. Result on 500K-token real sessions: ~49K retained vs ~162K, with higher recall when paired with recovery (see `evals/compaction/results/`). Costs a few extra summarizer calls at the compaction boundary. Old tool results inside the lean tail are demoted to one-line stubs carrying a recovery pointer | | `protect_last_n` | `20` | ≥1 | Minimum number of recent messages always preserved | | `min_tail_user_messages` | `1` | ≥1 | Minimum number of REAL (actionable) user messages guaranteed to survive in the uncompressed tail. `1` = the existing single last-user anchor (behavior-preserving default). Raise to e.g. `3` to keep the last 3 real user turns verbatim even when bulky tool outputs fill the tail token budget. Blank platform echoes, compaction handoffs, and synthetic continuation rows never count toward N. The guarantee wins over the tail token budget — the tail may exceed the budget when the anchor pulls the cut back | | `protect_first_n` | `3` | (hardcoded) | System prompt + first exchange always preserved | diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index ec1ae0bcfb..50c84b1ac7 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -826,7 +826,7 @@ compression: threshold: 0.50 # Compress at this % of context limit threshold_tokens: null # Absolute token cap (optional) — takes lower of ratio vs absolute target_ratio: 0.20 # Fraction of threshold to preserve as recent tail - tail_mode: legacy # Tail retention: "legacy" (0.20×window verbatim tail) or "lean" (clamped 2.5% tail, 10K-25K, with digests + anchor index + session_search recovery pointers in the summary — ~3x fewer retained tokens after compaction) + tail_mode: lean # Tail retention: "lean" (default — clamped 2.5% tail, 10K-25K, with digests + anchor index + session_search recovery pointers in the summary; ~3x fewer retained tokens after compaction) or "legacy" (0.20×threshold verbatim tail) protect_last_n: 20 # Min recent messages to keep uncompressed protect_first_n: 3 # Non-system head messages pinned across compactions (0 = pin nothing) in_place: true # Compact on the same session id (no rotation) — see below