From 19cd839d54c88a89995dd0cdff27dc7495aa1dc1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 8 Sep 2026 10:54:51 -0700 Subject: [PATCH] fix(compression): keep lean tails lean after auxiliary feasibility Lowering the session trigger must not replace the window-relative lean selection budget with threshold times target_ratio. Invalidate the lean cache through the existing property while preserving explicit legacy and external-engine fallback behavior. Narrow adaptation of the aux-sync diagnosis and invariants in #93576, without adding a required recalibration method to context engines. Related: #95681, #93576 Co-authored-by: Turgut Kural <58116817+TurgutKural@users.noreply.github.com> --- agent/conversation_compression.py | 9 ++-- .../run_agent/test_compression_feasibility.py | 41 +++++++++++++++++++ .../context-compression-and-caching.md | 11 +++++ 3 files changed, 58 insertions(+), 3 deletions(-) diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 7a82999937..12359913ed 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -1697,13 +1697,16 @@ def _lower_threshold_to_aux_context( ) -> None: """Lower the live threshold to the aux model's window and tell the user how to fix config. The summariser sends one user prompt (no system/tools), so threshold == aux_context is safe. - tail_token_budget and threshold_percent are kept in lockstep (as update_model does) or the 1.5x tail - ceiling exceeds the trigger and re-fires.""" + Retention is recalibrated through its selected policy: lean is window-relative; + only legacy follows the lowered threshold.""" compressor = agent.context_compressor old_threshold = compressor.threshold_tokens new_threshold = compressor.threshold_tokens = aux_context summary_target_ratio = getattr(compressor, "summary_target_ratio", None) - if isinstance(summary_target_ratio, (int, float)): + if getattr(compressor, "tail_mode", None) == "lean": + # Keep the window-relative policy owned by the compressor property. + compressor._tail_token_budget = None + elif isinstance(summary_target_ratio, (int, float)): compressor.tail_token_budget = int(new_threshold * summary_target_ratio) main_ctx = compressor.context_length if main_ctx: diff --git a/tests/run_agent/test_compression_feasibility.py b/tests/run_agent/test_compression_feasibility.py index 9017ea67d9..a6ba637e87 100644 --- a/tests/run_agent/test_compression_feasibility.py +++ b/tests/run_agent/test_compression_feasibility.py @@ -66,6 +66,47 @@ def _make_agent( return agent +@pytest.mark.parametrize("main_context,aux_context", [(1_000_000, 512_000), (400_000, 80_000)]) +def test_aux_sync_keeps_lean_tail_policy(main_context, aux_context): + """Lowering only the trigger must not change window-relative retention.""" + agent = _make_agent(main_context=main_context) + compressor = agent.context_compressor = ContextCompressor( + "test-main-model", config_context_length=main_context, + threshold_percent=0.85, quiet_mode=True, + ) + before = compressor.tail_token_budget + agent._emit_status = lambda message: None + client = MagicMock(base_url="http://localhost/v1", api_key="test-key") + with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(client, "aux")), \ + patch("agent.model_metadata.get_model_context_length", return_value=aux_context): + agent._check_compression_model_feasibility() + assert compressor.threshold_tokens == aux_context + assert compressor.tail_token_budget == before + # Repeated feasibility and subsequent model recalibration retain policy. + agent._check_compression_model_feasibility() + assert compressor.tail_token_budget == before + compressor.update_model("test-main-model", context_length=main_context) + assert compressor.tail_token_budget == before + + +def test_aux_sync_legacy_tail_follows_lowered_threshold(): + """Explicit legacy retention follows the current trigger, not its old cache.""" + agent = _make_agent(main_context=1_000_000) + compressor = agent.context_compressor = ContextCompressor( + "test-main-model", config_context_length=1_000_000, + threshold_percent=0.85, tail_mode="legacy", quiet_mode=True, + ) + before = compressor.tail_token_budget + agent._emit_status = lambda message: None + client = MagicMock(base_url="http://localhost/v1", api_key="test-key") + with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(client, "aux")), \ + patch("agent.model_metadata.get_model_context_length", return_value=512_000): + agent._check_compression_model_feasibility() + assert compressor.threshold_tokens == 512_000 + assert compressor.tail_token_budget < before + assert compressor.tail_token_budget == int(compressor.threshold_tokens * compressor.summary_target_ratio) + + # ── Core warning logic ────────────────────────────────────────────── diff --git a/website/docs/developer-guide/context-compression-and-caching.md b/website/docs/developer-guide/context-compression-and-caching.md index 0d08bdcc57..9221b2ccff 100644 --- a/website/docs/developer-guide/context-compression-and-caching.md +++ b/website/docs/developer-guide/context-compression-and-caching.md @@ -216,6 +216,17 @@ Consumers observe the mode rather than diffing session ids: Set `in_place: false` to restore the legacy rotating path, where each compaction commits a new session id linked to the previous one via `parent_session_id`. +### Auxiliary feasibility and tail retention + +A smaller auxiliary compression model can lower the live compression trigger without +changing the selected tail policy. In `lean` mode the selection budget remains based +on the **main model's context window**: 2.5%, clamped to 10K–25K tokens. For example, +a 1M main model with a 512K auxiliary model retains a 25K selection budget even when +feasibility lowers its trigger from 850K to 512K. Explicit `legacy` mode instead +recomputes `threshold_tokens × target_ratio` (102,400 tokens at 512K × 0.20). +These are tail-selection budgets, not strict limits on the entire compacted context: +protected messages, boundary alignment, summaries, and anchors can add tokens. + ### Per-model threshold overrides `compression.model_thresholds` lets you trigger compaction at different points