fix(compression): keep lean tails lean after auxiliary feasibility

Lowering the session trigger must not replace the window-relative lean
selection budget with threshold times target_ratio. Invalidate the lean
cache through the existing property while preserving explicit legacy and
external-engine fallback behavior.

Narrow adaptation of the aux-sync diagnosis and invariants in #93576,
without adding a required recalibration method to context engines.
Related: #95681, #93576

Co-authored-by: Turgut Kural <58116817+TurgutKural@users.noreply.github.com>
This commit is contained in:
Teknium
2026-09-08 10:54:51 -07:00
parent 9d661c2c92
commit 19cd839d54
3 changed files with 58 additions and 3 deletions
+6 -3
View File
@@ -1697,13 +1697,16 @@ def _lower_threshold_to_aux_context(
) -> None:
"""Lower the live threshold to the aux model's window and tell the user how to fix config.
The summariser sends one user prompt (no system/tools), so threshold == aux_context is safe.
tail_token_budget and threshold_percent are kept in lockstep (as update_model does) or the 1.5x tail
ceiling exceeds the trigger and re-fires."""
Retention is recalibrated through its selected policy: lean is window-relative;
only legacy follows the lowered threshold."""
compressor = agent.context_compressor
old_threshold = compressor.threshold_tokens
new_threshold = compressor.threshold_tokens = aux_context
summary_target_ratio = getattr(compressor, "summary_target_ratio", None)
if isinstance(summary_target_ratio, (int, float)):
if getattr(compressor, "tail_mode", None) == "lean":
# Keep the window-relative policy owned by the compressor property.
compressor._tail_token_budget = None
elif isinstance(summary_target_ratio, (int, float)):
compressor.tail_token_budget = int(new_threshold * summary_target_ratio)
main_ctx = compressor.context_length
if main_ctx:
@@ -66,6 +66,47 @@ def _make_agent(
return agent
@pytest.mark.parametrize("main_context,aux_context", [(1_000_000, 512_000), (400_000, 80_000)])
def test_aux_sync_keeps_lean_tail_policy(main_context, aux_context):
"""Lowering only the trigger must not change window-relative retention."""
agent = _make_agent(main_context=main_context)
compressor = agent.context_compressor = ContextCompressor(
"test-main-model", config_context_length=main_context,
threshold_percent=0.85, quiet_mode=True,
)
before = compressor.tail_token_budget
agent._emit_status = lambda message: None
client = MagicMock(base_url="http://localhost/v1", api_key="test-key")
with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(client, "aux")), \
patch("agent.model_metadata.get_model_context_length", return_value=aux_context):
agent._check_compression_model_feasibility()
assert compressor.threshold_tokens == aux_context
assert compressor.tail_token_budget == before
# Repeated feasibility and subsequent model recalibration retain policy.
agent._check_compression_model_feasibility()
assert compressor.tail_token_budget == before
compressor.update_model("test-main-model", context_length=main_context)
assert compressor.tail_token_budget == before
def test_aux_sync_legacy_tail_follows_lowered_threshold():
"""Explicit legacy retention follows the current trigger, not its old cache."""
agent = _make_agent(main_context=1_000_000)
compressor = agent.context_compressor = ContextCompressor(
"test-main-model", config_context_length=1_000_000,
threshold_percent=0.85, tail_mode="legacy", quiet_mode=True,
)
before = compressor.tail_token_budget
agent._emit_status = lambda message: None
client = MagicMock(base_url="http://localhost/v1", api_key="test-key")
with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(client, "aux")), \
patch("agent.model_metadata.get_model_context_length", return_value=512_000):
agent._check_compression_model_feasibility()
assert compressor.threshold_tokens == 512_000
assert compressor.tail_token_budget < before
assert compressor.tail_token_budget == int(compressor.threshold_tokens * compressor.summary_target_ratio)
# ── Core warning logic ──────────────────────────────────────────────
@@ -216,6 +216,17 @@ Consumers observe the mode rather than diffing session ids:
Set `in_place: false` to restore the legacy rotating path, where each compaction commits a new session id linked to the previous one via `parent_session_id`.
### Auxiliary feasibility and tail retention
A smaller auxiliary compression model can lower the live compression trigger without
changing the selected tail policy. In `lean` mode the selection budget remains based
on the **main model's context window**: 2.5%, clamped to 10K–25K tokens. For example,
a 1M main model with a 512K auxiliary model retains a 25K selection budget even when
feasibility lowers its trigger from 850K to 512K. Explicit `legacy` mode instead
recomputes `threshold_tokens × target_ratio` (102,400 tokens at 512K × 0.20).
These are tail-selection budgets, not strict limits on the entire compacted context:
protected messages, boundary alignment, summaries, and anchors can add tokens.
### Per-model threshold overrides
`compression.model_thresholds` lets you trigger compaction at different points