fix(compression): keep lean tails lean after auxiliary feasibility
Lowering the session trigger must not replace the window-relative lean selection budget with threshold times target_ratio. Invalidate the lean cache through the existing property while preserving explicit legacy and external-engine fallback behavior. Narrow adaptation of the aux-sync diagnosis and invariants in #93576, without adding a required recalibration method to context engines. Related: #95681, #93576 Co-authored-by: Turgut Kural <58116817+TurgutKural@users.noreply.github.com>
This commit is contained in:
@@ -1697,13 +1697,16 @@ def _lower_threshold_to_aux_context(
|
||||
) -> None:
|
||||
"""Lower the live threshold to the aux model's window and tell the user how to fix config.
|
||||
The summariser sends one user prompt (no system/tools), so threshold == aux_context is safe.
|
||||
tail_token_budget and threshold_percent are kept in lockstep (as update_model does) or the 1.5x tail
|
||||
ceiling exceeds the trigger and re-fires."""
|
||||
Retention is recalibrated through its selected policy: lean is window-relative;
|
||||
only legacy follows the lowered threshold."""
|
||||
compressor = agent.context_compressor
|
||||
old_threshold = compressor.threshold_tokens
|
||||
new_threshold = compressor.threshold_tokens = aux_context
|
||||
summary_target_ratio = getattr(compressor, "summary_target_ratio", None)
|
||||
if isinstance(summary_target_ratio, (int, float)):
|
||||
if getattr(compressor, "tail_mode", None) == "lean":
|
||||
# Keep the window-relative policy owned by the compressor property.
|
||||
compressor._tail_token_budget = None
|
||||
elif isinstance(summary_target_ratio, (int, float)):
|
||||
compressor.tail_token_budget = int(new_threshold * summary_target_ratio)
|
||||
main_ctx = compressor.context_length
|
||||
if main_ctx:
|
||||
|
||||
@@ -66,6 +66,47 @@ def _make_agent(
|
||||
return agent
|
||||
|
||||
|
||||
@pytest.mark.parametrize("main_context,aux_context", [(1_000_000, 512_000), (400_000, 80_000)])
|
||||
def test_aux_sync_keeps_lean_tail_policy(main_context, aux_context):
|
||||
"""Lowering only the trigger must not change window-relative retention."""
|
||||
agent = _make_agent(main_context=main_context)
|
||||
compressor = agent.context_compressor = ContextCompressor(
|
||||
"test-main-model", config_context_length=main_context,
|
||||
threshold_percent=0.85, quiet_mode=True,
|
||||
)
|
||||
before = compressor.tail_token_budget
|
||||
agent._emit_status = lambda message: None
|
||||
client = MagicMock(base_url="http://localhost/v1", api_key="test-key")
|
||||
with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(client, "aux")), \
|
||||
patch("agent.model_metadata.get_model_context_length", return_value=aux_context):
|
||||
agent._check_compression_model_feasibility()
|
||||
assert compressor.threshold_tokens == aux_context
|
||||
assert compressor.tail_token_budget == before
|
||||
# Repeated feasibility and subsequent model recalibration retain policy.
|
||||
agent._check_compression_model_feasibility()
|
||||
assert compressor.tail_token_budget == before
|
||||
compressor.update_model("test-main-model", context_length=main_context)
|
||||
assert compressor.tail_token_budget == before
|
||||
|
||||
|
||||
def test_aux_sync_legacy_tail_follows_lowered_threshold():
|
||||
"""Explicit legacy retention follows the current trigger, not its old cache."""
|
||||
agent = _make_agent(main_context=1_000_000)
|
||||
compressor = agent.context_compressor = ContextCompressor(
|
||||
"test-main-model", config_context_length=1_000_000,
|
||||
threshold_percent=0.85, tail_mode="legacy", quiet_mode=True,
|
||||
)
|
||||
before = compressor.tail_token_budget
|
||||
agent._emit_status = lambda message: None
|
||||
client = MagicMock(base_url="http://localhost/v1", api_key="test-key")
|
||||
with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(client, "aux")), \
|
||||
patch("agent.model_metadata.get_model_context_length", return_value=512_000):
|
||||
agent._check_compression_model_feasibility()
|
||||
assert compressor.threshold_tokens == 512_000
|
||||
assert compressor.tail_token_budget < before
|
||||
assert compressor.tail_token_budget == int(compressor.threshold_tokens * compressor.summary_target_ratio)
|
||||
|
||||
|
||||
# ── Core warning logic ──────────────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -216,6 +216,17 @@ Consumers observe the mode rather than diffing session ids:
|
||||
|
||||
Set `in_place: false` to restore the legacy rotating path, where each compaction commits a new session id linked to the previous one via `parent_session_id`.
|
||||
|
||||
### Auxiliary feasibility and tail retention
|
||||
|
||||
A smaller auxiliary compression model can lower the live compression trigger without
|
||||
changing the selected tail policy. In `lean` mode the selection budget remains based
|
||||
on the **main model's context window**: 2.5%, clamped to 10K–25K tokens. For example,
|
||||
a 1M main model with a 512K auxiliary model retains a 25K selection budget even when
|
||||
feasibility lowers its trigger from 850K to 512K. Explicit `legacy` mode instead
|
||||
recomputes `threshold_tokens × target_ratio` (102,400 tokens at 512K × 0.20).
|
||||
These are tail-selection budgets, not strict limits on the entire compacted context:
|
||||
protected messages, boundary alignment, summaries, and anchors can add tokens.
|
||||
|
||||
### Per-model threshold overrides
|
||||
|
||||
`compression.model_thresholds` lets you trigger compaction at different points
|
||||
|
||||
Reference in New Issue
Block a user