fix(compression): announce the compacting status before the lazy feasibility probe

The first compaction of a session runs check_compression_model_feasibility()
inside compress_context() BEFORE _announce_compression_start(). That probe is
network-bound (live model catalog / provider lookups; connect timeouts stack up
through proxies and slow remote gateways), so every automatic entrypoint that
reaches compress_context without its own pre-emit — the post-tool gate in
agent/turn_preflight.py::compress_after_tool_results, overflow recovery,
manual /compress — left the client with no `kind="compacting"` status for the
whole probe. On Desktop that is a bare working-row spinner with no
"Summarizing thread" label (#111294).

Move the announcement ahead of the probe in the one choke point so every
caller is covered, and retire the announced phase (force_terminal) when the
probe's hard rejection propagates so the client never stays "compacting" for
an attempt that never started.

Live repro: /tmp probe driving compress_after_tool_results with a 2 s
feasibility stand-in — before: first compacting status at t=2.29 s (after the
block); after: t=0.13 s.
This commit is contained in:
teknium1
2026-09-14 18:42:05 -07:00
committed by Teknium
parent ee6fb0ed22
commit d6d9e67f54
2 changed files with 52 additions and 6 deletions
+14 -6
View File
@@ -3611,19 +3611,27 @@ def compress_context(
if not force and _automatic_compression_gate_blocks(agent, bypass_cooldown):
return messages, _existing_system_prompt(agent, system_message)
# Lazy feasibility probe (~400ms cold) on first attempt, not __init__; it sets
# _compression_warning so status replay still surfaces the warning. Marked checked
# only after the probe completes (transient failures are swallowed inside).
if not getattr(agent, "_compression_feasibility_checked", False):
check_compression_model_feasibility(agent)
agent._compression_feasibility_checked = True
_pre_msg_count = len(messages)
# In-place keeps the SAME session_id (no rotation/child/renumber/re-sync). A
# missing attribute must default True, not rotation, which can wedge sessions.
in_place = bool(getattr(agent, "compression_in_place", True))
# Announce BEFORE the lazy feasibility probe: its live catalog / provider lookups are
# network-bound (connect timeouts stack up through proxies), and until this status lands
# the Desktop working row is a bare spinner with no "Summarizing thread" label (#111294).
lifecycle = _announce_compression_start(
agent, message_count=_pre_msg_count, approx_tokens=approx_tokens, focus_topic=focus_topic, force=force
)
# Lazy feasibility probe (~400ms cold) on first attempt, not __init__; it sets
# _compression_warning so status replay still surfaces the warning. Marked checked
# only after the probe completes (transient failures are swallowed inside). A hard
# rejection propagates; retire the announced phase first so the client is not left compacting.
if not getattr(agent, "_compression_feasibility_checked", False):
try:
check_compression_model_feasibility(agent)
except Exception:
lifecycle.complete(force_terminal=True)
raise
agent._compression_feasibility_checked = True
lease, _abort_prompt = _acquire_compression_lease(
agent, commit_fence=commit_fence, lifecycle=lifecycle, system_message=system_message,
approx_tokens=approx_tokens, attempt_started_at=attempt.started_at,
+38
View File
@@ -499,6 +499,44 @@ class TestPreflightCompression:
("compacted", COMPACTION_DONE_STATUS),
]
def test_compress_context_announces_before_lazy_feasibility_probe(self, agent):
"""The compacting status must land BEFORE the first-attempt feasibility probe (live catalog /
provider lookups): a slow probe otherwise leaves the Desktop working row on a bare spinner with no
\"Summarizing thread\" label (#111294). The probe's hard rejection still retires the phase."""
import agent.conversation_compression as cc
agent.compression_enabled = True
agent._compression_feasibility_checked = False
events = []
agent.status_callback = lambda ev, msg: events.append((ev, msg))
def _fake_compress(messages, current_tokens=None, focus_topic=None, force=False, memory_context=""):
events.append(("compress", "started"))
return [{"role": "user", "content": f"{SUMMARY_PREFIX}\nPrevious conversation"}]
with (
patch.object(cc, "check_compression_model_feasibility",
side_effect=lambda a: events.append(("probe", "feasibility"))),
patch.object(agent.context_compressor, "compress", side_effect=_fake_compress),
patch.object(agent, "_build_system_prompt", return_value="new system prompt"),
patch("agent.conversation_compression.estimate_request_tokens_rough", return_value=42),
):
agent._compress_context([{"role": "user", "content": "hello"}], "system prompt", approx_tokens=1234)
assert events[:2] == [("lifecycle", COMPACTION_STATUS), ("probe", "feasibility")]
assert events[-1] == ("compacted", COMPACTION_DONE_STATUS)
assert agent._compression_feasibility_checked is True
# Hard rejection (aux window below minimum) propagates AND retires the announced phase.
agent._compression_feasibility_checked = False
events.clear()
with (
patch.object(cc, "check_compression_model_feasibility", side_effect=ValueError("aux too small")),
pytest.raises(ValueError),
):
agent._compress_context([{"role": "user", "content": "hello"}], "system prompt", approx_tokens=1234)
assert events == [("lifecycle", COMPACTION_STATUS), ("compacted", COMPACTION_DONE_STATUS)]
def test_compress_context_emits_one_terminal_status_when_lock_is_unavailable(self, agent):
"""A rejected lock must retire the started desktop compaction phase."""
agent.compression_enabled = False