fix(compression): truncated summaries no longer become compaction checkpoints (port of earendil-works/pi#7048)
A summarization response with finish_reason == "length" contains PARTIAL text — the generation stopped on the output-token cap mid-summary. Previously all compressor summarization sites accepted such responses as complete: the cut-off text replaced the real middle turns AND was fed back into every subsequent iterative-update prompt, compounding the loss across compactions. Guards added at all four summarization sites (whole bug class): - _generate_summary: length stop raises, gets the existing one-shot main-model fallback (a larger output budget may finish the summary), and on terminal failure ABORTS compression preserving the session unchanged (new _last_summary_truncated_failure flag, same class as empty-content). - _micro_summarize_one: partial rolling-summary merge is discarded; the exchange stays unabsorbed for a later pass. - _build_chunk_digests: partial lean digest degrades to the recover-via-session_search placeholder. - trajectory_compressor (sync + async): length stop raises into the existing retry/backoff loop. _response_finish_reason() reads dict- and object-shaped responses and returns "" when the provider omits the field, so proxies that never send finish_reason are unaffected. Ported from earendil-works/pi commit 97fa14e39 (pi#7048), adapted to hermes' abort-preserving compression failure machinery. Tests: tests/agent/test_compressor_truncated_summary_guard.py (12 tests; sabotage-verified — disabling the guards fails 4).
This commit is contained in:
@@ -56,6 +56,31 @@ _project_env = Path(__file__).parent / ".env"
|
||||
load_hermes_dotenv(hermes_home=_hermes_home, project_env=_project_env)
|
||||
|
||||
|
||||
def _response_finish_reason(response: Any) -> str:
|
||||
"""Return ``choices[0].finish_reason`` from a dict- or object-shaped response.
|
||||
|
||||
Local copy of ``agent.context_compressor._response_finish_reason`` —
|
||||
trajectory_compressor is a standalone CLI tool and deliberately avoids
|
||||
importing the (heavy) context compressor module. Returns the lowercased
|
||||
finish reason, or ``""`` when absent/unreadable.
|
||||
"""
|
||||
try:
|
||||
if isinstance(response, dict):
|
||||
choices = response.get("choices") or [{}]
|
||||
first = choices[0] if choices else {}
|
||||
reason = (
|
||||
first.get("finish_reason")
|
||||
if isinstance(first, dict)
|
||||
else getattr(first, "finish_reason", None)
|
||||
)
|
||||
else:
|
||||
choices = getattr(response, "choices", None) or []
|
||||
reason = getattr(choices[0], "finish_reason", None) if choices else None
|
||||
return str(reason).strip().lower() if reason else ""
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _effective_temperature_for_model(
|
||||
model: str,
|
||||
requested_temperature: float,
|
||||
@@ -658,6 +683,16 @@ Write only the summary, starting with "[CONTEXT SUMMARY]:" prefix."""
|
||||
_create_kwargs["temperature"] = summary_temperature
|
||||
response = self.client.chat.completions.create(**_create_kwargs)
|
||||
|
||||
_fr = _response_finish_reason(response)
|
||||
if _fr == "length":
|
||||
# Length stop = partial summary; storing it as the turn
|
||||
# replacement silently truncates the trajectory's memory.
|
||||
# Raise so the retry/backoff loop treats it as a failure
|
||||
# (pi#7048 class).
|
||||
raise RuntimeError(
|
||||
"trajectory summarization hit the output token cap "
|
||||
"(finish_reason=length); summary is incomplete"
|
||||
)
|
||||
summary = self._coerce_summary_content(response.choices[0].message.content)
|
||||
return self._ensure_summary_prefix(summary)
|
||||
|
||||
@@ -727,6 +762,13 @@ Write only the summary, starting with "[CONTEXT SUMMARY]:" prefix."""
|
||||
_create_kwargs["temperature"] = summary_temperature
|
||||
response = await self._get_async_client().chat.completions.create(**_create_kwargs)
|
||||
|
||||
if _response_finish_reason(response) == "length":
|
||||
# Length stop = partial summary; see sync sibling above
|
||||
# (pi#7048 class).
|
||||
raise RuntimeError(
|
||||
"trajectory summarization hit the output token cap "
|
||||
"(finish_reason=length); summary is incomplete"
|
||||
)
|
||||
summary = self._coerce_summary_content(response.choices[0].message.content)
|
||||
return self._ensure_summary_prefix(summary)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user