diff --git a/agent/native_compaction.py b/agent/native_compaction.py index dd4cd1d08c..c8835f1392 100644 --- a/agent/native_compaction.py +++ b/agent/native_compaction.py @@ -303,6 +303,22 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool: return any(marker in text for marker in rejection_markers) +def has_compaction_checkpoint(items: Any) -> bool: + """Does this ``codex_reasoning_items`` sidecar carry a compaction checkpoint? + + A ``type: "compaction"`` item is the server-side stand-in for history that + has already been pruned — cumulative context, not per-turn reasoning. It + rides the same sidecar as ordinary reasoning items, so anything that + rewrites or discards that sidecar (or the message carrying it) has to ask + this question first: the checkpoint exists in exactly one place, and the + request that loses it loses the compacted history with it. + """ + return any( + isinstance(item, dict) and item.get("type") == "compaction" + for item in (items if isinstance(items, list) else ()) + ) + + def merge_interim_reasoning_items( prior_items: Any, new_items: Any, @@ -324,10 +340,6 @@ def merge_interim_reasoning_items( if isinstance(item, dict) and item.get("type") == "compaction" ] new_list = list(new_items) if isinstance(new_items, list) else [] - new_has_checkpoint = any( - isinstance(item, dict) and item.get("type") == "compaction" - for item in new_list - ) - if new_has_checkpoint or not kept_checkpoints: + if has_compaction_checkpoint(new_list) or not kept_checkpoints: return new_list return kept_checkpoints + new_list diff --git a/run_agent.py b/run_agent.py index 2cf9c455dc..0c82700505 100644 --- a/run_agent.py +++ b/run_agent.py @@ -4668,6 +4668,16 @@ class AIAgent: # empty-turn handling instead of being dropped here. codex_items = msg.get("codex_reasoning_items") if drop_codex_reasoning_items and isinstance(codex_items, list): + # A native compaction checkpoint rides this same sidecar and is + # the server-side stand-in for already-pruned history. Dropping + # the turn takes the checkpoint with it — the request then carries + # neither the compacted history nor the checkpoint that replaces + # it. Compaction pruning filters items for this reason rather than + # popping the key; a carrier is never thinking-only. + from agent.native_compaction import has_compaction_checkpoint + + if has_compaction_checkpoint(codex_items): + return False return any( isinstance(item, dict) and item.get("type") == "reasoning" for item in codex_items diff --git a/tests/run_agent/test_thinking_only_sanitizer.py b/tests/run_agent/test_thinking_only_sanitizer.py index 0ca0eab436..8acdda8286 100644 --- a/tests/run_agent/test_thinking_only_sanitizer.py +++ b/tests/run_agent/test_thinking_only_sanitizer.py @@ -178,3 +178,50 @@ class TestDropThinkingOnlyAndMergeUsers: assert out[0]["role"] == "system" assert out[1]["role"] == "user" assert out[1]["content"] == "u1\n\nu2" + + +# --------------------------------------------------------------------------- +# Native compaction checkpoints ride the same sidecar +# --------------------------------------------------------------------------- + + +class TestCompactionCheckpointCarrier: + """``type: "compaction"`` items are cumulative context, not per-turn + reasoning: they stand in for history the server already pruned, and they + exist in exactly one place. Compaction pruning filters the sidecar rather + than popping it so they survive on every retained message; dropping the + carrier message defeats that from the other direction. + """ + + CHECKPOINT = {"type": "compaction", "encrypted_content": "ckpt"} + REASONING = {"type": "reasoning", "encrypted_content": "per-turn"} + + def _carrier(self, items): + return {"role": "assistant", "content": "", "codex_reasoning_items": list(items)} + + def test_reasoning_only_carrier_is_still_thinking_only(self): + """The existing contract is unchanged for ordinary reasoning turns.""" + assert AIAgent._is_thinking_only_assistant(self._carrier([self.REASONING])) + + def test_checkpoint_alongside_reasoning_is_not_thinking_only(self): + msg = self._carrier([self.REASONING, self.CHECKPOINT]) + assert not AIAgent._is_thinking_only_assistant(msg) + + def test_checkpoint_order_does_not_matter(self): + msg = self._carrier([self.CHECKPOINT, self.REASONING]) + assert not AIAgent._is_thinking_only_assistant(msg) + + def test_checkpoint_survives_the_sanitizer_pass(self): + msgs = [ + {"role": "user", "content": "u1"}, + self._carrier([self.REASONING, self.CHECKPOINT]), + {"role": "user", "content": "u2"}, + ] + out = AIAgent._drop_thinking_only_and_merge_users(msgs) + surviving = [ + item + for m in out + for item in (m.get("codex_reasoning_items") or []) + if item.get("type") == "compaction" + ] + assert surviving == [self.CHECKPOINT]