fix(compaction): count api_content in tail budget
This commit is contained in:
@@ -1128,7 +1128,9 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) -
|
||||
and always-replayed provider fields. Always-replayed fields are charged because the preflight estimator sees
|
||||
the full shape; a mismatched size class protects blob-heavy rows as "small" and compaction re-fires.
|
||||
``charge_stale_thinking=False`` skips newest-turn-only thinking keys. Accounting only; never mutates."""
|
||||
content = msg.get("content") or ""
|
||||
# Charge the wire substitute, not both it and the clean display content.
|
||||
sidecar = msg.get("api_content")
|
||||
content = sidecar if isinstance(sidecar, str) and sidecar and msg.get("role") in ("user", "assistant") else msg.get("content") or ""
|
||||
text_tokens = estimate_tokens_rough(content) if isinstance(content, str) else _content_length_for_budget(content) // _CHARS_PER_TOKEN
|
||||
tokens = text_tokens + 10 # +10 for role/key overhead
|
||||
tokens += sum(estimate_tokens_rough(str(tc)) for tc in msg.get("tool_calls") or [] if isinstance(tc, dict))
|
||||
|
||||
@@ -12,6 +12,8 @@ ARE wire-replayed on every retained turn and stay charged unconditionally
|
||||
(#55572) — including native compaction checkpoints (#81747).
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.context_compressor import (
|
||||
_ALWAYS_REPLAYED_BUDGET_KEYS,
|
||||
_NEWEST_TURN_ONLY_BUDGET_KEYS,
|
||||
@@ -19,12 +21,46 @@ from agent.context_compressor import (
|
||||
_estimate_msg_budget_tokens,
|
||||
_last_assistant_index,
|
||||
)
|
||||
from agent.model_metadata import estimate_tokens_rough
|
||||
from agent.turn_context import substitute_api_content
|
||||
|
||||
|
||||
BIG_THINKING = "deliberation " * 400 # ~1.3K tokens of stale thinking text
|
||||
BIG_BLOB = [{"type": "reasoning", "encrypted_content": "x" * 4000}]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"message",
|
||||
[
|
||||
{"role": "user", "content": "clean", "api_content": "wire user"},
|
||||
{"role": "assistant", "content": "clean", "api_content": "wire assistant"},
|
||||
{"role": "system", "content": "clean", "api_content": "ignored"},
|
||||
{"role": "tool", "content": "clean", "api_content": "ignored"},
|
||||
{"role": "user", "content": "clean", "api_content": ""},
|
||||
{"role": "user", "content": "clean", "api_content": ["ignored"]},
|
||||
],
|
||||
ids=[
|
||||
"user-sidecar",
|
||||
"assistant-sidecar",
|
||||
"system-role",
|
||||
"tool-role",
|
||||
"empty-sidecar",
|
||||
"non-string-sidecar",
|
||||
],
|
||||
)
|
||||
def test_api_content_matches_wire_substitution_without_mutation(message):
|
||||
"""Tail budgeting must mirror ``substitute_api_content`` exactly."""
|
||||
original = dict(message)
|
||||
wire_message = dict(message)
|
||||
substitute_api_content(wire_message)
|
||||
|
||||
tokens = _estimate_msg_budget_tokens(message)
|
||||
|
||||
expected_content = wire_message.get("content") or ""
|
||||
assert tokens == estimate_tokens_rough(expected_content) + 10
|
||||
assert message == original
|
||||
|
||||
|
||||
def _assistant(thinking=False, codex=False):
|
||||
msg = {"role": "assistant", "content": "done"}
|
||||
if thinking:
|
||||
|
||||
Reference in New Issue
Block a user