fix(delegation): summary headroom uses the aggregator's own prompt size; unknown usage means the static ceiling, never zero context

Independent review found two holes in the first fix. A parent with no usage
row yet was treated as 0 tokens used, so a 190K/200K prompt received a
384K-char dynamic summary budget instead of ~4K; the budget now returns None
(static ceiling only) when nothing is known. And under MoA the folded usage
includes advisor prompts that are not in the parent's context, over-stating
the prompt size and wrongly truncating summaries; turn_usage now records the
aggregator's pre-fold prompt_tokens as _last_prompt_size_tokens and the
budget reads that first.

Tests (2 new): unknown usage -> None; MoA-folded and unfolded parents with
the same real prompt get the same budget.
This commit is contained in:
Teknium
2026-09-05 06:18:41 -07:00
parent 903b9bf187
commit 0913885250
3 changed files with 40 additions and 4 deletions
+3
View File
@@ -143,6 +143,9 @@ def record_response_usage(
# Stash canonical usage for on_turn_complete(); keep the latest call's.
agent._last_turn_usage = dict(usage_dict)
# The parent's CURRENT prompt size for headroom math (delegate summary budgets): the
# aggregator's own prompt, never the MoA-folded total (advisor prompts are not in this context).
agent._last_prompt_size_tokens = int(aggregator_usage.prompt_tokens or 0)
# Persist only provider-confirmed context lengths, not probe tiers.
if getattr(compressor, "_context_probed", False):
@@ -88,3 +88,23 @@ def test_budget_uses_current_prompt_size_not_the_session_sum():
fresh = _FakeParent(context_length=200_000, used_tokens=30_000, max_tokens=8_000)
assert _parent_summary_char_budget(long_lived, 1) == _parent_summary_char_budget(fresh, 1)
assert _parent_summary_char_budget(long_lived, 1) > _MIN_SUMMARY_CHARS
def test_unknown_parent_usage_means_static_ceiling_not_zero_context():
"""Independent-review witness: a parent with no usage yet was treated as 0 tokens used, so a 190K/200K
prompt got a 384K-char summary budget instead of ~4K."""
from types import SimpleNamespace
from tools.delegate_tool_results import _parent_summary_char_budget
parent = SimpleNamespace(context_compressor=SimpleNamespace(context_length=200_000, max_tokens=0),
_last_turn_usage=None)
assert _parent_summary_char_budget(parent, 1) is None
def test_moa_fold_does_not_inflate_the_parents_prompt_size():
"""MoA folds advisor prompts into reported usage; the parent's context holds only the aggregator's."""
from types import SimpleNamespace
from tools.delegate_tool_results import _parent_summary_char_budget
cc = SimpleNamespace(context_length=200_000, max_tokens=0)
folded = SimpleNamespace(context_compressor=cc, _last_turn_usage={"prompt_tokens": 190_000}, _last_prompt_size_tokens=50_000)
unfolded = SimpleNamespace(context_compressor=cc, _last_turn_usage={"prompt_tokens": 50_000})
assert _parent_summary_char_budget(folded, 1) == _parent_summary_char_budget(unfolded, 1)
+17 -4
View File
@@ -218,6 +218,20 @@ def _trim_summary_with_footer(summary: str, cap: int, task_index: int) -> tuple[
footer_lines.append("─" * 37)
return head + "\n\n[... middle omitted — see footer ...]\n\n" + tail + "\n".join(footer_lines), spill_path
def _parent_prompt_size_tokens(parent_agent) -> Optional[int]:
"""The parent's current prompt size: the aggregator's own last ``prompt_tokens`` (pre-MoA-fold), else
the last provider usage. ``None`` when no request has completed yet: the caller then applies the
static ceiling only. Treating "no usage yet" as zero handed a 190K/200K parent a 384K-char budget."""
size = getattr(parent_agent, "_last_prompt_size_tokens", None)
if isinstance(size, (int, float)) and size > 0:
return int(size)
last_usage = getattr(parent_agent, "_last_turn_usage", None) or {}
used = last_usage.get("prompt_tokens") if isinstance(last_usage, dict) else None
if isinstance(used, (int, float)) and used > 0:
return int(used)
return None
def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int]:
"""Per-summary char budget from the parent's *remaining* context headroom (context length − the parent's
current prompt size − the compressor's output reserve), a fraction of it split across the batch at ~4
@@ -233,10 +247,9 @@ def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int]
context_length = getattr(compressor, "context_length", None)
if not isinstance(context_length, int) or context_length <= 0:
return None
last_usage = getattr(parent_agent, "_last_turn_usage", None) or {}
used_tokens = last_usage.get("prompt_tokens") if isinstance(last_usage, dict) else None
if not isinstance(used_tokens, (int, float)) or used_tokens < 0:
used_tokens = 0
used_tokens = _parent_prompt_size_tokens(parent_agent)
if used_tokens is None:
return None # no usage yet and nothing to estimate from: static ceiling only, never "zero context"
headroom_tokens = context_length - int(used_tokens) - int(getattr(compressor, "max_tokens", 0) or 0)
if headroom_tokens <= 0:
return _MIN_SUMMARY_CHARS # parent already over budget: floor only