Merge pull request #103513 from NousResearch/fix/subagent-context-cap

fix(delegation): compression_threshold_tokens is opt-in (default off, children keep the 500K ratio trigger); validate the value
This commit is contained in:
Teknium
2026-09-06 12:02:49 -07:00
committed by GitHub
5 changed files with 104 additions and 0 deletions
+8
View File
@@ -1220,6 +1220,14 @@ DEFAULT_CONFIG = {
# {"extra_body": {"provider": {"sort": "throughput"}}}. Explicit values win OVER
# runtime/parent overrides (extra_body deep-merged 1 level).
"request_overrides": {},
# compression_threshold_tokens: optional absolute cap on a subagent's compaction TRIGGER
# (not the request payload), applied as the lower of this and the child's ratio threshold.
# 0 (default) = no subagent-specific cap; children compact at the same 0.50 x window as the
# parent (500K on a 1M model). A replay of a 1,393-agent run showed 200K-400K caps within
# 5% of each other in cost once cache prefixes are intact, and every compaction is a
# chance to lose detail, so the default stays off. A token count >= 16000 enables it;
# other values (true, "200k") are config errors: warned and ignored.
"compression_threshold_tokens": 0,
# When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so
# toolsets=["web"] doesn't strip MCP). false = strict intersection.
"inherit_mcp_toolsets": True,
@@ -0,0 +1,53 @@
"""``delegation.compression_threshold_tokens`` is an OPTIONAL absolute cap on a subagent's compaction
trigger, off by default, and its value is validated rather than coerced.
Default off: a 1M-window child compacts at the same 0.50 x window as its parent (500K). A replay of a
1,393-agent run put 200K-400K caps within 5% of each other once cache prefixes are intact, and every
compaction is a chance to lose detail, so the cap is opt-in. The validation matters because YAML
``true`` coerces to int 1 (a one-token trigger) and ``"200k"`` would silently read as no cap.
"""
from types import SimpleNamespace
from agent.context_compressor import ContextCompressor
from tools.delegate_tool import _apply_child_compression_cap, _child_compression_cap_tokens
def _child(window=1_000_000, threshold=0.50, cap=None):
cc = ContextCompressor(model="anthropic/claude-fable-5.1", threshold_percent=threshold,
config_context_length=window, threshold_tokens_cap=cap)
return SimpleNamespace(context_compressor=cc)
def test_default_is_no_cap_child_keeps_the_ratio_trigger():
child = _child()
_apply_child_compression_cap(child, {})
assert child.context_compressor.threshold_tokens == 500_000
_apply_child_compression_cap(child, {"compression_threshold_tokens": 0})
assert child.context_compressor.threshold_tokens == 500_000
def test_explicit_cap_is_the_lower_of_delegation_and_global_and_never_raises():
child = _child()
_apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000})
assert child.context_compressor.threshold_tokens == 300_000
child = _child(cap=150_000)
_apply_child_compression_cap(child, {"compression_threshold_tokens": 200_000})
assert child.context_compressor.threshold_tokens == 150_000
small = _child(window=128_000)
before = small.context_compressor.threshold_tokens
_apply_child_compression_cap(small, {"compression_threshold_tokens": 200_000})
assert small.context_compressor.threshold_tokens == before # cap above the ratio trigger: no effect
def test_config_values_are_validated_not_coerced():
"""Independent-review witnesses: YAML ``true`` -> int 1 (a one-token trigger) and ``"200k"`` -> silently
disabled. Both are ignored with a warning; the child keeps the ratio trigger."""
for bad in (True, "200k", 5, 15_999, -1, 1.5):
assert _child_compression_cap_tokens(bad) is None, bad
for off in (None, 0, False):
assert _child_compression_cap_tokens(off) is None
assert _child_compression_cap_tokens(16_000) == 16_000
assert _child_compression_cap_tokens(300_000.0) == 300_000
child = _child()
_apply_child_compression_cap(child, {"compression_threshold_tokens": True})
assert child.context_compressor.threshold_tokens == 500_000
+40
View File
@@ -110,6 +110,45 @@ def _apply_child_cache_ttl(child) -> None:
if getattr(child, "_cache_ttl", None) == "1h":
child._cache_ttl = "5m"
_CHILD_CAP_MIN = 16_000 # below this a child compresses on every call; treat as a config error
def _child_compression_cap_tokens(raw) -> "int | None":
"""Validated ``delegation.compression_threshold_tokens``: an int >= 16000, or None for "no cap".
Unset / ``0`` / ``false`` / ``null`` mean no subagent-specific cap: the child compacts at the
same ratio trigger as everyone else (0.50 x window). A bool ``true`` (YAML) would coerce to 1
and make every call compress; a string like ``"200k"`` would silently read as no cap. Both are
config errors: warn and treat as unset so a typo never changes compaction behaviour."""
if raw is None or raw is False or raw == 0:
return None
if isinstance(raw, bool) or not isinstance(raw, (int, float)) or int(raw) < _CHILD_CAP_MIN:
logger.warning(
"delegation.compression_threshold_tokens=%r is not a token count >= %d; ignoring it "
"(children keep the ratio trigger).", raw, _CHILD_CAP_MIN,
)
return None
return int(raw)
def _apply_child_compression_cap(child, delegation_cfg: dict) -> None:
"""Optional absolute cap on the child's compaction trigger, ``delegation.compression_threshold_tokens``
(lower of it and any global ``compression.threshold_tokens``). Off by default: a 1M-window child
compacts at 500K like its parent. The compressor applies the cap on first window resolution, which
happens after construction, so setting it here is exactly equivalent to config."""
from agent.context_compressor import ContextCompressor
cc = getattr(child, "context_compressor", None)
if not isinstance(cc, ContextCompressor):
return
cap = _child_compression_cap_tokens((delegation_cfg or {}).get("compression_threshold_tokens"))
if cap is None:
return
existing = cc.threshold_tokens_cap
cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap
if cc._threshold_tokens is not None: # already resolved: re-clamp now
cc._apply_threshold_tokens_cap()
def _build_child_agent(
task_index: int,
@@ -213,6 +252,7 @@ def _build_child_agent(
child._progress_identity_ref = child_session_ref
child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role
child._subagent_id, child._parent_subagent_id = subagent_id, parent_subagent_id
_apply_child_compression_cap(child, delegation_cfg)
# Ownership chain for action=list/steer/stop; weakref so a finished parent
# can be collected while a detached child record lingers in the registry.
try:
+1
View File
@@ -2664,6 +2664,7 @@ delegation:
# base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider)
# api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY)
# api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect.
compression_threshold_tokens: 0 # Optional absolute cap on a subagent's compaction trigger (>= 16000); 0 = off, children use the ratio threshold
# request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches).
# extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints:
# provider:
@@ -551,6 +551,8 @@ delegation:
When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare).
Subagents compact at the same ratio trigger as their parent (`compression.threshold`, 0.50 × window by default). `delegation.compression_threshold_tokens` (default `0`, off) adds an optional absolute cap on a child's compaction *trigger*, applied as the lower of it and the ratio threshold; it never touches the request payload or the parent. A token count of at least 16000 enables it; `true` or `"200k"` are config errors that are warned and ignored. It stays off by default because a replay of a 1,393-agent run put 200K–400K caps within 5% of each other in cost once cache prefixes are intact, and every compaction is a chance to lose detail.
`delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details.
:::tip