Merge pull request #103513 from NousResearch/fix/subagent-context-cap
fix(delegation): compression_threshold_tokens is opt-in (default off, children keep the 500K ratio trigger); validate the value
This commit is contained in:
@@ -1220,6 +1220,14 @@ DEFAULT_CONFIG = {
|
||||
# {"extra_body": {"provider": {"sort": "throughput"}}}. Explicit values win OVER
|
||||
# runtime/parent overrides (extra_body deep-merged 1 level).
|
||||
"request_overrides": {},
|
||||
# compression_threshold_tokens: optional absolute cap on a subagent's compaction TRIGGER
|
||||
# (not the request payload), applied as the lower of this and the child's ratio threshold.
|
||||
# 0 (default) = no subagent-specific cap; children compact at the same 0.50 x window as the
|
||||
# parent (500K on a 1M model). A replay of a 1,393-agent run showed 200K-400K caps within
|
||||
# 5% of each other in cost once cache prefixes are intact, and every compaction is a
|
||||
# chance to lose detail, so the default stays off. A token count >= 16000 enables it;
|
||||
# other values (true, "200k") are config errors: warned and ignored.
|
||||
"compression_threshold_tokens": 0,
|
||||
# When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so
|
||||
# toolsets=["web"] doesn't strip MCP). false = strict intersection.
|
||||
"inherit_mcp_toolsets": True,
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
"""``delegation.compression_threshold_tokens`` is an OPTIONAL absolute cap on a subagent's compaction
|
||||
trigger, off by default, and its value is validated rather than coerced.
|
||||
|
||||
Default off: a 1M-window child compacts at the same 0.50 x window as its parent (500K). A replay of a
|
||||
1,393-agent run put 200K-400K caps within 5% of each other once cache prefixes are intact, and every
|
||||
compaction is a chance to lose detail, so the cap is opt-in. The validation matters because YAML
|
||||
``true`` coerces to int 1 (a one-token trigger) and ``"200k"`` would silently read as no cap.
|
||||
"""
|
||||
from types import SimpleNamespace
|
||||
|
||||
from agent.context_compressor import ContextCompressor
|
||||
from tools.delegate_tool import _apply_child_compression_cap, _child_compression_cap_tokens
|
||||
|
||||
|
||||
def _child(window=1_000_000, threshold=0.50, cap=None):
|
||||
cc = ContextCompressor(model="anthropic/claude-fable-5.1", threshold_percent=threshold,
|
||||
config_context_length=window, threshold_tokens_cap=cap)
|
||||
return SimpleNamespace(context_compressor=cc)
|
||||
|
||||
|
||||
def test_default_is_no_cap_child_keeps_the_ratio_trigger():
|
||||
child = _child()
|
||||
_apply_child_compression_cap(child, {})
|
||||
assert child.context_compressor.threshold_tokens == 500_000
|
||||
_apply_child_compression_cap(child, {"compression_threshold_tokens": 0})
|
||||
assert child.context_compressor.threshold_tokens == 500_000
|
||||
|
||||
|
||||
def test_explicit_cap_is_the_lower_of_delegation_and_global_and_never_raises():
|
||||
child = _child()
|
||||
_apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000})
|
||||
assert child.context_compressor.threshold_tokens == 300_000
|
||||
child = _child(cap=150_000)
|
||||
_apply_child_compression_cap(child, {"compression_threshold_tokens": 200_000})
|
||||
assert child.context_compressor.threshold_tokens == 150_000
|
||||
small = _child(window=128_000)
|
||||
before = small.context_compressor.threshold_tokens
|
||||
_apply_child_compression_cap(small, {"compression_threshold_tokens": 200_000})
|
||||
assert small.context_compressor.threshold_tokens == before # cap above the ratio trigger: no effect
|
||||
|
||||
|
||||
def test_config_values_are_validated_not_coerced():
|
||||
"""Independent-review witnesses: YAML ``true`` -> int 1 (a one-token trigger) and ``"200k"`` -> silently
|
||||
disabled. Both are ignored with a warning; the child keeps the ratio trigger."""
|
||||
for bad in (True, "200k", 5, 15_999, -1, 1.5):
|
||||
assert _child_compression_cap_tokens(bad) is None, bad
|
||||
for off in (None, 0, False):
|
||||
assert _child_compression_cap_tokens(off) is None
|
||||
assert _child_compression_cap_tokens(16_000) == 16_000
|
||||
assert _child_compression_cap_tokens(300_000.0) == 300_000
|
||||
child = _child()
|
||||
_apply_child_compression_cap(child, {"compression_threshold_tokens": True})
|
||||
assert child.context_compressor.threshold_tokens == 500_000
|
||||
@@ -110,6 +110,45 @@ def _apply_child_cache_ttl(child) -> None:
|
||||
if getattr(child, "_cache_ttl", None) == "1h":
|
||||
child._cache_ttl = "5m"
|
||||
|
||||
_CHILD_CAP_MIN = 16_000 # below this a child compresses on every call; treat as a config error
|
||||
|
||||
|
||||
def _child_compression_cap_tokens(raw) -> "int | None":
|
||||
"""Validated ``delegation.compression_threshold_tokens``: an int >= 16000, or None for "no cap".
|
||||
|
||||
Unset / ``0`` / ``false`` / ``null`` mean no subagent-specific cap: the child compacts at the
|
||||
same ratio trigger as everyone else (0.50 x window). A bool ``true`` (YAML) would coerce to 1
|
||||
and make every call compress; a string like ``"200k"`` would silently read as no cap. Both are
|
||||
config errors: warn and treat as unset so a typo never changes compaction behaviour."""
|
||||
if raw is None or raw is False or raw == 0:
|
||||
return None
|
||||
if isinstance(raw, bool) or not isinstance(raw, (int, float)) or int(raw) < _CHILD_CAP_MIN:
|
||||
logger.warning(
|
||||
"delegation.compression_threshold_tokens=%r is not a token count >= %d; ignoring it "
|
||||
"(children keep the ratio trigger).", raw, _CHILD_CAP_MIN,
|
||||
)
|
||||
return None
|
||||
return int(raw)
|
||||
|
||||
|
||||
def _apply_child_compression_cap(child, delegation_cfg: dict) -> None:
|
||||
"""Optional absolute cap on the child's compaction trigger, ``delegation.compression_threshold_tokens``
|
||||
(lower of it and any global ``compression.threshold_tokens``). Off by default: a 1M-window child
|
||||
compacts at 500K like its parent. The compressor applies the cap on first window resolution, which
|
||||
happens after construction, so setting it here is exactly equivalent to config."""
|
||||
from agent.context_compressor import ContextCompressor
|
||||
|
||||
cc = getattr(child, "context_compressor", None)
|
||||
if not isinstance(cc, ContextCompressor):
|
||||
return
|
||||
cap = _child_compression_cap_tokens((delegation_cfg or {}).get("compression_threshold_tokens"))
|
||||
if cap is None:
|
||||
return
|
||||
existing = cc.threshold_tokens_cap
|
||||
cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap
|
||||
if cc._threshold_tokens is not None: # already resolved: re-clamp now
|
||||
cc._apply_threshold_tokens_cap()
|
||||
|
||||
|
||||
def _build_child_agent(
|
||||
task_index: int,
|
||||
@@ -213,6 +252,7 @@ def _build_child_agent(
|
||||
child._progress_identity_ref = child_session_ref
|
||||
child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role
|
||||
child._subagent_id, child._parent_subagent_id = subagent_id, parent_subagent_id
|
||||
_apply_child_compression_cap(child, delegation_cfg)
|
||||
# Ownership chain for action=list/steer/stop; weakref so a finished parent
|
||||
# can be collected while a detached child record lingers in the registry.
|
||||
try:
|
||||
|
||||
@@ -2664,6 +2664,7 @@ delegation:
|
||||
# base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider)
|
||||
# api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY)
|
||||
# api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect.
|
||||
compression_threshold_tokens: 0 # Optional absolute cap on a subagent's compaction trigger (>= 16000); 0 = off, children use the ratio threshold
|
||||
# request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches).
|
||||
# extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints:
|
||||
# provider:
|
||||
|
||||
@@ -551,6 +551,8 @@ delegation:
|
||||
|
||||
When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare).
|
||||
|
||||
Subagents compact at the same ratio trigger as their parent (`compression.threshold`, 0.50 × window by default). `delegation.compression_threshold_tokens` (default `0`, off) adds an optional absolute cap on a child's compaction *trigger*, applied as the lower of it and the ratio threshold; it never touches the request payload or the parent. A token count of at least 16000 enables it; `true` or `"200k"` are config errors that are warned and ignored. It stays off by default because a replay of a 1,393-agent run put 200K–400K caps within 5% of each other in cost once cache prefixes are intact, and every compaction is a chance to lose detail.
|
||||
|
||||
`delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details.
|
||||
|
||||
:::tip
|
||||
|
||||
Reference in New Issue
Block a user