diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 5f15909118..1b01342cb0 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1220,6 +1220,14 @@ DEFAULT_CONFIG = { # {"extra_body": {"provider": {"sort": "throughput"}}}. Explicit values win OVER # runtime/parent overrides (extra_body deep-merged 1 level). "request_overrides": {}, + # compression_threshold_tokens: optional absolute cap on a subagent's compaction TRIGGER + # (not the request payload), applied as the lower of this and the child's ratio threshold. + # 0 (default) = no subagent-specific cap; children compact at the same 0.50 x window as the + # parent (500K on a 1M model). A replay of a 1,393-agent run showed 200K-400K caps within + # 5% of each other in cost once cache prefixes are intact, and every compaction is a + # chance to lose detail, so the default stays off. A token count >= 16000 enables it; + # other values (true, "200k") are config errors: warned and ignored. + "compression_threshold_tokens": 0, # When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so # toolsets=["web"] doesn't strip MCP). false = strict intersection. "inherit_mcp_toolsets": True, diff --git a/tests/tools/test_delegate_child_compression_cap.py b/tests/tools/test_delegate_child_compression_cap.py new file mode 100644 index 0000000000..a7df734d09 --- /dev/null +++ b/tests/tools/test_delegate_child_compression_cap.py @@ -0,0 +1,53 @@ +"""``delegation.compression_threshold_tokens`` is an OPTIONAL absolute cap on a subagent's compaction +trigger, off by default, and its value is validated rather than coerced. + +Default off: a 1M-window child compacts at the same 0.50 x window as its parent (500K). A replay of a +1,393-agent run put 200K-400K caps within 5% of each other once cache prefixes are intact, and every +compaction is a chance to lose detail, so the cap is opt-in. The validation matters because YAML +``true`` coerces to int 1 (a one-token trigger) and ``"200k"`` would silently read as no cap. +""" +from types import SimpleNamespace + +from agent.context_compressor import ContextCompressor +from tools.delegate_tool import _apply_child_compression_cap, _child_compression_cap_tokens + + +def _child(window=1_000_000, threshold=0.50, cap=None): + cc = ContextCompressor(model="anthropic/claude-fable-5.1", threshold_percent=threshold, + config_context_length=window, threshold_tokens_cap=cap) + return SimpleNamespace(context_compressor=cc) + + +def test_default_is_no_cap_child_keeps_the_ratio_trigger(): + child = _child() + _apply_child_compression_cap(child, {}) + assert child.context_compressor.threshold_tokens == 500_000 + _apply_child_compression_cap(child, {"compression_threshold_tokens": 0}) + assert child.context_compressor.threshold_tokens == 500_000 + + +def test_explicit_cap_is_the_lower_of_delegation_and_global_and_never_raises(): + child = _child() + _apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000}) + assert child.context_compressor.threshold_tokens == 300_000 + child = _child(cap=150_000) + _apply_child_compression_cap(child, {"compression_threshold_tokens": 200_000}) + assert child.context_compressor.threshold_tokens == 150_000 + small = _child(window=128_000) + before = small.context_compressor.threshold_tokens + _apply_child_compression_cap(small, {"compression_threshold_tokens": 200_000}) + assert small.context_compressor.threshold_tokens == before # cap above the ratio trigger: no effect + + +def test_config_values_are_validated_not_coerced(): + """Independent-review witnesses: YAML ``true`` -> int 1 (a one-token trigger) and ``"200k"`` -> silently + disabled. Both are ignored with a warning; the child keeps the ratio trigger.""" + for bad in (True, "200k", 5, 15_999, -1, 1.5): + assert _child_compression_cap_tokens(bad) is None, bad + for off in (None, 0, False): + assert _child_compression_cap_tokens(off) is None + assert _child_compression_cap_tokens(16_000) == 16_000 + assert _child_compression_cap_tokens(300_000.0) == 300_000 + child = _child() + _apply_child_compression_cap(child, {"compression_threshold_tokens": True}) + assert child.context_compressor.threshold_tokens == 500_000 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 99b2e2e4b0..e6b12f4fd6 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -110,6 +110,45 @@ def _apply_child_cache_ttl(child) -> None: if getattr(child, "_cache_ttl", None) == "1h": child._cache_ttl = "5m" +_CHILD_CAP_MIN = 16_000 # below this a child compresses on every call; treat as a config error + + +def _child_compression_cap_tokens(raw) -> "int | None": + """Validated ``delegation.compression_threshold_tokens``: an int >= 16000, or None for "no cap". + + Unset / ``0`` / ``false`` / ``null`` mean no subagent-specific cap: the child compacts at the + same ratio trigger as everyone else (0.50 x window). A bool ``true`` (YAML) would coerce to 1 + and make every call compress; a string like ``"200k"`` would silently read as no cap. Both are + config errors: warn and treat as unset so a typo never changes compaction behaviour.""" + if raw is None or raw is False or raw == 0: + return None + if isinstance(raw, bool) or not isinstance(raw, (int, float)) or int(raw) < _CHILD_CAP_MIN: + logger.warning( + "delegation.compression_threshold_tokens=%r is not a token count >= %d; ignoring it " + "(children keep the ratio trigger).", raw, _CHILD_CAP_MIN, + ) + return None + return int(raw) + + +def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: + """Optional absolute cap on the child's compaction trigger, ``delegation.compression_threshold_tokens`` + (lower of it and any global ``compression.threshold_tokens``). Off by default: a 1M-window child + compacts at 500K like its parent. The compressor applies the cap on first window resolution, which + happens after construction, so setting it here is exactly equivalent to config.""" + from agent.context_compressor import ContextCompressor + + cc = getattr(child, "context_compressor", None) + if not isinstance(cc, ContextCompressor): + return + cap = _child_compression_cap_tokens((delegation_cfg or {}).get("compression_threshold_tokens")) + if cap is None: + return + existing = cc.threshold_tokens_cap + cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap + if cc._threshold_tokens is not None: # already resolved: re-clamp now + cc._apply_threshold_tokens_cap() + def _build_child_agent( task_index: int, @@ -213,6 +252,7 @@ def _build_child_agent( child._progress_identity_ref = child_session_ref child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role child._subagent_id, child._parent_subagent_id = subagent_id, parent_subagent_id + _apply_child_compression_cap(child, delegation_cfg) # Ownership chain for action=list/steer/stop; weakref so a finished parent # can be collected while a detached child record lingers in the registry. try: diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 4d21bc3df8..cccec83cfb 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2664,6 +2664,7 @@ delegation: # base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider) # api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY) # api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect. + compression_threshold_tokens: 0 # Optional absolute cap on a subagent's compaction trigger (>= 16000); 0 = off, children use the ratio threshold # request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches). # extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints: # provider: diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index e417e55fbe..90c5b9fe16 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -551,6 +551,8 @@ delegation: When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). +Subagents compact at the same ratio trigger as their parent (`compression.threshold`, 0.50 × window by default). `delegation.compression_threshold_tokens` (default `0`, off) adds an optional absolute cap on a child's compaction *trigger*, applied as the lower of it and the ratio threshold; it never touches the request payload or the parent. A token count of at least 16000 enables it; `true` or `"200k"` are config errors that are warned and ignored. It stays off by default because a replay of a 1,393-agent run put 200K–400K caps within 5% of each other in cost once cache prefixes are intact, and every compaction is a chance to lose detail. + `delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details. :::tip