From 40da0fd52f018199571841ccb2618d6b3ac8ed22 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:07:09 -0700 Subject: [PATCH] feat(delegation): subagents compress at an absolute context cap (delegation.compression_threshold_tokens, default 200K) A delegate_task child inherits the compression threshold as a RATIO of the model window. On a 1M-window model at the run's configured 0.85 that is an 850K-token trigger: in the 1,393-agent refactor run 1,373 of 1,375 children never compressed once, 62% of all API calls carried >150K of context, and the calls above 200K carried ~$10.9k of the $19.3k bill (58% of it cache WRITES, i.e. re-sending a 300-800K prefix on every call). A sawtooth replay of the logged calls with a 200K cap / 65K floor cuts context spend by ~49% (~$7k). Children are brief-driven and disposable; they re-read their brief and the files they touch, so a large window buys them little. New delegation.compression_threshold_tokens (default 200000) is applied to the child's ContextCompressor right after construction as the lower of it and any global compression.threshold_tokens; the parent's own trigger is untouched. 0 disables the subagent-specific cap. The compressor applies threshold_tokens_cap on first window resolution, so this is byte-equivalent to the user having set compression.threshold_tokens for the child. Live through the real spawn path (_build_child_agent, real imports, temp HERMES_HOME, 1M-window model): main child trigger 500,000 / branch 200,000; parent 500,000 on both. Tests (3): default caps a 1M child at 200K; the cap is the lower of the delegation and global values and never raises a small-window child's trigger; 0 disables and an already-resolved trigger is re-clamped. Docs: delegation.md, configuration.md. --- hermes_cli/config_defaults.py | 7 ++++ .../test_delegate_child_compression_cap.py | 40 +++++++++++++++++++ tools/delegate_tool.py | 22 ++++++++++ website/docs/user-guide/configuration.md | 1 + .../docs/user-guide/features/delegation.md | 2 + 5 files changed, 72 insertions(+) create mode 100644 tests/tools/test_delegate_child_compression_cap.py diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 234ef5dcc1..8ccb38cf30 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1216,6 +1216,13 @@ DEFAULT_CONFIG = { # {"extra_body": {"provider": {"sort": "throughput"}}}. Explicit values win OVER # runtime/parent overrides (extra_body deep-merged 1 level). "request_overrides": {}, + # compression_threshold_tokens: absolute context cap for subagents, applied as the lower of + # this and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window + # model otherwise compresses at threshold x window (850K at 0.85) and re-sends a 300-800K + # prefix on every call: in one 1,393-agent run 1,373 children never compressed and calls + # above 200K context carried ~55% of the bill. Independent of the parent's own threshold. + # 0 = no subagent-specific cap. + "compression_threshold_tokens": 200000, # When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so # toolsets=["web"] doesn't strip MCP). false = strict intersection. "inherit_mcp_toolsets": True, diff --git a/tests/tools/test_delegate_child_compression_cap.py b/tests/tools/test_delegate_child_compression_cap.py new file mode 100644 index 0000000000..024c3f0855 --- /dev/null +++ b/tests/tools/test_delegate_child_compression_cap.py @@ -0,0 +1,40 @@ +"""Subagents spawned by delegate_task compress at an absolute context cap, not threshold x window. + +In a 1,393-agent run on a 1M-window model the children's trigger was 0.85 x 1M = 850K; 1,373 of 1,375 +never compressed and calls above 200K context carried ~55% of the bill. +""" +from types import SimpleNamespace + +from agent.context_compressor import ContextCompressor +from tools.delegate_tool import _apply_child_compression_cap + + +def _child(window=1_000_000, threshold=0.85, cap=None): + cc = ContextCompressor(model="anthropic/claude-fable-5.1", threshold_percent=threshold, + config_context_length=window, threshold_tokens_cap=cap) + return SimpleNamespace(context_compressor=cc) + + +def test_default_cap_bounds_a_big_window_child(): + child = _child() + _apply_child_compression_cap(child, {}) + assert child.context_compressor.threshold_tokens == 200_000 + + +def test_cap_is_the_lower_of_delegation_and_global_and_never_raises(): + child = _child(cap=150_000) + _apply_child_compression_cap(child, {"compression_threshold_tokens": 200_000}) + assert child.context_compressor.threshold_tokens == 150_000 + small = _child(window=128_000) + before = small.context_compressor.threshold_tokens + _apply_child_compression_cap(small, {"compression_threshold_tokens": 200_000}) + assert small.context_compressor.threshold_tokens == before # cap above the ratio trigger: no effect + + +def test_zero_disables_and_an_already_resolved_trigger_is_reclamped(): + child = _child() + assert child.context_compressor.threshold_tokens == 850_000 # resolve first + _apply_child_compression_cap(child, {"compression_threshold_tokens": 0}) + assert child.context_compressor.threshold_tokens == 850_000 + _apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000}) + assert child.context_compressor.threshold_tokens == 300_000 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 70e98fe6a6..c0660647fd 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -102,6 +102,27 @@ def _open_child_session_db(parent_agent) -> Any: return acquire(_parent_db_path) if _parent_db_path is not None else acquire() return None + +def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: + """Cap the child's compaction trigger at ``delegation.compression_threshold_tokens`` (lower of it + and any global ``compression.threshold_tokens``). The compressor applies the cap on first window + resolution, which happens after construction, so setting it here is exactly equivalent to config.""" + from agent.context_compressor import ContextCompressor + + cc = getattr(child, "context_compressor", None) + if not isinstance(cc, ContextCompressor): + return + try: + cap = int((delegation_cfg or {}).get("compression_threshold_tokens", 200_000) or 0) + except (TypeError, ValueError): + cap = 0 + if cap <= 0: + return + existing = cc.threshold_tokens_cap + cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap + if cc._threshold_tokens is not None: # already resolved: re-clamp now + cc._apply_threshold_tokens_cap() + def _build_child_agent( task_index: int, goal: str, @@ -203,6 +224,7 @@ def _build_child_agent( child._progress_identity_ref = child_session_ref child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role child._subagent_id, child._parent_subagent_id = subagent_id, parent_subagent_id + _apply_child_compression_cap(child, delegation_cfg) # Ownership chain for action=list/steer/stop; weakref so a finished parent # can be collected while a detached child record lingers in the registry. try: diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index a663d1b317..2a18c3c318 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2636,6 +2636,7 @@ delegation: # base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider) # api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY) # api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect. + compression_threshold_tokens: 200000 # Subagent context cap (lower of this and the child's ratio threshold); 0 = none # request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches). # extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints: # provider: diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index a4ce8aba3b..dcc51599ac 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -537,6 +537,8 @@ delegation: When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). +Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold; `0` disables the subagent-specific cap. + `delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details. :::tip