From 40da0fd52f018199571841ccb2618d6b3ac8ed22 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:07:09 -0700 Subject: [PATCH 1/2] feat(delegation): subagents compress at an absolute context cap (delegation.compression_threshold_tokens, default 200K) A delegate_task child inherits the compression threshold as a RATIO of the model window. On a 1M-window model at the run's configured 0.85 that is an 850K-token trigger: in the 1,393-agent refactor run 1,373 of 1,375 children never compressed once, 62% of all API calls carried >150K of context, and the calls above 200K carried ~$10.9k of the $19.3k bill (58% of it cache WRITES, i.e. re-sending a 300-800K prefix on every call). A sawtooth replay of the logged calls with a 200K cap / 65K floor cuts context spend by ~49% (~$7k). Children are brief-driven and disposable; they re-read their brief and the files they touch, so a large window buys them little. New delegation.compression_threshold_tokens (default 200000) is applied to the child's ContextCompressor right after construction as the lower of it and any global compression.threshold_tokens; the parent's own trigger is untouched. 0 disables the subagent-specific cap. The compressor applies threshold_tokens_cap on first window resolution, so this is byte-equivalent to the user having set compression.threshold_tokens for the child. Live through the real spawn path (_build_child_agent, real imports, temp HERMES_HOME, 1M-window model): main child trigger 500,000 / branch 200,000; parent 500,000 on both. Tests (3): default caps a 1M child at 200K; the cap is the lower of the delegation and global values and never raises a small-window child's trigger; 0 disables and an already-resolved trigger is re-clamped. Docs: delegation.md, configuration.md. --- hermes_cli/config_defaults.py | 7 ++++ .../test_delegate_child_compression_cap.py | 40 +++++++++++++++++++ tools/delegate_tool.py | 22 ++++++++++ website/docs/user-guide/configuration.md | 1 + .../docs/user-guide/features/delegation.md | 2 + 5 files changed, 72 insertions(+) create mode 100644 tests/tools/test_delegate_child_compression_cap.py diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 234ef5dcc1..8ccb38cf30 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1216,6 +1216,13 @@ DEFAULT_CONFIG = { # {"extra_body": {"provider": {"sort": "throughput"}}}. Explicit values win OVER # runtime/parent overrides (extra_body deep-merged 1 level). "request_overrides": {}, + # compression_threshold_tokens: absolute context cap for subagents, applied as the lower of + # this and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window + # model otherwise compresses at threshold x window (850K at 0.85) and re-sends a 300-800K + # prefix on every call: in one 1,393-agent run 1,373 children never compressed and calls + # above 200K context carried ~55% of the bill. Independent of the parent's own threshold. + # 0 = no subagent-specific cap. + "compression_threshold_tokens": 200000, # When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so # toolsets=["web"] doesn't strip MCP). false = strict intersection. "inherit_mcp_toolsets": True, diff --git a/tests/tools/test_delegate_child_compression_cap.py b/tests/tools/test_delegate_child_compression_cap.py new file mode 100644 index 0000000000..024c3f0855 --- /dev/null +++ b/tests/tools/test_delegate_child_compression_cap.py @@ -0,0 +1,40 @@ +"""Subagents spawned by delegate_task compress at an absolute context cap, not threshold x window. + +In a 1,393-agent run on a 1M-window model the children's trigger was 0.85 x 1M = 850K; 1,373 of 1,375 +never compressed and calls above 200K context carried ~55% of the bill. +""" +from types import SimpleNamespace + +from agent.context_compressor import ContextCompressor +from tools.delegate_tool import _apply_child_compression_cap + + +def _child(window=1_000_000, threshold=0.85, cap=None): + cc = ContextCompressor(model="anthropic/claude-fable-5.1", threshold_percent=threshold, + config_context_length=window, threshold_tokens_cap=cap) + return SimpleNamespace(context_compressor=cc) + + +def test_default_cap_bounds_a_big_window_child(): + child = _child() + _apply_child_compression_cap(child, {}) + assert child.context_compressor.threshold_tokens == 200_000 + + +def test_cap_is_the_lower_of_delegation_and_global_and_never_raises(): + child = _child(cap=150_000) + _apply_child_compression_cap(child, {"compression_threshold_tokens": 200_000}) + assert child.context_compressor.threshold_tokens == 150_000 + small = _child(window=128_000) + before = small.context_compressor.threshold_tokens + _apply_child_compression_cap(small, {"compression_threshold_tokens": 200_000}) + assert small.context_compressor.threshold_tokens == before # cap above the ratio trigger: no effect + + +def test_zero_disables_and_an_already_resolved_trigger_is_reclamped(): + child = _child() + assert child.context_compressor.threshold_tokens == 850_000 # resolve first + _apply_child_compression_cap(child, {"compression_threshold_tokens": 0}) + assert child.context_compressor.threshold_tokens == 850_000 + _apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000}) + assert child.context_compressor.threshold_tokens == 300_000 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 70e98fe6a6..c0660647fd 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -102,6 +102,27 @@ def _open_child_session_db(parent_agent) -> Any: return acquire(_parent_db_path) if _parent_db_path is not None else acquire() return None + +def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: + """Cap the child's compaction trigger at ``delegation.compression_threshold_tokens`` (lower of it + and any global ``compression.threshold_tokens``). The compressor applies the cap on first window + resolution, which happens after construction, so setting it here is exactly equivalent to config.""" + from agent.context_compressor import ContextCompressor + + cc = getattr(child, "context_compressor", None) + if not isinstance(cc, ContextCompressor): + return + try: + cap = int((delegation_cfg or {}).get("compression_threshold_tokens", 200_000) or 0) + except (TypeError, ValueError): + cap = 0 + if cap <= 0: + return + existing = cc.threshold_tokens_cap + cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap + if cc._threshold_tokens is not None: # already resolved: re-clamp now + cc._apply_threshold_tokens_cap() + def _build_child_agent( task_index: int, goal: str, @@ -203,6 +224,7 @@ def _build_child_agent( child._progress_identity_ref = child_session_ref child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role child._subagent_id, child._parent_subagent_id = subagent_id, parent_subagent_id + _apply_child_compression_cap(child, delegation_cfg) # Ownership chain for action=list/steer/stop; weakref so a finished parent # can be collected while a detached child record lingers in the registry. try: diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index a663d1b317..2a18c3c318 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2636,6 +2636,7 @@ delegation: # base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider) # api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY) # api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect. + compression_threshold_tokens: 200000 # Subagent context cap (lower of this and the child's ratio threshold); 0 = none # request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches). # extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints: # provider: diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index a4ce8aba3b..dcc51599ac 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -537,6 +537,8 @@ delegation: When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). +Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold; `0` disables the subagent-specific cap. + `delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details. :::tip From ec4c1e0c985578ddbd1e883109167459b5b88706 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:29:28 -0700 Subject: [PATCH 2/2] fix(delegation): validate compression_threshold_tokens; state that it caps the trigger, not the payload Independent review: a YAML `true` coerced to int 1 and gave every child a one-token compression trigger; "200k" silently disabled the default cap. Values are now validated: an int >= 16000 is used, 0/false/null disable on purpose, anything else warns and falls back to the 200K default so a typo never costs money. Config comment and docs say it caps the compaction trigger, not a hard request-size limit. Test: true / "200k" / 5 -> default; 0 / false / None -> disabled; valid ints pass. --- hermes_cli/config_defaults.py | 3 ++- .../test_delegate_child_compression_cap.py | 9 +++++++ tools/delegate_tool.py | 27 +++++++++++++++---- .../docs/user-guide/features/delegation.md | 2 +- 4 files changed, 34 insertions(+), 7 deletions(-) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 8ccb38cf30..b8a1f2a5b9 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1221,7 +1221,8 @@ DEFAULT_CONFIG = { # model otherwise compresses at threshold x window (850K at 0.85) and re-sends a 300-800K # prefix on every call: in one 1,393-agent run 1,373 children never compressed and calls # above 200K context carried ~55% of the bill. Independent of the parent's own threshold. - # 0 = no subagent-specific cap. + # This caps the compaction TRIGGER, not the request payload. A token count >= 16000; 0 disables + # the subagent-specific cap. Other values (true, "200k") are config errors: warned, default used. "compression_threshold_tokens": 200000, # When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so # toolsets=["web"] doesn't strip MCP). false = strict intersection. diff --git a/tests/tools/test_delegate_child_compression_cap.py b/tests/tools/test_delegate_child_compression_cap.py index 024c3f0855..9277a99cb7 100644 --- a/tests/tools/test_delegate_child_compression_cap.py +++ b/tests/tools/test_delegate_child_compression_cap.py @@ -38,3 +38,12 @@ def test_zero_disables_and_an_already_resolved_trigger_is_reclamped(): assert child.context_compressor.threshold_tokens == 850_000 _apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000}) assert child.context_compressor.threshold_tokens == 300_000 + + +def test_config_values_are_validated_not_coerced(): + """Independent-review witnesses: YAML `true` coerced to int 1 (a one-token trigger) and "200k" + silently disabled the cap. Both fall back to the default with a warning; 0/false/null disable.""" + from tools.delegate_tool import _child_compression_cap_tokens as cap + assert cap(True) == 200_000 and cap("200k") == 200_000 and cap(5) == 200_000 + assert cap(0) is None and cap(False) is None and cap(None) is None + assert cap(150_000) == 150_000 and cap(300_000.0) == 300_000 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index c0660647fd..4b68cd0b49 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -103,6 +103,26 @@ def _open_child_session_db(parent_agent) -> Any: return None +_CHILD_CAP_DEFAULT = 200_000 +_CHILD_CAP_MIN = 16_000 # below this a child compresses on every call; treat as a config error + + +def _child_compression_cap_tokens(raw) -> "int | None": + """Validated ``delegation.compression_threshold_tokens``: an int >= 16000, or None for "no cap". + ``0``/``false``/``null`` disable on purpose. A bool ``true`` (YAML) would coerce to 1 and make + every call compress; a string like ``"200k"`` would silently disable the default. Both are + config errors: warn once and fall back to the default so the mistake never costs money.""" + if raw is None or raw is False or raw == 0: + return None + if isinstance(raw, bool) or not isinstance(raw, (int, float)) or int(raw) < _CHILD_CAP_MIN: + logger.warning( + "delegation.compression_threshold_tokens=%r is not a token count >= %d; using the default %d " + "(set 0 to disable the subagent cap).", raw, _CHILD_CAP_MIN, _CHILD_CAP_DEFAULT, + ) + return _CHILD_CAP_DEFAULT + return int(raw) + + def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: """Cap the child's compaction trigger at ``delegation.compression_threshold_tokens`` (lower of it and any global ``compression.threshold_tokens``). The compressor applies the cap on first window @@ -112,11 +132,8 @@ def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: cc = getattr(child, "context_compressor", None) if not isinstance(cc, ContextCompressor): return - try: - cap = int((delegation_cfg or {}).get("compression_threshold_tokens", 200_000) or 0) - except (TypeError, ValueError): - cap = 0 - if cap <= 0: + cap = _child_compression_cap_tokens((delegation_cfg or {}).get("compression_threshold_tokens", 200_000)) + if cap is None: return existing = cc.threshold_tokens_cap cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index dcc51599ac..1950755281 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -537,7 +537,7 @@ delegation: When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). -Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold; `0` disables the subagent-specific cap. +Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold and caps the compaction *trigger*, not the request payload. Accepted values: a token count of at least 16000, or `0` to disable the subagent-specific cap; anything else (a bare `true`, `"200k"`) is a config error that falls back to the default with a warning. `delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details.