diff --git a/MagicMock/mock._session_db.db_path/123598073808400 b/MagicMock/mock._session_db.db_path/123598073808400 new file mode 100644 index 0000000000..f08e555a11 Binary files /dev/null and b/MagicMock/mock._session_db.db_path/123598073808400 differ diff --git a/MagicMock/mock._session_db.db_path/123598073808400.fts_rebuild.lock b/MagicMock/mock._session_db.db_path/123598073808400.fts_rebuild.lock new file mode 100644 index 0000000000..e69de29bb2 diff --git a/MagicMock/mock._session_db.db_path/123598073808400.quarantine.lock b/MagicMock/mock._session_db.db_path/123598073808400.quarantine.lock new file mode 100644 index 0000000000..e69de29bb2 diff --git a/MagicMock/mock._session_db.db_path/123598093110032 b/MagicMock/mock._session_db.db_path/123598093110032 new file mode 100644 index 0000000000..d20f981571 Binary files /dev/null and b/MagicMock/mock._session_db.db_path/123598093110032 differ diff --git a/MagicMock/mock._session_db.db_path/123598093110032.fts_rebuild.lock b/MagicMock/mock._session_db.db_path/123598093110032.fts_rebuild.lock new file mode 100644 index 0000000000..e69de29bb2 diff --git a/MagicMock/mock._session_db.db_path/123598093110032.quarantine.lock b/MagicMock/mock._session_db.db_path/123598093110032.quarantine.lock new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 8dbac85c3b..7c178d40eb 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -2,19 +2,13 @@ """ Delegate Tool -- Subagent Architecture -Spawns child AIAgent instances with isolated context, inherited toolsets, -and their own terminal sessions. Supports single-task and batch (parallel) -modes. Top-level model calls run in the background; orchestrator children -wait for their own workers so they can synthesize the results. - -Each child gets: - - A fresh conversation (no parent history) - - Its own task_id (own terminal session, file ops cache) - - The parent's toolsets, with child-only blocked tools stripped - - A focused system prompt built from the delegated goal + context - -The parent's context only sees the delegation call and the summary result, -never the child's intermediate tool calls or reasoning. +Spawns child AIAgent instances with a fresh conversation, their own task_id +(terminal session, file-ops cache), the parent's toolsets minus child-blocked +tools, and a focused system prompt built from goal + context. Single-task and +batch (parallel) modes; top-level model calls run in the background while +orchestrator children wait for their own workers. The parent only ever sees +the delegation call and the summary result, never the child's intermediate +tool calls or reasoning. """ import json @@ -32,10 +26,9 @@ from utils import is_truthy_value logger = logging.getLogger(__name__) -# The delegate_tool_* siblings hold the pieces split out of this module. Every -# moved name is re-imported here so ``tools.delegate_tool.`` keeps -# resolving for callers and for tests that patch it. Mutable flag globals -# (_spawn_paused, *_WARNED) live only in their owning module. +# The delegate_tool_* siblings hold the pieces split out of this module; every +# moved name is re-imported so ``tools.delegate_tool.`` keeps resolving for +# callers and patching tests. Mutable flag globals live only in their owning module. from tools.delegate_tool_child_run import ( # noqa: F401 _WorktreeReporter, _append_missed_steer, @@ -133,8 +126,6 @@ DELEGATE_BLOCKED_TOOLS = frozenset( # Nested delegation is granted by depth/role in _build_child_agent, never by the # model naming toolsets (there is no model-facing toolsets argument). - - def _normalize_role(r: Optional[str]) -> str: """'leaf' | 'orchestrator'; None/empty/unknown -> 'leaf' (unknown warns).""" if r is None or not r: @@ -246,20 +237,18 @@ def _resolve_child_toolsets( """Return ``(enabled_toolsets, disabled_toolsets)`` for a child. Children never gain tools the parent lacks: explicit ``toolsets`` are - intersected with the parent's (composite-expanded) set, otherwise the - parent's enabled set is inherited. Blocked tools are stripped twice — whole - blocked toolsets here, and exact one-tool deny toolsets via - ``disabled_toolsets`` so blocked names inside mixed bundles (hermes-cli) - are subtracted AFTER composite expansion and survive registry refreshes. - Orchestrators get ``delegation`` re-added unconditionally (role-granted, - not inherited). + intersected with the parent's (composite-expanded) set, else the parent's + enabled set is inherited. Blocked tools are stripped twice — whole blocked + toolsets here, and exact one-tool deny toolsets via ``disabled_toolsets`` so + blocked names inside mixed bundles (hermes-cli) are subtracted AFTER + composite expansion and survive registry refreshes. Orchestrators get + ``delegation`` re-added unconditionally (role-granted, not inherited). """ # enabled_toolsets=None means "all tools", so derive from loaded tool names. parent_enabled = getattr(parent_agent, "enabled_toolsets", None) if parent_enabled is not None: parent_toolsets = set(parent_enabled) elif parent_agent and hasattr(parent_agent, "valid_tool_names"): - # enabled_toolsets is None (all tools) — derive from loaded tool names import model_tools parent_toolsets = { @@ -284,20 +273,16 @@ def _resolve_child_toolsets( child_toolsets = _strip_blocked_tools(DEFAULT_TOOLSETS) raw_parent_disabled = getattr(parent_agent, "disabled_toolsets", None) - if isinstance(raw_parent_disabled, (list, tuple, set)): - inherited_disabled = [str(name) for name in raw_parent_disabled] - else: - inherited_disabled = [] + inherited_disabled = ( + [str(name) for name in raw_parent_disabled] if isinstance(raw_parent_disabled, (list, tuple, set)) else [] + ) if effective_role == "orchestrator": inherited_disabled = [name for name in inherited_disabled if name != "delegation"] + if "delegation" not in child_toolsets: + child_toolsets.append("delegation") child_disabled_toolsets = list( - dict.fromkeys( - inherited_disabled + _blocked_toolsets_for_role(effective_role) + ["kanban"] - ) + dict.fromkeys(inherited_disabled + _blocked_toolsets_for_role(effective_role) + ["kanban"]) ) - - if effective_role == "orchestrator" and "delegation" not in child_toolsets: - child_toolsets.append("delegation") return child_toolsets, child_disabled_toolsets @@ -789,16 +774,13 @@ def _recover_tasks_from_json_string(tasks: Any) -> tuple[Optional[List[Dict[str, return parsed, None -# Placeholder shapes for batch goal validation: bare 'TODO', bare 'task N' -# labels, or goals still carrying unexpanded template markers. -# -# The marker regex is deliberately NARROW: it only fires on snake_case / -# space-separated placeholder identifiers (``, `{file path}`, -# ``) — the shape LLM templates actually leave behind. Bare -# single-word brackets are left alone because legitimate coding goals are -# full of them: generics (`Vec`, `Result`), HTML tags (`
`), -# JSON/dict snippets (`{"key": 1}`), glob braces (`{a,b}`), and f-string -# style (`{i}`) must never be rejected (post-merge audit of #81141). +# Placeholder shapes for batch goal validation: bare 'TODO' / 'task N' labels, +# or unexpanded template markers. The marker regex is deliberately NARROW — +# only snake_case / space-separated placeholder identifiers (``, +# `{file path}`, ``), the shape LLM templates leave behind. Bare +# single-word brackets must never be rejected: legitimate goals are full of +# generics (`Vec`), HTML tags (`
`), dict snippets (`{"key": 1}`), glob +# braces (`{a,b}`) and f-string style (`{i}`). _PLACEHOLDER_GOAL_RE = re.compile(r"^(todo|task\s*\d+)$", re.IGNORECASE) _TEMPLATE_MARKER_RE = re.compile( r"<[A-Za-z][A-Za-z0-9]*(?:[ _-][A-Za-z0-9]+)+>" @@ -808,20 +790,14 @@ _MIN_BATCH_GOAL_LEN = 10 def _validate_batch_tasks(task_list: List[Dict[str, Any]]) -> Optional[str]: - """Validate a tasks=[...] batch beyond per-task goal presence. + """Validate a tasks=[...] batch beyond per-task goal presence; actionable + error string or None. - Returns an actionable error string, or None when the batch is valid. - - A one-entry array is the canonical single-task shape (the advertised - interface is tasks-only; legacy top-level `goal` is wrapped into a - one-entry batch), so no minimum count is enforced. The placeholder/ - template checks below still run on every entry. - - Duplicate goals are deliberately NOT rejected: identical-goal fan-outs - are a legitimate pattern (best-of-N / ensemble sampling), and blocking - them broke real workflows (post-merge audit of #81141). + No minimum count: a one-entry array is the canonical single-task shape + (legacy top-level `goal` is wrapped into one). Duplicate goals are + deliberately NOT rejected — identical-goal fan-outs (best-of-N / ensemble + sampling) are legitimate and blocking them broke real workflows. """ - for i, task in enumerate(task_list): goal = str(task.get("goal", "")).strip() normalized = " ".join(goal.lower().split()) @@ -1238,11 +1214,8 @@ def delegate_task( def _build_top_level_description() -> str: - """Compose the delegate_task tool description. - - Carries ONLY guidance stated nowhere else in the schema (limits live in - the 'tasks' parameter description, rebuilt per get_definitions() call). - """ + """delegate_task description: ONLY guidance stated nowhere else in the schema + (limits live in the 'tasks' parameter description, rebuilt per get_definitions()).""" try: orchestration_available = _get_max_spawn_depth() >= 2 and _get_orchestrator_enabled() except Exception: @@ -1311,17 +1284,11 @@ def _build_tasks_param_description() -> str: def _build_dynamic_schema_overrides() -> dict: - """Return per-call schema overrides reflecting current config. - - Plugged into ToolEntry.dynamic_schema_overrides so every - get_definitions() pass rewrites the description fields to the user's - actual limits. - """ + """Per-call schema overrides (ToolEntry.dynamic_schema_overrides): every + get_definitions() pass rewrites the descriptions to the user's actual limits.""" overrides_params = {**DELEGATE_TASK_SCHEMA["parameters"]} - # Deep-copy properties so we don't mutate the static schema dict. - overrides_params["properties"] = { - k: dict(v) for k, v in DELEGATE_TASK_SCHEMA["parameters"]["properties"].items() - } + # Copy properties so the static schema dict is never mutated. + overrides_params["properties"] = {k: dict(v) for k, v in DELEGATE_TASK_SCHEMA["parameters"]["properties"].items()} overrides_params["properties"]["tasks"]["description"] = _build_tasks_param_description() return {"description": _build_top_level_description(), "parameters": overrides_params} @@ -1329,14 +1296,11 @@ def _build_dynamic_schema_overrides() -> dict: DELEGATE_TASK_SCHEMA = { "name": "delegate_task", - # NOTE: description / tasks.description / role.description are placeholder - # values. The real text is generated per get_definitions() call by - # _build_dynamic_schema_overrides() (registered via - # dynamic_schema_overrides below) so the model sees the user's actual - # delegation.max_concurrent_children / max_spawn_depth, not the framework - # defaults. Building these lazily (instead of at module import) also - # avoids forcing cli.CLI_CONFIG to load before the test conftest can - # redirect HERMES_HOME. + # description / tasks.description are placeholders: the real text is built per + # get_definitions() call by _build_dynamic_schema_overrides() so the model sees + # the user's actual max_concurrent_children / max_spawn_depth. Lazy (not at + # import) so cli.CLI_CONFIG isn't forced to load before the test conftest + # redirects HERMES_HOME. "description": ( "Spawn one or more subagents in isolated contexts. " "Description is rebuilt at every get_definitions() call to reflect " @@ -1345,11 +1309,9 @@ DELEGATE_TASK_SCHEMA = { "parameters": { "type": "object", "properties": { - # NOTE: the handler also accepts the legacy single-goal shape — - # top-level `goal` (string), `context` (string), `output_schema` - # (object) — wrapped into a one-entry batch at dispatch. Legacy, - # unadvertised (old transcripts/callers only); tasks=[...] is the - # only advertised shape. Do not re-add these to the schema. + # The handler also accepts the legacy single-goal shape (top-level + # `goal`/`context`/`output_schema`), wrapped into a one-entry batch at + # dispatch. Unadvertised on purpose (old transcripts only); do not re-add. "tasks": { "type": "array", "minItems": 1, @@ -1388,18 +1350,14 @@ DELEGATE_TASK_SCHEMA = { }, "required": ["goal"], }, - # No maxItems — the runtime limit is configurable via - # delegation.max_concurrent_children (default 3) and - # enforced with a clear error in delegate_task(). - # NOTE: the handler also accepts a per-task `role` — legacy, - # ignored: delegation capability is depth-derived, not - # caller-declared. Unadvertised on purpose; do not re-add. + # No maxItems — the runtime limit (delegation.max_concurrent_children) + # is enforced with a clear error in delegate_task(). A per-task `role` + # is also accepted — legacy, ignored (capability is depth-derived); + # unadvertised on purpose, do not re-add. "description": "(rebuilt at get_definitions() time)", }, - # NOTE: the handler also accepts `background` (bool) — DEPRECATED, - # ignored: top-level delegations always run in the background. - # Deliberately unadvertised (old transcripts/callers only); do not - # re-add to the schema. + # `background` (bool) is also accepted — DEPRECATED, ignored: top-level + # delegations always run in the background. Unadvertised; do not re-add. "action": { "type": "string", "enum": ["spawn", "list", "steer", "stop"], @@ -1441,18 +1399,14 @@ from tools.registry import registry, tool_error def _model_background_value(args: dict, parent_agent=None) -> bool: """Background flag for the MODEL-facing dispatch path (registry fallback). - Delegations from the top-level agent always run in the background — the - model does not choose. This applies to both a single task and a fan-out - batch (the whole batch is one async unit that joins on all children and - returns one consolidated result). The one - exception is a delegation from an orchestrator subagent (depth > 0), which - needs its workers' results within its own turn. The live path is - ``run_agent._dispatch_delegate_task``; this lambda mirrors it for the rare - case the intercept is bypassed. Direct Python callers of ``delegate_task`` - keep the historical synchronous default. + Top-level delegations always run in the background — the model does not + choose — for single tasks and fan-out batches alike (one async unit, one + consolidated result). The exception is an orchestrator subagent (depth > 0), + which needs its workers' results within its own turn. The live path is + ``run_agent._dispatch_delegate_task``; this mirrors it for the rare case the + intercept is bypassed. Direct Python callers keep the synchronous default. """ - is_subagent = getattr(parent_agent, "_delegate_depth", 0) > 0 - return not is_subagent + return not getattr(parent_agent, "_delegate_depth", 0) > 0 _MODEL_HIDDEN_TASK_FIELDS = {"acp_command", "acp_args"} diff --git a/tools/delegate_tool_child_run.py b/tools/delegate_tool_child_run.py index adc58821b7..a929fb2033 100644 --- a/tools/delegate_tool_child_run.py +++ b/tools/delegate_tool_child_run.py @@ -502,14 +502,13 @@ def _defer_close_after_timeout(child: Any, child_future: Any) -> None: """Hand ``child.close()`` to a Future done-callback and drain its transports. The interrupt is cooperative: the worker still runs its finally path, so - closing the child now could close SQLite under its final write — the - done-callback is the first safe close boundary. The abandoned worker is - usually parked in an OpenSSL read; NEVER hard-close that transport from this - thread (cross-thread FD release under a live SSL read corrupts native - state) — shutdown() the pooled sockets instead, which settles the read with - EOF so the worker unwinds. One immediate sweep + one delayed re-sweep for a - connection opened in between; a worker that still won't settle keeps its - resources until process exit. + closing now could close SQLite under its final write — the done-callback is + the first safe boundary. The abandoned worker is usually parked in an + OpenSSL read; NEVER hard-close that transport from this thread (cross-thread + FD release under a live SSL read corrupts native state) — shutdown() the + pooled sockets so the read settles with EOF and the worker unwinds. One + immediate sweep + one delayed re-sweep for a connection opened in between; a + worker that still won't settle keeps its resources until process exit. """ child_future.add_done_callback( lambda _done: _close_child(child, "Failed to close timed-out child after worker exit") @@ -575,16 +574,15 @@ def _await_child( child_progress_cb: Any, worktree: _WorktreeReporter, ) -> tuple[Optional[Dict[str, Any]], Optional[_ChildFailure]]: - """Run the child's conversation on a daemon worker and wait for it. + """Run the child's conversation on a daemon worker; ``(result, None)`` or + ``(None, failure)`` on timeout/exception. - Returns ``(result, None)`` or ``(None, failure)`` on timeout/exception. - The hard timeout is off by default (``result(timeout=None)`` blocks until - the child finishes; stuck-child protection is the heartbeat). The worker is - a daemon: a timed-out child is abandoned and a stdlib non-daemon worker - would block interpreter exit at atexit-join time. The worker installs a - non-interactive approval callback so dangerous-command prompts never fall - back to ``input()`` and deadlock the parent TUI (deny vs approve follows - delegation.subagent_auto_approve). + The hard timeout is off by default (``result(timeout=None)`` blocks; stuck + children are the heartbeat's job). Daemon worker: a timed-out child is + abandoned and a non-daemon thread would block interpreter exit at atexit + join. The worker installs a non-interactive approval callback so dangerous + command prompts never fall back to ``input()`` and deadlock the parent TUI + (deny vs approve follows delegation.subagent_auto_approve). """ from tools.delegate_tool import (_get_child_timeout, _get_subagent_approval_callback, _set_subagent_approval_cb) from tools.daemon_pool import DaemonThreadPoolExecutor diff --git a/tools/delegate_tool_dispatch.py b/tools/delegate_tool_dispatch.py index b535b65b63..0f593f7b82 100644 --- a/tools/delegate_tool_dispatch.py +++ b/tools/delegate_tool_dispatch.py @@ -196,15 +196,13 @@ def _resolve_async_wake_sid(origin_wake_sid: str) -> Optional[str]: def _resolve_async_session_key(parent_agent: Any, origin_ui_session_id: str) -> tuple[str, str]: """``(session_key, origin_ui_session_id)`` the async registry routes completions by. - In desktop/TUI the routable key is the durable AIAgent.session_id: context - compression can rotate it mid-turn before the TUI-side session dict is - re-anchored, and a stale approval-context key would orphan the completion - for any desktop poller to consume. Gateway chats keep the platform - conversation key (agent:main:...). The CLI (single-process) has no bound - approval contextvar and no HERMES_SESSION_KEY, so the key resolves empty; - it drains completions through a positive-ownership filter keyed on the - durable session_id, so an empty key would fail closed — stamp the parent's - durable id (compression rotations resolve on the drain side via lineage). + Desktop/TUI: the routable key is the durable AIAgent.session_id — compression + can rotate it mid-turn before the TUI-side dict is re-anchored, and a stale + approval-context key would orphan the completion. Gateway chats keep the + platform conversation key (agent:main:...). The CLI has no bound approval + contextvar and no HERMES_SESSION_KEY, so the key resolves empty; its drain + is a positive-ownership filter keyed on the durable session_id, so an empty + key would fail closed — stamp the parent's durable id. """ from tools.approval import get_current_session_key @@ -227,17 +225,12 @@ def _resolve_async_session_key(parent_agent: Any, origin_ui_session_id: str) -> def _batch_progress_token(child_agents: List[Any]) -> tuple: - """Progress token for the async registry's stale monitor. - - The combined (api_call_count, current_tool, last_activity_ts) of every - child; last_activity_ts ticks on every streamed chunk, tool transition and - API-call start/completion, so a child streaming a long response is alive - even though api_call_count only advances when the call completes. A fully - frozen token past the stale threshold means the detached batch is wedged - (e.g. stuck inside the first model API call). ``in_tool`` is True while ANY - child is inside a tool so slow tools get the higher staleness ceiling, - mirroring the sync-path heartbeat monitor. - """ + """Progress token for the async registry's stale monitor: every child's + (api_call_count, current_tool, last_activity_ts). last_activity_ts ticks on + streamed chunks, tool transitions and API-call start/completion, so a child + streaming a long response counts as alive; a fully frozen token past the + threshold means the batch is wedged. ``in_tool`` is True while ANY child is + inside a tool so slow tools get the higher ceiling (mirrors the sync heartbeat).""" parts = [] in_tool = False for c in child_agents: diff --git a/tools/delegate_tool_registry.py b/tools/delegate_tool_registry.py index d066f6d131..50b43b4d93 100644 --- a/tools/delegate_tool_registry.py +++ b/tools/delegate_tool_registry.py @@ -159,20 +159,16 @@ def steer_subagent( owner_transport: Any = None, owner_session_record: Any = None, ) -> bool: - """Queue steering text into a single running subagent without stopping it. + """Queue steering text into a running subagent without stopping it. - The redirection-side mirror of interrupt_subagent(): resolves the live - child in the registry and calls AIAgent.steer(), which appends the text - to the child's last tool result at its next iteration boundary — the - current tool call is never cut. Returns True if a matching subagent - QUEUED the text while the child was still accepting work; False for an - unknown/closed id, an ownership mismatch, a record with no live agent, or - empty text. ``owner_session_id=None`` deliberately preserves the internal - in-process helper contract; gateway callers must pass exact authority. - - Acceptance and completion are linearized by the registry lock. If - acceptance wins but no delivery boundary remains, ``_run_single_child`` - drains the exact text into the completion entry as ``missed_steer``. + Mirror of interrupt_subagent(): calls AIAgent.steer(), which appends the + text to the child's last tool result at its next iteration boundary — the + current tool call is never cut. True iff the text was QUEUED while the child + still accepted work; False for unknown/closed id, ownership mismatch, no + live agent, or empty text. ``owner_session_id=None`` keeps the in-process + helper contract; gateway callers must pass exact authority. Acceptance and + completion are linearized by the registry lock: if acceptance wins but no + delivery boundary remains, the text lands in the entry as ``missed_steer``. """ if not text or not text.strip(): return False @@ -277,24 +273,16 @@ def _resolve_session_lineage(session_id: Optional[str], parent_agent: Any) -> st def _owns_subagent_record(record: Dict[str, Any], parent_agent: Any) -> bool: """True when *parent_agent*'s conversation owns this live-child record. - Two-tier check: - - 1. Object identity — the ``_delegate_parent_ref`` weakref chain stamped - at build time reaches *parent_agent*. Fast path for the common case - where the parent AIAgent object survives the whole run. - 2. Durable conversation lineage — the child was registered with the - owning conversation's durable session id - (``owner_agent_session_id``); match it against the calling parent's - ``session_id``, resolving compression-rotation lineage on both sides. - - Tier 2 exists because the identity chain is BRITTLE across parent-agent - rebuilds: the CLI sets ``self.agent = None`` mid-session (route-signature - change, credential refresh, /model, MoA one-shots) and constructs a NEW - AIAgent for the next turn while the child keeps running with a weakref to - the old object. The delivery path always survived this (it routes by - durable session id); the control path must use the same durable spine or - running children go invisible/unsteerable (observed live: deleg_88454b70 - / sa-0-dc0100f4, 2026-08-17). + Tier 1: object identity — the ``_delegate_parent_ref`` weakref chain reaches + *parent_agent* (fast path while the parent AIAgent survives the run). + Tier 2: durable lineage — the record's ``owner_agent_session_id`` matches the + caller's ``session_id``, resolving compression-rotation lineage on both + sides. Tier 2 exists because the identity chain is BRITTLE across parent + rebuilds: the CLI sets ``self.agent = None`` mid-session (route change, + credential refresh, /model, MoA one-shots) and builds a NEW AIAgent while + the child keeps a weakref to the old one. Delivery always routed by durable + session id; control must use the same spine or running children go + invisible/unsteerable. """ agent = record.get("agent") if _is_descendant_of(agent, parent_agent): diff --git a/tools/delegate_tool_results.py b/tools/delegate_tool_results.py index b1bcc891ce..7c99b982d0 100644 --- a/tools/delegate_tool_results.py +++ b/tools/delegate_tool_results.py @@ -343,19 +343,12 @@ def _trim_summary_with_footer(summary: str, cap: int, task_index: int) -> tuple[ def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int]: - """Per-summary character budget sized against the parent's *remaining* - context headroom, split across the batch. - - The overflow this guards against is N summaries entering the parent - context at once (batch fan-out), not any single summary being large. We - take a fraction of the headroom the parent has left (resolved context - length minus what's already in its prompt) and divide it across the batch, - converting tokens→chars at the standard ~4 chars/token estimate. - - Returns the per-summary char budget, or None when the parent's context - state is unknown (no compressor / no token count) — in which case the - caller falls back to the static char ceiling only. - """ + """Per-summary char budget from the parent's *remaining* context headroom + (context length minus prompt tokens minus the compressor's output reserve), + a fraction of it split across the batch at ~4 chars/token. Guards against N + summaries landing at once, not one large summary. None when the parent's + context state is unknown (no compressor / token count) — caller then uses + the static ceiling only.""" try: compressor = getattr(parent_agent, "context_compressor", None) context_length = getattr(compressor, "context_length", None) @@ -383,18 +376,14 @@ def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int] def _apply_summary_budget(results: List[Dict[str, Any]], parent_agent) -> None: - """Trim subagent summaries in-place so the batch can't overflow the - parent's context window, spilling full text to disk so nothing is lost. + """Trim subagent summaries in-place so a batch can't overflow the parent's + context window; full text is spilled to disk so nothing is lost. - The effective per-summary cap is the MIN of: - - the dynamic headroom budget (remaining parent context ÷ batch size), and - - the static ``delegation.max_summary_chars`` ceiling (0 = disabled). - - When a summary exceeds the cap, its full text is written to a file and the - in-context summary becomes a head slice plus a pointer to that file. This - addresses issue/PR #9126: batch fan-out returned N full summaries verbatim, - blowing the parent context and (on rate-limited providers) triggering a - compression/429 death spiral. + Per-summary cap = MIN(dynamic headroom budget: remaining parent context ÷ + batch size, static ``delegation.max_summary_chars`` ceiling; 0 = disabled). + Over-cap summaries become a head+tail slice plus a pointer to the spill file. + Without this, fan-out returned N full summaries verbatim, blowing the parent + context and (on rate-limited providers) triggering a compression/429 death spiral. """ from tools.delegate_tool import _load_config summaries = [r for r in results if isinstance(r, dict) and isinstance(r.get("summary"), str) and r["summary"]]