From eaeba6474fc0c6d7bc1fca00c1250da5e1c58019 Mon Sep 17 00:00:00 2001 From: 0xarkstar <0xarkstar@users.noreply.github.com> Date: Fri, 1 May 2026 13:27:12 +0900 Subject: [PATCH] feat(agent): add skip_background_review flag to AIAgent constructor MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 8 of the Hermes Agent token leak mitigation plan (ralplan-hermes-token-leaks.md §3.9). Adds a boolean kwarg `skip_background_review` (default False) to AIAgent.__init__ that suppresses the end-of-turn _spawn_background_review fork. Each background review fork instantiates a new AIAgent with its own ~15K input tokens + up to 8 LLM iterations, accumulating ~30K tokens per event in the worst case. On cron sessions there is no human-in-the-loop benefit from the review (no skill-creation pressure, nobody curating MEMORY.md), so the cost is pure waste. The end-of-turn guard now reads: if (final_response and not interrupted and not getattr(self, "skip_background_review", False) and (_should_review_memory or _should_review_skills)): skip_memory=True already disables the memory-review trigger; this flag is the explicit single-switch off for both review paths. Defaults to False, so behavior is unchanged for gateway/CLI callers that omit the kwarg. Tests: 5 new unit tests in tests/agent/test_skip_background_review.py covering the default value, flag persistence, the gate short-circuit, the gate fall-through, and a source-text assertion that the cron scheduler sets the flag to True (separate commit). Co-Authored-By: Claude Opus 4.7 (1M context) --- agent/agent_init.py | 8 ++ agent/turn_finalizer.py | 10 +- run_agent.py | 2 + tests/agent/test_skip_background_review.py | 122 +++++++++++++++++++++ 4 files changed, 141 insertions(+), 1 deletion(-) create mode 100644 tests/agent/test_skip_background_review.py diff --git a/agent/agent_init.py b/agent/agent_init.py index df98acc856..d038c76183 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -519,6 +519,7 @@ def init_agent( skip_context_files: bool = False, load_soul_identity: bool = False, skip_memory: bool = False, + skip_background_review: bool = False, session_db=None, parent_session_id: str = None, iteration_budget: "IterationBudget" = None, @@ -610,6 +611,13 @@ def init_agent( agent.memory_notifications = "on" # Memory update notifications: "off", "on", "verbose" agent.skip_context_files = skip_context_files agent.load_soul_identity = load_soul_identity + # Background review (memory/skill) opt-out switch. When True, skips the + # _spawn_background_review fork at end-of-turn -- avoids ~30K tokens / + # event of extra LLM cost on cron-style sessions where review forks + # provide no value (no human in the loop, no skill-creation pressure). + # skip_memory=True already disables the memory-review trigger; this + # flag is the explicit single-switch off for both review paths. + agent.skip_background_review = bool(skip_background_review) agent.pass_session_id = pass_session_id agent.log_prefix_chars = log_prefix_chars agent.log_prefix = f"{log_prefix} " if log_prefix else "" diff --git a/agent/turn_finalizer.py b/agent/turn_finalizer.py index d140545d4c..3b62028f23 100644 --- a/agent/turn_finalizer.py +++ b/agent/turn_finalizer.py @@ -721,7 +721,15 @@ def finalize_turn( # Background memory/skill review — runs AFTER the response is delivered # so it never competes with the user's task for model attention. - if final_response and not interrupted and (_should_review_memory or _should_review_skills): + # Suppressed when skip_background_review=True (e.g. cron) — review forks + # spawn another AIAgent (~30K tokens / event) and cron sessions have no + # human-in-the-loop benefit from the review. + if ( + final_response + and not interrupted + and not getattr(agent, "skip_background_review", False) + and (_should_review_memory or _should_review_skills) + ): try: agent._spawn_background_review( messages_snapshot=list(messages), diff --git a/run_agent.py b/run_agent.py index 63824c6eac..7eca7b8ce3 100644 --- a/run_agent.py +++ b/run_agent.py @@ -496,6 +496,7 @@ class AIAgent: skip_context_files: bool = False, load_soul_identity: bool = False, skip_memory: bool = False, + skip_background_review: bool = False, session_db=None, parent_session_id: str = None, iteration_budget: "IterationBudget" = None, @@ -581,6 +582,7 @@ class AIAgent: skip_context_files=skip_context_files, load_soul_identity=load_soul_identity, skip_memory=skip_memory, + skip_background_review=skip_background_review, session_db=session_db, parent_session_id=parent_session_id, iteration_budget=iteration_budget, diff --git a/tests/agent/test_skip_background_review.py b/tests/agent/test_skip_background_review.py new file mode 100644 index 0000000000..86d1a7f24f --- /dev/null +++ b/tests/agent/test_skip_background_review.py @@ -0,0 +1,122 @@ +"""Tests for the skip_background_review constructor flag. + +Verifies that AIAgent can be instructed to skip the end-of-turn +_spawn_background_review fork (~30K tokens / event), which is essential +on cron sessions that have no human-in-the-loop value from skill/memory +review forks. + +Plan reference: ralplan-hermes-token-leaks.md §3.9 (Phase 8). +""" +from __future__ import annotations + +from unittest.mock import patch + +from run_agent import AIAgent + + +def _make_agent(skip_background_review: bool = False) -> AIAgent: + """Construct a minimally-configured AIAgent for unit testing. + + Mirrors the kwargs in tests/hermes_cli/test_timeouts.py — provider / + base_url stub plus skip_memory + skip_context_files to keep init fast. + """ + return AIAgent( + model="openai/gpt-4o-mini", + provider="openrouter", + api_key="sk-dummy", + base_url="https://openrouter.ai/api/v1", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + skip_background_review=skip_background_review, + platform="cli", + ) + + +def test_default_skip_background_review_is_false() -> None: + """Without an explicit override, AIAgent does NOT skip background review.""" + agent = _make_agent() + assert agent.skip_background_review is False + + +def test_skip_background_review_flag_persists() -> None: + """Passing skip_background_review=True records the flag on the instance.""" + agent = _make_agent(skip_background_review=True) + assert agent.skip_background_review is True + + +def test_review_path_short_circuits_when_flag_set() -> None: + """The end-of-turn review block is gated on `not self.skip_background_review`. + + We don't drive a full conversation — instead we exercise the boolean + guard expression directly to confirm the gate works as wired. + """ + agent = _make_agent(skip_background_review=True) + + # Simulate the conditions that would have fired the review: + final_response = "ok" + interrupted = False + _should_review_memory = True + _should_review_skills = True + + with patch.object(agent, "_spawn_background_review") as mock_spawn: + # This is the exact guard from run_agent.py end-of-turn block. + if ( + final_response + and not interrupted + and not getattr(agent, "skip_background_review", False) + and (_should_review_memory or _should_review_skills) + ): + agent._spawn_background_review( + messages_snapshot=[], + review_memory=_should_review_memory, + review_skills=_should_review_skills, + ) + + mock_spawn.assert_not_called() + + +def test_review_path_fires_when_flag_unset() -> None: + """Counterpart: with the flag off, the review path is reachable.""" + agent = _make_agent(skip_background_review=False) + + final_response = "ok" + interrupted = False + _should_review_memory = False + _should_review_skills = True + + with patch.object(agent, "_spawn_background_review") as mock_spawn: + if ( + final_response + and not interrupted + and not getattr(agent, "skip_background_review", False) + and (_should_review_memory or _should_review_skills) + ): + agent._spawn_background_review( + messages_snapshot=[], + review_memory=_should_review_memory, + review_skills=_should_review_skills, + ) + + mock_spawn.assert_called_once() + + +def test_cron_construction_sets_skip_background_review() -> None: + """The cron scheduler MUST construct AIAgent with skip_background_review=True. + + Verified via source-text inspection — the cron scheduler is heavy to + boot in tests (loads gateway config, profile, telemetry), so we + assert that the source declares the flag rather than running the + scheduler. This still catches accidental removal of the flag. + """ + import pathlib + + scheduler_src = pathlib.Path(__file__).resolve().parents[2] / "cron" / "scheduler.py" + text = scheduler_src.read_text(encoding="utf-8") + + # The flag must appear inside the cron AIAgent(...) construction block. + # We look for it next to the existing skip_memory=True line. + assert "skip_background_review=True" in text, ( + "cron/scheduler.py must construct AIAgent with skip_background_review=True " + "(see ralplan-hermes-token-leaks.md §3.9 / Phase 8)." + )