feat(agent): add skip_background_review flag to AIAgent constructor

Phase 8 of the Hermes Agent token leak mitigation plan
(ralplan-hermes-token-leaks.md §3.9). Adds a boolean kwarg
`skip_background_review` (default False) to AIAgent.__init__ that
suppresses the end-of-turn _spawn_background_review fork.

Each background review fork instantiates a new AIAgent with its own
~15K input tokens + up to 8 LLM iterations, accumulating ~30K tokens
per event in the worst case. On cron sessions there is no
human-in-the-loop benefit from the review (no skill-creation pressure,
nobody curating MEMORY.md), so the cost is pure waste.

The end-of-turn guard now reads:

    if (final_response and not interrupted
            and not getattr(self, "skip_background_review", False)
            and (_should_review_memory or _should_review_skills)):

skip_memory=True already disables the memory-review trigger; this
flag is the explicit single-switch off for both review paths.

Defaults to False, so behavior is unchanged for gateway/CLI callers
that omit the kwarg.

Tests: 5 new unit tests in tests/agent/test_skip_background_review.py
covering the default value, flag persistence, the gate short-circuit,
the gate fall-through, and a source-text assertion that the cron
scheduler sets the flag to True (separate commit).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
0xarkstar
2026-05-01 13:27:12 +09:00
committed by kshitij
parent 8370141f1c
commit eaeba6474f
4 changed files with 141 additions and 1 deletions
+8
View File
@@ -519,6 +519,7 @@ def init_agent(
skip_context_files: bool = False,
load_soul_identity: bool = False,
skip_memory: bool = False,
skip_background_review: bool = False,
session_db=None,
parent_session_id: str = None,
iteration_budget: "IterationBudget" = None,
@@ -610,6 +611,13 @@ def init_agent(
agent.memory_notifications = "on" # Memory update notifications: "off", "on", "verbose"
agent.skip_context_files = skip_context_files
agent.load_soul_identity = load_soul_identity
# Background review (memory/skill) opt-out switch. When True, skips the
# _spawn_background_review fork at end-of-turn -- avoids ~30K tokens /
# event of extra LLM cost on cron-style sessions where review forks
# provide no value (no human in the loop, no skill-creation pressure).
# skip_memory=True already disables the memory-review trigger; this
# flag is the explicit single-switch off for both review paths.
agent.skip_background_review = bool(skip_background_review)
agent.pass_session_id = pass_session_id
agent.log_prefix_chars = log_prefix_chars
agent.log_prefix = f"{log_prefix} " if log_prefix else ""
+9 -1
View File
@@ -721,7 +721,15 @@ def finalize_turn(
# Background memory/skill review — runs AFTER the response is delivered
# so it never competes with the user's task for model attention.
if final_response and not interrupted and (_should_review_memory or _should_review_skills):
# Suppressed when skip_background_review=True (e.g. cron) — review forks
# spawn another AIAgent (~30K tokens / event) and cron sessions have no
# human-in-the-loop benefit from the review.
if (
final_response
and not interrupted
and not getattr(agent, "skip_background_review", False)
and (_should_review_memory or _should_review_skills)
):
try:
agent._spawn_background_review(
messages_snapshot=list(messages),
+2
View File
@@ -496,6 +496,7 @@ class AIAgent:
skip_context_files: bool = False,
load_soul_identity: bool = False,
skip_memory: bool = False,
skip_background_review: bool = False,
session_db=None,
parent_session_id: str = None,
iteration_budget: "IterationBudget" = None,
@@ -581,6 +582,7 @@ class AIAgent:
skip_context_files=skip_context_files,
load_soul_identity=load_soul_identity,
skip_memory=skip_memory,
skip_background_review=skip_background_review,
session_db=session_db,
parent_session_id=parent_session_id,
iteration_budget=iteration_budget,
+122
View File
@@ -0,0 +1,122 @@
"""Tests for the skip_background_review constructor flag.
Verifies that AIAgent can be instructed to skip the end-of-turn
_spawn_background_review fork (~30K tokens / event), which is essential
on cron sessions that have no human-in-the-loop value from skill/memory
review forks.
Plan reference: ralplan-hermes-token-leaks.md §3.9 (Phase 8).
"""
from __future__ import annotations
from unittest.mock import patch
from run_agent import AIAgent
def _make_agent(skip_background_review: bool = False) -> AIAgent:
"""Construct a minimally-configured AIAgent for unit testing.
Mirrors the kwargs in tests/hermes_cli/test_timeouts.py — provider /
base_url stub plus skip_memory + skip_context_files to keep init fast.
"""
return AIAgent(
model="openai/gpt-4o-mini",
provider="openrouter",
api_key="sk-dummy",
base_url="https://openrouter.ai/api/v1",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
skip_background_review=skip_background_review,
platform="cli",
)
def test_default_skip_background_review_is_false() -> None:
"""Without an explicit override, AIAgent does NOT skip background review."""
agent = _make_agent()
assert agent.skip_background_review is False
def test_skip_background_review_flag_persists() -> None:
"""Passing skip_background_review=True records the flag on the instance."""
agent = _make_agent(skip_background_review=True)
assert agent.skip_background_review is True
def test_review_path_short_circuits_when_flag_set() -> None:
"""The end-of-turn review block is gated on `not self.skip_background_review`.
We don't drive a full conversation — instead we exercise the boolean
guard expression directly to confirm the gate works as wired.
"""
agent = _make_agent(skip_background_review=True)
# Simulate the conditions that would have fired the review:
final_response = "ok"
interrupted = False
_should_review_memory = True
_should_review_skills = True
with patch.object(agent, "_spawn_background_review") as mock_spawn:
# This is the exact guard from run_agent.py end-of-turn block.
if (
final_response
and not interrupted
and not getattr(agent, "skip_background_review", False)
and (_should_review_memory or _should_review_skills)
):
agent._spawn_background_review(
messages_snapshot=[],
review_memory=_should_review_memory,
review_skills=_should_review_skills,
)
mock_spawn.assert_not_called()
def test_review_path_fires_when_flag_unset() -> None:
"""Counterpart: with the flag off, the review path is reachable."""
agent = _make_agent(skip_background_review=False)
final_response = "ok"
interrupted = False
_should_review_memory = False
_should_review_skills = True
with patch.object(agent, "_spawn_background_review") as mock_spawn:
if (
final_response
and not interrupted
and not getattr(agent, "skip_background_review", False)
and (_should_review_memory or _should_review_skills)
):
agent._spawn_background_review(
messages_snapshot=[],
review_memory=_should_review_memory,
review_skills=_should_review_skills,
)
mock_spawn.assert_called_once()
def test_cron_construction_sets_skip_background_review() -> None:
"""The cron scheduler MUST construct AIAgent with skip_background_review=True.
Verified via source-text inspection — the cron scheduler is heavy to
boot in tests (loads gateway config, profile, telemetry), so we
assert that the source declares the flag rather than running the
scheduler. This still catches accidental removal of the flag.
"""
import pathlib
scheduler_src = pathlib.Path(__file__).resolve().parents[2] / "cron" / "scheduler.py"
text = scheduler_src.read_text(encoding="utf-8")
# The flag must appear inside the cron AIAgent(...) construction block.
# We look for it next to the existing skip_memory=True line.
assert "skip_background_review=True" in text, (
"cron/scheduler.py must construct AIAgent with skip_background_review=True "
"(see ralplan-hermes-token-leaks.md §3.9 / Phase 8)."
)