feat(agent): add skip_background_review flag to AIAgent constructor
Phase 8 of the Hermes Agent token leak mitigation plan
(ralplan-hermes-token-leaks.md §3.9). Adds a boolean kwarg
`skip_background_review` (default False) to AIAgent.__init__ that
suppresses the end-of-turn _spawn_background_review fork.
Each background review fork instantiates a new AIAgent with its own
~15K input tokens + up to 8 LLM iterations, accumulating ~30K tokens
per event in the worst case. On cron sessions there is no
human-in-the-loop benefit from the review (no skill-creation pressure,
nobody curating MEMORY.md), so the cost is pure waste.
The end-of-turn guard now reads:
if (final_response and not interrupted
and not getattr(self, "skip_background_review", False)
and (_should_review_memory or _should_review_skills)):
skip_memory=True already disables the memory-review trigger; this
flag is the explicit single-switch off for both review paths.
Defaults to False, so behavior is unchanged for gateway/CLI callers
that omit the kwarg.
Tests: 5 new unit tests in tests/agent/test_skip_background_review.py
covering the default value, flag persistence, the gate short-circuit,
the gate fall-through, and a source-text assertion that the cron
scheduler sets the flag to True (separate commit).
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -519,6 +519,7 @@ def init_agent(
|
||||
skip_context_files: bool = False,
|
||||
load_soul_identity: bool = False,
|
||||
skip_memory: bool = False,
|
||||
skip_background_review: bool = False,
|
||||
session_db=None,
|
||||
parent_session_id: str = None,
|
||||
iteration_budget: "IterationBudget" = None,
|
||||
@@ -610,6 +611,13 @@ def init_agent(
|
||||
agent.memory_notifications = "on" # Memory update notifications: "off", "on", "verbose"
|
||||
agent.skip_context_files = skip_context_files
|
||||
agent.load_soul_identity = load_soul_identity
|
||||
# Background review (memory/skill) opt-out switch. When True, skips the
|
||||
# _spawn_background_review fork at end-of-turn -- avoids ~30K tokens /
|
||||
# event of extra LLM cost on cron-style sessions where review forks
|
||||
# provide no value (no human in the loop, no skill-creation pressure).
|
||||
# skip_memory=True already disables the memory-review trigger; this
|
||||
# flag is the explicit single-switch off for both review paths.
|
||||
agent.skip_background_review = bool(skip_background_review)
|
||||
agent.pass_session_id = pass_session_id
|
||||
agent.log_prefix_chars = log_prefix_chars
|
||||
agent.log_prefix = f"{log_prefix} " if log_prefix else ""
|
||||
|
||||
@@ -721,7 +721,15 @@ def finalize_turn(
|
||||
|
||||
# Background memory/skill review — runs AFTER the response is delivered
|
||||
# so it never competes with the user's task for model attention.
|
||||
if final_response and not interrupted and (_should_review_memory or _should_review_skills):
|
||||
# Suppressed when skip_background_review=True (e.g. cron) — review forks
|
||||
# spawn another AIAgent (~30K tokens / event) and cron sessions have no
|
||||
# human-in-the-loop benefit from the review.
|
||||
if (
|
||||
final_response
|
||||
and not interrupted
|
||||
and not getattr(agent, "skip_background_review", False)
|
||||
and (_should_review_memory or _should_review_skills)
|
||||
):
|
||||
try:
|
||||
agent._spawn_background_review(
|
||||
messages_snapshot=list(messages),
|
||||
|
||||
@@ -496,6 +496,7 @@ class AIAgent:
|
||||
skip_context_files: bool = False,
|
||||
load_soul_identity: bool = False,
|
||||
skip_memory: bool = False,
|
||||
skip_background_review: bool = False,
|
||||
session_db=None,
|
||||
parent_session_id: str = None,
|
||||
iteration_budget: "IterationBudget" = None,
|
||||
@@ -581,6 +582,7 @@ class AIAgent:
|
||||
skip_context_files=skip_context_files,
|
||||
load_soul_identity=load_soul_identity,
|
||||
skip_memory=skip_memory,
|
||||
skip_background_review=skip_background_review,
|
||||
session_db=session_db,
|
||||
parent_session_id=parent_session_id,
|
||||
iteration_budget=iteration_budget,
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Tests for the skip_background_review constructor flag.
|
||||
|
||||
Verifies that AIAgent can be instructed to skip the end-of-turn
|
||||
_spawn_background_review fork (~30K tokens / event), which is essential
|
||||
on cron sessions that have no human-in-the-loop value from skill/memory
|
||||
review forks.
|
||||
|
||||
Plan reference: ralplan-hermes-token-leaks.md §3.9 (Phase 8).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
from run_agent import AIAgent
|
||||
|
||||
|
||||
def _make_agent(skip_background_review: bool = False) -> AIAgent:
|
||||
"""Construct a minimally-configured AIAgent for unit testing.
|
||||
|
||||
Mirrors the kwargs in tests/hermes_cli/test_timeouts.py — provider /
|
||||
base_url stub plus skip_memory + skip_context_files to keep init fast.
|
||||
"""
|
||||
return AIAgent(
|
||||
model="openai/gpt-4o-mini",
|
||||
provider="openrouter",
|
||||
api_key="sk-dummy",
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
skip_background_review=skip_background_review,
|
||||
platform="cli",
|
||||
)
|
||||
|
||||
|
||||
def test_default_skip_background_review_is_false() -> None:
|
||||
"""Without an explicit override, AIAgent does NOT skip background review."""
|
||||
agent = _make_agent()
|
||||
assert agent.skip_background_review is False
|
||||
|
||||
|
||||
def test_skip_background_review_flag_persists() -> None:
|
||||
"""Passing skip_background_review=True records the flag on the instance."""
|
||||
agent = _make_agent(skip_background_review=True)
|
||||
assert agent.skip_background_review is True
|
||||
|
||||
|
||||
def test_review_path_short_circuits_when_flag_set() -> None:
|
||||
"""The end-of-turn review block is gated on `not self.skip_background_review`.
|
||||
|
||||
We don't drive a full conversation — instead we exercise the boolean
|
||||
guard expression directly to confirm the gate works as wired.
|
||||
"""
|
||||
agent = _make_agent(skip_background_review=True)
|
||||
|
||||
# Simulate the conditions that would have fired the review:
|
||||
final_response = "ok"
|
||||
interrupted = False
|
||||
_should_review_memory = True
|
||||
_should_review_skills = True
|
||||
|
||||
with patch.object(agent, "_spawn_background_review") as mock_spawn:
|
||||
# This is the exact guard from run_agent.py end-of-turn block.
|
||||
if (
|
||||
final_response
|
||||
and not interrupted
|
||||
and not getattr(agent, "skip_background_review", False)
|
||||
and (_should_review_memory or _should_review_skills)
|
||||
):
|
||||
agent._spawn_background_review(
|
||||
messages_snapshot=[],
|
||||
review_memory=_should_review_memory,
|
||||
review_skills=_should_review_skills,
|
||||
)
|
||||
|
||||
mock_spawn.assert_not_called()
|
||||
|
||||
|
||||
def test_review_path_fires_when_flag_unset() -> None:
|
||||
"""Counterpart: with the flag off, the review path is reachable."""
|
||||
agent = _make_agent(skip_background_review=False)
|
||||
|
||||
final_response = "ok"
|
||||
interrupted = False
|
||||
_should_review_memory = False
|
||||
_should_review_skills = True
|
||||
|
||||
with patch.object(agent, "_spawn_background_review") as mock_spawn:
|
||||
if (
|
||||
final_response
|
||||
and not interrupted
|
||||
and not getattr(agent, "skip_background_review", False)
|
||||
and (_should_review_memory or _should_review_skills)
|
||||
):
|
||||
agent._spawn_background_review(
|
||||
messages_snapshot=[],
|
||||
review_memory=_should_review_memory,
|
||||
review_skills=_should_review_skills,
|
||||
)
|
||||
|
||||
mock_spawn.assert_called_once()
|
||||
|
||||
|
||||
def test_cron_construction_sets_skip_background_review() -> None:
|
||||
"""The cron scheduler MUST construct AIAgent with skip_background_review=True.
|
||||
|
||||
Verified via source-text inspection — the cron scheduler is heavy to
|
||||
boot in tests (loads gateway config, profile, telemetry), so we
|
||||
assert that the source declares the flag rather than running the
|
||||
scheduler. This still catches accidental removal of the flag.
|
||||
"""
|
||||
import pathlib
|
||||
|
||||
scheduler_src = pathlib.Path(__file__).resolve().parents[2] / "cron" / "scheduler.py"
|
||||
text = scheduler_src.read_text(encoding="utf-8")
|
||||
|
||||
# The flag must appear inside the cron AIAgent(...) construction block.
|
||||
# We look for it next to the existing skip_memory=True line.
|
||||
assert "skip_background_review=True" in text, (
|
||||
"cron/scheduler.py must construct AIAgent with skip_background_review=True "
|
||||
"(see ralplan-hermes-token-leaks.md §3.9 / Phase 8)."
|
||||
)
|
||||
Reference in New Issue
Block a user