diff --git a/agent/background_review.py b/agent/background_review.py index 01820cfffd..b5728f3fe8 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -1032,12 +1032,19 @@ def spawn_background_review_thread( messages_snapshot: List[Dict], review_memory: bool = False, review_skills: bool = False, + focus: Optional[str] = None, ): """Build the review thread target and prompt for a background review. Returns a ``(target, prompt)`` tuple. The caller (``AIAgent._spawn_background_review``) owns the actual ``threading.Thread`` construction so test-level patches of ``run_agent.threading.Thread`` keep working. + + ``focus`` is optional user steering (the ``/refine [instructions]`` + path): appended to the chosen review prompt so the fork prioritizes what + the user asked for while keeping the same guardrails. Automatic + post-turn reviews pass ``None`` — their prompts are byte-identical to + before this parameter existed. """ # Pick the right prompt based on which triggers fired. Allow per-agent # override (the prompts moved to module-level constants but old code paths @@ -1049,6 +1056,15 @@ def spawn_background_review_thread( else: prompt = getattr(agent, "_SKILL_REVIEW_PROMPT", _SKILL_REVIEW_PROMPT) + focus = (focus or "").strip() + if focus: + prompt = ( + f"{prompt}\n\n" + f"The user explicitly requested this review with the following " + f"focus — prioritize it over the general instructions above:\n" + f"{focus}" + ) + def _target() -> None: _run_review_in_thread(agent, messages_snapshot, prompt) diff --git a/cli.py b/cli.py index 08ec446dda..26e202711d 100644 --- a/cli.py +++ b/cli.py @@ -10302,6 +10302,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): self._handle_goal_command(cmd_original) elif canonical == "heartbeat": self._handle_heartbeat_command(cmd_original) + elif canonical == "refine": + self._handle_refine_command(cmd_original) elif canonical == "moa": # /moa is one-shot sugar only: run a single prompt through the # default MoA preset, then restore the prior model. To *switch* to a diff --git a/gateway/run.py b/gateway/run.py index 8542558e0b..4aee4c91ef 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -15299,6 +15299,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if canonical == "heartbeat": return await self._handle_heartbeat_command(event) + if canonical == "refine": + return await self._handle_refine_command(event) if canonical == "moa": # /moa is one-shot sugar only: run a single prompt through the diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index d8475ccf28..c65acc814f 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -2844,6 +2844,50 @@ class GatewaySlashCommandsMixin: "elapsed. Lives while the gateway runs — use `hermes cron` for durable schedules." ) + async def _handle_refine_command(self, event: "MessageEvent") -> str: + """Handle /refine — run the memory/skill review fork on demand. + + Uses the session's cached AIAgent (idle agents live in + ``_agent_cache``). The review runs in a daemon thread against a + snapshot of the conversation; the live session and prompt cache are + untouched. Requires the session to have at least one completed turn. + """ + args = (event.get_command_args() or "").strip() + quick_key = self._session_key_for_source(event.source) if event.source else None + if not quick_key: + return "Refine unavailable (no session)." + if quick_key in self._running_agents: + return "Agent is running — wait for the turn to finish, then /refine." + + agent = None + cache_lock = getattr(self, "_agent_cache_lock", None) + if cache_lock is not None: + with cache_lock: + cached = self._agent_cache.get(quick_key) + agent = cached[0] if isinstance(cached, tuple) else cached if cached else None + if agent is None: + return "Nothing to refine yet — send a message first." + + snapshot = list(getattr(agent, "_session_messages", None) or []) + if not snapshot: + return "Nothing to refine yet — the conversation is empty." + + review_skills = "skill_manage" in getattr(agent, "valid_tool_names", set()) + try: + agent._spawn_background_review( + messages_snapshot=snapshot, + review_memory=True, + review_skills=review_skills, + focus=args or None, + ) + except Exception as exc: + return f"/refine failed to start: {exc}" + tail = f" (focus: {args})" if args else "" + return ( + f"⚗ Reviewing this conversation in the background{tail} — " + f"any memory/skill updates will be reported when done." + ) + async def _handle_subgoal_command(self, event: "MessageEvent") -> str: """Handle /subgoal for gateway platforms (mirror of CLI handler). diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index f549e655bd..50bb4f7345 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -2453,6 +2453,47 @@ class CLICommandsMixin: f"durable schedules.{_RST}" ) + def _handle_refine_command(self, cmd: str) -> None: + """Dispatch /refine — run the memory/skill review fork on demand. + + Same machinery as the automatic post-turn self-improvement loop + (``AIAgent._spawn_background_review``), but user-triggered and with + optional focus instructions. Writes go to the memory + skill stores + in a background fork; the live conversation and prompt cache are + never touched. + """ + from cli import _DIM, _RST, _cprint + + parts = (cmd or "").strip().split(None, 1) + focus = parts[1].strip() if len(parts) > 1 else "" + + agent = getattr(self, "agent", None) + if agent is None: + _cprint(f" {_DIM}Nothing to refine yet — send a message first.{_RST}") + return + + snapshot = list(getattr(self, "conversation_history", None) or []) + if not snapshot: + _cprint(f" {_DIM}Nothing to refine yet — the conversation is empty.{_RST}") + return + + review_skills = "skill_manage" in getattr(agent, "valid_tool_names", set()) + try: + agent._spawn_background_review( + messages_snapshot=snapshot, + review_memory=True, + review_skills=review_skills, + focus=focus or None, + ) + except Exception as exc: + _cprint(f" /refine failed to start: {exc}") + return + tail = f" (focus: {focus})" if focus else "" + _cprint( + f" ⚗ Reviewing this conversation in the background{tail} — " + f"any memory/skill updates will be reported when done." + ) + def _handle_goal_command(self, cmd: str) -> None: """Dispatch /goal subcommands: set / draft / show / gate / status / pause / resume / clear.""" from cli import _DIM, _RST, _cprint diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index df2c53ffa5..f3323f020a 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -163,6 +163,8 @@ COMMAND_REGISTRY: list[CommandDef] = [ aliases=("hb",), args_hint="[every | status | pause | resume | clear]", subcommands=("status", "pause", "resume", "clear"), busy_policy="dispatch"), + CommandDef("refine", "Review this conversation now and save lessons to memory/skills", "Session", + args_hint="[focus instructions]"), CommandDef("moa", "Run one prompt through the default Mixture of Agents preset, then restore your model", "Session", args_hint="", busy_policy="reject", busy_handler="moa"), CommandDef("subgoal", "Add or manage extra criteria on the active goal", "Session", @@ -1267,7 +1269,10 @@ _SLACK_PRIORITY_ALIASES = ("btw", "bg") # and silently clamps /update off, breaking Telegram parity. # - heartbeat: session heartbeat management; reached via /hermes heartbeat # on Slack. Added at the 50-cap — a native slot would clamp /insights. -_SLACK_VIA_HERMES_ONLY = frozenset({"topup", "moa", "debug", "egress", "init", "version", "diff", "update", "heartbeat"}) +# - refine: on-demand memory/skill review; reached via /hermes refine on +# Slack. Added at the 50-cap — a native slot would clamp an existing +# native slash. +_SLACK_VIA_HERMES_ONLY = frozenset({"topup", "moa", "debug", "egress", "init", "version", "diff", "update", "heartbeat", "refine"}) def _sanitize_slack_name(raw: str) -> str: diff --git a/run_agent.py b/run_agent.py index 92942116e9..67cb22a71e 100644 --- a/run_agent.py +++ b/run_agent.py @@ -1586,8 +1586,6 @@ class AIAgent: ) -> bool: """Return True when this provider/model pair should use Responses API.""" normalized_provider = (provider or "").strip().lower() - if normalized_provider == "actual": - return True # Nous serves GPT-5.x models via its OpenAI-compatible chat # completions endpoint; its /v1/responses endpoint returns 404. if normalized_provider == "nous": @@ -1798,6 +1796,7 @@ class AIAgent: messages_snapshot: List[Dict], review_memory: bool = False, review_skills: bool = False, + focus: Optional[str] = None, ) -> None: """Spawn the background memory/skill review thread. @@ -1806,6 +1805,10 @@ class AIAgent: returns the thread target. ``threading.Thread`` is constructed here so existing tests that patch ``run_agent.threading.Thread`` keep working. + + ``focus`` is optional user-supplied steering (from ``/refine``) + appended to the review prompt — e.g. "save the deploy workflow as a + skill". The automatic post-turn triggers never set it. """ from agent.background_review import spawn_background_review_thread from tools.thread_context import propagate_context_to_thread @@ -1814,6 +1817,7 @@ class AIAgent: messages_snapshot, review_memory=review_memory, review_skills=review_skills, + focus=focus, ) # Carry the active profile into the review thread so MEMORY.md / skill # review writes land in the right profile (#54937). diff --git a/tests/agent/test_refine_focus.py b/tests/agent/test_refine_focus.py new file mode 100644 index 0000000000..49c7ff52df --- /dev/null +++ b/tests/agent/test_refine_focus.py @@ -0,0 +1,56 @@ +"""Tests for the /refine focus parameter on spawn_background_review_thread.""" + +from unittest.mock import MagicMock + +from agent.background_review import ( + _COMBINED_REVIEW_PROMPT, + _MEMORY_REVIEW_PROMPT, + spawn_background_review_thread, +) + + +def _bare_agent(): + agent = MagicMock() + # Ensure getattr(agent, "_COMBINED_REVIEW_PROMPT", default) hits the default. + del agent._COMBINED_REVIEW_PROMPT + del agent._MEMORY_REVIEW_PROMPT + del agent._SKILL_REVIEW_PROMPT + return agent + + +def test_no_focus_prompt_is_byte_identical(): + agent = _bare_agent() + _target, prompt = spawn_background_review_thread( + agent, [], review_memory=True, review_skills=True + ) + assert prompt == _COMBINED_REVIEW_PROMPT + + _target, prompt = spawn_background_review_thread( + agent, [], review_memory=True, review_skills=True, focus=None + ) + assert prompt == _COMBINED_REVIEW_PROMPT + + _target, prompt = spawn_background_review_thread( + agent, [], review_memory=True, review_skills=True, focus=" " + ) + assert prompt == _COMBINED_REVIEW_PROMPT + + +def test_focus_is_appended_to_prompt(): + agent = _bare_agent() + _target, prompt = spawn_background_review_thread( + agent, [], review_memory=True, review_skills=True, + focus="save the deploy workflow as a skill", + ) + assert prompt.startswith(_COMBINED_REVIEW_PROMPT) + assert "save the deploy workflow as a skill" in prompt + assert "explicitly requested" in prompt + + +def test_focus_works_with_memory_only_prompt(): + agent = _bare_agent() + _target, prompt = spawn_background_review_thread( + agent, [], review_memory=True, review_skills=False, focus="remember my timezone", + ) + assert prompt.startswith(_MEMORY_REVIEW_PROMPT) + assert "remember my timezone" in prompt diff --git a/website/docs/reference/slash-commands.md b/website/docs/reference/slash-commands.md index 4502a7982d..c5f646fa6a 100644 --- a/website/docs/reference/slash-commands.md +++ b/website/docs/reference/slash-commands.md @@ -54,6 +54,7 @@ Type `/` in the CLI to open the autocomplete menu. Built-in commands are case-in | `/goal ` | Set a standing goal Hermes works toward across turns — our take on the Ralph loop. After each turn an auxiliary judge model decides whether the goal is done; if not, Hermes auto-continues. Subcommands: `/goal status`, `/goal pause`, `/goal resume`, `/goal clear`. Budget defaults to 20 turns (`goals.max_turns`); any real user message preempts the continuation loop, and state survives `/resume`. See [Persistent Goals](/user-guide/features/goals) for the full walkthrough. | | `/subgoal ` | Append a user-supplied criterion to the active goal mid-loop. The continuation prompt surfaces all subgoals to the agent verbatim, and the judge factors them into its DONE/CONTINUE verdict — so the goal isn't marked done until the original goal **and** every subgoal are met. Subcommands: `/subgoal` (list), `/subgoal remove `, `/subgoal clear`. Requires an active `/goal`. | | `/heartbeat every ` (alias: `/hb`) | Set a recurring prompt that re-enters **this session** as a normal user turn whenever it's idle and the interval has elapsed (min 60s; missed ticks coalesce). Subcommands: `/heartbeat status`, `/heartbeat pause`, `/heartbeat resume`, `/heartbeat clear`. Session-scoped and in-process — use `hermes cron` for durable isolated schedules. See [Session Heartbeats](/user-guide/features/heartbeat). | +| `/refine [focus]` | Run the background memory/skill self-improvement review **now** instead of waiting for the automatic post-turn trigger. Optional focus text steers the review (e.g. `/refine save the deploy workflow as a skill`). Runs in a background fork against a conversation snapshot — the live session and prompt cache are untouched; results are reported when done. | | `/moa ` | Run a single prompt through the default [Mixture of Agents](/user-guide/features/mixture-of-agents) preset, then restore your current model. One-shot — does not change your session model. | | `/resume [name]` | Resume a previously-named session | | `/sessions` (TUI alias: `/switch`) | Classic CLI: browse and resume previous sessions in an interactive picker. TUI: open the live session switcher for currently open TUI sessions. Use `/sessions new` in the TUI to start another live session immediately. | @@ -251,6 +252,7 @@ The messaging gateway supports the following built-in commands inside Telegram, | `/goal ` | Set a standing goal Hermes works toward across turns — our take on the Ralph loop. A judge model checks after each turn; if not done, Hermes auto-continues until it is, you pause/clear it, or the turn budget (default 20) is hit. Subcommands: `/goal status`, `/goal pause`, `/goal resume`, `/goal clear`. Safe to run mid-agent for status/pause/clear; setting a new goal requires `/stop` first. See [Persistent Goals](/user-guide/features/goals). | | `/subgoal ` | Append criteria to the active `/goal` mid-loop (`/subgoal`, `/subgoal remove `, `/subgoal clear`). | | `/heartbeat every ` (alias: `/hb`) | Set a recurring prompt that re-enters this session when idle. Subcommands: `status`, `pause`, `resume`, `clear`. On Slack use `/hermes heartbeat …`. | +| `/refine [focus]` | Run the memory/skill self-improvement review now, optionally with focus instructions. On Slack use `/hermes refine …`. | | `/moa ` | Run one prompt through the default [Mixture of Agents](/user-guide/features/mixture-of-agents) preset, then restore the session model. | | `/branch [name]` (alias: `/fork`) | Branch the current session (explore a different path). | | `/agents` (alias: `/tasks`) | Show active agents and running tasks. |