diff --git a/agent/agent_init.py b/agent/agent_init.py index 20e44607bb..60683fcafc 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1896,6 +1896,12 @@ def init_agent( _agent_section = {} agent._tool_use_enforcement = _agent_section.get("tool_use_enforcement", "auto") + # Execution-discipline guidance gate: "auto" (default — matches + # EXECUTION_GUIDANCE_MODELS), true (always), false (never), or list of + # model-name substrings. Independent of tool_use_enforcement — see + # agent/system_prompt.py for the injection gate. + agent._execution_guidance = _agent_section.get("execution_guidance", "auto") + # Empty-response retry guard config (NS-503): additive # ``agent.empty_response_guard`` subsection. Resolution is tolerant — # a malformed section falls back to the schema defaults (guard on, diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 998e5606a0..ed1533748b 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -358,6 +358,25 @@ TOOL_USE_ENFORCEMENT_GUIDANCE = ( # Add new patterns here when a model family needs explicit steering. TOOL_USE_ENFORCEMENT_MODELS = ("gpt", "codex", "gemini", "gemma", "grok", "glm", "qwen", "deepseek") +# Model name substrings whose sessions receive OPENAI_MODEL_EXECUTION_GUIDANCE +# (execution discipline: tool persistence, mandatory tool use for arithmetic, +# external-write read-back, count reconciliation, literal preservation, +# verification-gated completion) when agent.execution_guidance is "auto". +# +# gpt/codex/grok are the historical set; deepseek/kimi/qwen/glm/minimax/ +# mimo/mistral were added after Composio agentic-eval traces showed the same +# failure modes on those families (financial math in prose, no read-back after +# external writes, identifier "repair", completeness claims despite count +# mismatches). GLM's tool-calls-as-plain-text stall (#53847) and MiMo (#41874) +# are covered here too. Gemini/Gemma are excluded — they get the more specific +# GOOGLE_MODEL_OPERATIONAL_GUIDANCE block instead. Claude is excluded because +# it does not exhibit these failure modes; users can opt any model in via +# config.yaml `agent.execution_guidance: true` or a substring list. +EXECUTION_GUIDANCE_MODELS = ( + "gpt", "codex", "grok", + "deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral", +) + # Universal "finish the job" guidance — applied to ALL models, not gated # by model family. Addresses two cross-model failure modes: # 1. Stopping after a stub: writing a tiny file or running one command @@ -438,13 +457,22 @@ PARALLEL_TOOL_CALL_GUIDANCE = ( # without tool calls, suggests workarounds instead of using existing tools, # replies with plans/suggestions instead of executing). The body is # family-agnostic; the OPENAI_ prefix reflects origin, not exclusivity. +# +# As of the Composio agentic-eval follow-up, the block is no longer fenced to +# gpt/codex/grok: eval traces showed DeepSeek/Kimi doing financial math in +# prose, skipping read-back verification after external writes, "repairing" +# malformed identifiers, and claiming completeness despite count mismatches — +# exactly the failure modes this block targets. The injection gate lives in +# agent/system_prompt.py and is controlled by config.yaml +# ``agent.execution_guidance`` (auto/true/false/list); "auto" matches the +# EXECUTION_GUIDANCE_MODELS substring tuple below. OPENAI_MODEL_EXECUTION_GUIDANCE = ( "# Execution discipline\n" "\n" "- Use tools whenever they improve correctness, completeness, or grounding.\n" "- Do not stop early when another tool call would materially improve the result.\n" - "- If a tool returns empty or partial results, retry with a different query or " - "strategy before giving up.\n" + "- If a tool returns empty, partial, or suspiciously narrow results, retry " + "with a broader or different query or strategy before concluding.\n" "- Keep calling tools until: (1) the task is complete, AND (2) you have verified " "the result.\n" "\n" @@ -487,8 +515,30 @@ OPENAI_MODEL_EXECUTION_GUIDANCE = ( "- Formatting: does the output match the requested format or schema?\n" "- Safety: if the next step has side effects (file writes, commands, API calls), " "confirm scope before executing.\n" + "- Completion: 'done' means every named acceptance criterion is verified — " + "never a plausible subset. Completing your plan is not itself the answer; " + "the requested output must appear in your response.\n" "\n" "\n" + "\n" + "- After any state-changing write to an external system (API call, message " + "post, record update), verify the effect by reading back the exact target " + "before claiming success — a successful tool call is not a successful task. " + "Do NOT re-verify internal file edits a tool already confirmed.\n" + "- Declared totals in responses (total, reply_count, has_more, '...N more') " + "are hard assertions. If your enumerated count disagrees, re-fetch or parse " + "programmatically — never finalize on 'go with what I have'.\n" + "- When building write payloads, set fields explicitly rather than relying " + "on provider defaults that could contradict intent.\n" + "\n" + "\n" + "\n" + "- Preserve identifiers, commands, and values exactly as given — never " + "'repair' or normalize a token that fails a stated format. A successful " + "lookup does not validate a malformed source token; validate format first, " + "then look up.\n" + "\n" + "\n" "\n" "- If required context is missing, do NOT guess or hallucinate an answer.\n" "- Use the appropriate lookup tool when missing information is retrievable " diff --git a/agent/system_prompt.py b/agent/system_prompt.py index 2e2579a79c..7d5bba76b3 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -33,6 +33,7 @@ from typing import Any, Dict, List, Optional from agent.prompt_builder import ( DEFAULT_AGENT_IDENTITY, + EXECUTION_GUIDANCE_MODELS, GOOGLE_MODEL_OPERATIONAL_GUIDANCE, HERMES_AGENT_HELP_GUIDANCE, KANBAN_GUIDANCE, @@ -477,13 +478,36 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) # paths, parallel tool calls, verify-before-edit, etc.) if "gemini" in _model_lower or "gemma" in _model_lower: stable_parts.append(GOOGLE_MODEL_OPERATIONAL_GUIDANCE) - # OpenAI GPT/Codex execution discipline (tool persistence, - # prerequisite checks, verification, anti-hallucination). - # Also applied to xAI Grok — same failure modes (claims completion - # without tool calls, suggests workarounds instead of using - # existing tools, replies with plans instead of executing). - if "gpt" in _model_lower or "codex" in _model_lower or "grok" in _model_lower: - stable_parts.append(OPENAI_MODEL_EXECUTION_GUIDANCE) + + # Execution-discipline guidance (tool persistence, mandatory tool use + # for arithmetic, external-write read-back, count reconciliation, + # literal preservation, verification-gated completion). Historically + # nested inside the tool-use-enforcement branch and fenced to + # gpt/codex/grok; now an independent gate so DeepSeek/Kimi/Qwen-class + # models receive it even when tool_use_enforcement is off. Controlled + # by config.yaml agent.execution_guidance: + # "auto" (default) — matches EXECUTION_GUIDANCE_MODELS + # true — always inject (all models) + # false — never inject + # list — custom model-name substrings to match + # Resolved once at session start keyed on the (fixed) model name, so + # the system prompt stays byte-stable for the life of the conversation. + if agent.valid_tool_names: + _exec_guidance = getattr(agent, "_execution_guidance", "auto") + _exec_inject = False + if _exec_guidance is True or (isinstance(_exec_guidance, str) and _exec_guidance.lower() in {"true", "always", "yes", "on"}): + _exec_inject = True + elif _exec_guidance is False or (isinstance(_exec_guidance, str) and _exec_guidance.lower() in {"false", "never", "no", "off"}): + _exec_inject = False + elif isinstance(_exec_guidance, list): + model_lower = (agent.model or "").lower() + _exec_inject = any(p.lower() in model_lower for p in _exec_guidance if isinstance(p, str)) + else: + # "auto" or any unrecognised value — use hardcoded defaults + model_lower = (agent.model or "").lower() + _exec_inject = any(p in model_lower for p in EXECUTION_GUIDANCE_MODELS) + if _exec_inject: + stable_parts.append(OPENAI_MODEL_EXECUTION_GUIDANCE) has_skills_tools = any(name in agent.valid_tool_names for name in ['skills_list', 'skill_view', 'skill_manage']) if has_skills_tools: diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index c1dd275401..5a6173d6d2 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -146,6 +146,15 @@ DEFAULT_CONFIG = { # (force on/off for all models), or a list of model-name substrings # to match (e.g. ["gpt", "codex", "gemini", "qwen"]). "tool_use_enforcement": "auto", + # Execution-discipline guidance: injects a system prompt block covering + # tool persistence, mandatory tool use for arithmetic/system facts, + # external-write read-back, count reconciliation, literal preservation + # of identifiers, and verification-gated completion. Chosen once at + # session start keyed on model name (prompt stays byte-stable). + # Values: "auto" (default — applies to gpt/codex/grok/deepseek/kimi/ + # qwen/glm/minimax/mimo/mistral models), true/false (force on/off for + # all models), or a list of model-name substrings to match. + "execution_guidance": "auto", # Intent-ack continuation: when the model opens a turn by narrating an # action it will take ("I'll go check the logs...") but emits no tool # call, intercept the turn-end, inject a "continue now, execute the diff --git a/hermes_cli/dump.py b/hermes_cli/dump.py index a8d3e992bd..e29675ce13 100644 --- a/hermes_cli/dump.py +++ b/hermes_cli/dump.py @@ -238,6 +238,7 @@ def _config_overrides(config: dict) -> dict[str, str]: ("agent", "gateway_timeout"), ("agent", "session_stall_timeout"), ("agent", "tool_use_enforcement"), + ("agent", "execution_guidance"), ("terminal", "backend"), ("terminal", "docker_image"), ("terminal", "persistent_shell"), diff --git a/tests/agent/test_prompt_builder.py b/tests/agent/test_prompt_builder.py index 38bc36efba..4d94272023 100644 --- a/tests/agent/test_prompt_builder.py +++ b/tests/agent/test_prompt_builder.py @@ -986,6 +986,51 @@ class TestOpenAIModelExecutionGuidance: assert isinstance(OPENAI_MODEL_EXECUTION_GUIDANCE, str) assert len(OPENAI_MODEL_EXECUTION_GUIDANCE) > 100 + def test_guidance_covers_external_write_readback(self): + text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower() + assert "read" in text and "back" in text + assert "successful tool call is not a successful task" in text + + def test_guidance_covers_count_reconciliation(self): + text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower() + assert "has_more" in text + assert "hard assertions" in text + + def test_guidance_covers_literal_preservation(self): + text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower() + assert "normalize" in text + assert "malformed" in text + + def test_guidance_covers_retry_differently(self): + text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower() + assert "suspiciously narrow" in text + assert "retry" in text + + def test_guidance_gates_completion_on_verification(self): + text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower() + assert "plausible subset" in text + + +class TestExecutionGuidanceModels: + """Behavior contracts for the default auto-match model list.""" + + def test_includes_historical_families(self): + from agent.prompt_builder import EXECUTION_GUIDANCE_MODELS + for fam in ("gpt", "codex", "grok"): + assert fam in EXECUTION_GUIDANCE_MODELS + + def test_includes_composio_eval_families(self): + from agent.prompt_builder import EXECUTION_GUIDANCE_MODELS + for fam in ("deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral"): + assert fam in EXECUTION_GUIDANCE_MODELS + + def test_excludes_google_and_claude(self): + # Gemini/Gemma get GOOGLE_MODEL_OPERATIONAL_GUIDANCE instead; + # Claude doesn't exhibit the targeted failure modes. + from agent.prompt_builder import EXECUTION_GUIDANCE_MODELS + for fam in ("gemini", "gemma", "claude"): + assert fam not in EXECUTION_GUIDANCE_MODELS + class TestParallelToolCallGuidance: """Behavior contracts for the universal parallel-tool-call guidance block. diff --git a/tests/agent/test_system_prompt.py b/tests/agent/test_system_prompt.py index aac357bd75..56dc09f1a1 100644 --- a/tests/agent/test_system_prompt.py +++ b/tests/agent/test_system_prompt.py @@ -116,6 +116,87 @@ class TestCodingContextBlock: assert "coding agent" not in _stable_prompt(agent) +class TestExecutionGuidanceInjection: + """Injection gate for OPENAI_MODEL_EXECUTION_GUIDANCE via + ``agent.execution_guidance`` (auto/true/false/list). + + Background — Composio agentic-eval traces (2026-08): the block was + historically fenced to gpt/codex/grok AND nested inside the + tool-use-enforcement branch, so DeepSeek/Kimi/Qwen-class models + received no execution discipline at all. The gate is now independent + of tool_use_enforcement and defaults to a broader family list. + """ + + def _prompt(self, model, execution_guidance="auto", *, + tool_use_enforcement=False, + valid_tool_names=("terminal", "read_file")): + agent = _make_agent( + valid_tool_names=list(valid_tool_names), + model=model, + _tool_use_enforcement=tool_use_enforcement, + _execution_guidance=execution_guidance, + ) + return _stable_prompt(agent) + + def test_deepseek_gets_guidance_by_default(self): + stable = self._prompt("deepseek/deepseek-v4-pro") + assert "Execution discipline" in stable + assert "" in stable + + def test_kimi_gets_guidance_by_default(self): + assert "Execution discipline" in self._prompt("moonshotai/kimi-k3") + + def test_qwen_glm_minimax_mimo_mistral_get_guidance_by_default(self): + for model in ("qwen/qwen-3-max", "z-ai/glm-5.2", + "minimax/minimax-m2", "xiaomi/mimo-v2", + "mistralai/mistral-large-3"): + assert "Execution discipline" in self._prompt(model), model + + def test_gpt_still_gets_guidance(self): + assert "Execution discipline" in self._prompt("openai/gpt-5.5") + + def test_grok_still_gets_guidance(self): + assert "Execution discipline" in self._prompt("xai/grok-4") + + def test_independent_of_tool_use_enforcement(self): + # The gate must not require tool-use enforcement to be on. + stable = self._prompt("deepseek/deepseek-v4-flash", + tool_use_enforcement=False) + assert "Execution discipline" in stable + assert "Tool-use enforcement" not in stable + + def test_claude_does_not_get_guidance_by_default(self): + assert "Execution discipline" not in self._prompt( + "anthropic/claude-opus-4.8") + + def test_gemini_does_not_get_guidance_by_default(self): + assert "Execution discipline" not in self._prompt( + "google/gemini-2.5-pro") + + def test_config_false_suppresses(self): + assert "Execution discipline" not in self._prompt( + "openai/gpt-5.5", execution_guidance=False) + assert "Execution discipline" not in self._prompt( + "deepseek/deepseek-v4-pro", execution_guidance="off") + + def test_config_true_forces_for_any_model(self): + assert "Execution discipline" in self._prompt( + "anthropic/claude-opus-4.8", execution_guidance=True) + + def test_config_list_matches_substring(self): + stable = self._prompt("mycorp/custom-llm-7b", + execution_guidance=["custom-llm", "gpt"]) + assert "Execution discipline" in stable + + def test_config_list_non_match_suppresses(self): + assert "Execution discipline" not in self._prompt( + "openai/gpt-5.5", execution_guidance=["deepseek"]) + + def test_no_tools_no_guidance(self): + assert "Execution discipline" not in self._prompt( + "deepseek/deepseek-v4-pro", valid_tool_names=()) + + class TestNamedProfileHintIntegration: """The same defect through the REAL resolution chain (#72894). diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 135e37e83f..cd4836c575 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -1092,6 +1092,72 @@ class TestToolUseEnforcementConfig: assert TOOL_USE_ENFORCEMENT_GUIDANCE not in prompt +class TestExecutionGuidanceConfig: + """End-to-end tests for the agent.execution_guidance config option — + from config.yaml through agent_init to the built system prompt.""" + + def _make_agent(self, model="deepseek/deepseek-v4-pro", execution_guidance=None): + agent_cfg = {"tool_use_enforcement": False} + if execution_guidance is not None: + agent_cfg["execution_guidance"] = execution_guidance + with ( + patch( + "run_agent.get_tool_definitions", + return_value=_make_tool_defs("terminal", "web_search"), + ), + patch("run_agent.check_toolset_requirements", return_value={}), + patch("run_agent.OpenAI"), + patch( + "hermes_cli.config.load_config", + return_value={"agent": agent_cfg}, + ), patch( + "hermes_cli.config.load_config_readonly", + return_value={"agent": agent_cfg}, + ), + ): + a = AIAgent( + model=model, + api_key="test-key-1234567890", + base_url="https://openrouter.ai/api/v1", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + a.client = MagicMock() + return a + + def test_deepseek_gets_guidance_by_default(self): + from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE + agent = self._make_agent(model="deepseek/deepseek-v4-pro") + assert OPENAI_MODEL_EXECUTION_GUIDANCE in agent._build_system_prompt() + + def test_gpt_still_gets_guidance(self): + from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE + agent = self._make_agent(model="openai/gpt-4.1") + assert OPENAI_MODEL_EXECUTION_GUIDANCE in agent._build_system_prompt() + + def test_config_false_suppresses(self): + from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE + agent = self._make_agent( + model="deepseek/deepseek-v4-pro", execution_guidance=False + ) + assert OPENAI_MODEL_EXECUTION_GUIDANCE not in agent._build_system_prompt() + + def test_config_list_matches(self): + from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE + agent = self._make_agent( + model="moonshotai/kimi-k3", execution_guidance=["kimi"] + ) + assert OPENAI_MODEL_EXECUTION_GUIDANCE in agent._build_system_prompt() + + def test_config_list_non_match_suppresses(self): + from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE + agent = self._make_agent( + model="openai/gpt-4.1", execution_guidance=["kimi"] + ) + assert OPENAI_MODEL_EXECUTION_GUIDANCE not in agent._build_system_prompt() + + class TestTaskCompletionGuidance: """Tests for the universal task-completion / no-fabrication guidance (config.yaml ``agent.task_completion_guidance``). diff --git a/tools/todo_tool.py b/tools/todo_tool.py index 1eea334f43..cc80fb92fb 100644 --- a/tools/todo_tool.py +++ b/tools/todo_tool.py @@ -296,6 +296,8 @@ TODO_SCHEMA = { "description": ( "Manage your task list for the current session. Use for complex tasks " "with 3+ steps or when the user provides multiple tasks. " + "For 'all N items' tasks, enumerate every instance as its own checklist " + "item so none are silently dropped. " "Call with no parameters to read the current list.\n\n" "Writing:\n" "- Provide 'todos' array to create/update items\n" @@ -304,7 +306,8 @@ TODO_SCHEMA = { "Each item: {id: string, content: string, " "status: pending|in_progress|completed|cancelled}\n" "List order is priority. Only ONE item in_progress at a time.\n" - "Mark items completed immediately when done. If something fails, " + "Mark an item completed only after the work is verified done, never " + "based on intent. If something fails, " "cancel it and add a revised item.\n\n" "Always returns the full current list." ), diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 0eccf2d5aa..e031cb639e 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1629,13 +1629,11 @@ agent: ### What it injects -When enabled, three layers of guidance may be added to the system prompt: +When enabled, two layers of guidance may be added to the system prompt: 1. **General tool-use enforcement** (all matched models) — instructs the model to make tool calls immediately instead of describing intentions, keep working until the task is complete, and never end a turn with a promise of future action. -2. **OpenAI execution discipline** (GPT, Codex, and Grok models) — additional guidance addressing GPT-specific failure modes: abandoning work on partial results, skipping prerequisite lookups, hallucinating instead of using tools, and declaring "done" without verification. - -3. **Google operational guidance** (Gemini and Gemma models only) — conciseness, absolute paths, parallel tool calls, and verify-before-edit patterns. +2. **Google operational guidance** (Gemini and Gemma models only) — conciseness, absolute paths, parallel tool calls, and verify-before-edit patterns. These are transparent to the user and only affect the system prompt. Models that already use tools reliably (like Claude) don't need this guidance, which is why `"auto"` excludes them. @@ -1648,6 +1646,33 @@ agent: tool_use_enforcement: ["gpt", "codex", "gemini", "grok", "my-custom-model"] ``` +## Execution-Discipline Guidance + +Separately from tool-use enforcement, Hermes injects an **execution-discipline** block for model families that share a set of agentic failure modes observed in eval traces: doing arithmetic in prose instead of code, skipping read-back verification after external writes, "repairing" malformed identifiers, claiming completeness despite count mismatches, and declaring "done" without verifying every acceptance criterion. + +```yaml +agent: + execution_guidance: "auto" # "auto" | true | false | ["model-substring", ...] +``` + +| Value | Behavior | +|-------|----------| +| `"auto"` (default) | Enabled for models matching: `gpt`, `codex`, `grok`, `deepseek`, `kimi`, `qwen`, `glm`, `minimax`, `mimo`, `mistral`. | +| `true` | Always enabled, regardless of model. | +| `false` | Always disabled, regardless of model. | +| `["deepseek", "my-custom-model"]` | Enabled only when the model name contains one of the listed substrings (case-insensitive). | + +The injected block covers: + +- **Tool persistence** — keep calling tools until the task is complete *and* verified; retry empty, partial, or suspiciously narrow lookup results with a broader or different query before concluding. +- **Mandatory tool use** — arithmetic, hashes, dates, system state, and file facts always come from a tool, never from mental computation. +- **External-write read-back** — after any state-changing write to an external system, read back the exact target before claiming success (internal file edits a tool already confirmed are not re-verified). +- **Count reconciliation** — declared totals (`total`, `reply_count`, `has_more`) are hard assertions; on mismatch, re-fetch or parse programmatically. +- **Literal preservation** — never normalize or "repair" identifiers that fail a stated format; a successful lookup does not validate a malformed source token. +- **Verification-gated completion** — "done" means every named acceptance criterion is verified, never a plausible subset. + +The gate is independent of `tool_use_enforcement` — either can be on without the other. The guidance is chosen once at session start keyed on the model name, so the system prompt stays byte-stable (and prompt-cache-friendly) for the life of the conversation. Gemini/Gemma are excluded from the auto list because they receive the more specific Google operational guidance; Claude is excluded because it doesn't exhibit these failure modes — opt any model in with `true` or a substring list. + ## Tool-Loop Guardrails Hermes detects when the agent is stuck in an unproductive tool-calling loop — the same tool call failing repeatedly, the same tool failing over and over, or an idempotent call returning the same result with no progress. By default it injects a **warning** into the tool result so the model self-corrects; it does not hard-stop, since a person watching the CLI/TUI can intervene.