diff --git a/agent/agent_init.py b/agent/agent_init.py
index 20e44607bb..60683fcafc 100644
--- a/agent/agent_init.py
+++ b/agent/agent_init.py
@@ -1896,6 +1896,12 @@ def init_agent(
_agent_section = {}
agent._tool_use_enforcement = _agent_section.get("tool_use_enforcement", "auto")
+ # Execution-discipline guidance gate: "auto" (default — matches
+ # EXECUTION_GUIDANCE_MODELS), true (always), false (never), or list of
+ # model-name substrings. Independent of tool_use_enforcement — see
+ # agent/system_prompt.py for the injection gate.
+ agent._execution_guidance = _agent_section.get("execution_guidance", "auto")
+
# Empty-response retry guard config (NS-503): additive
# ``agent.empty_response_guard`` subsection. Resolution is tolerant —
# a malformed section falls back to the schema defaults (guard on,
diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py
index 998e5606a0..ed1533748b 100644
--- a/agent/prompt_builder.py
+++ b/agent/prompt_builder.py
@@ -358,6 +358,25 @@ TOOL_USE_ENFORCEMENT_GUIDANCE = (
# Add new patterns here when a model family needs explicit steering.
TOOL_USE_ENFORCEMENT_MODELS = ("gpt", "codex", "gemini", "gemma", "grok", "glm", "qwen", "deepseek")
+# Model name substrings whose sessions receive OPENAI_MODEL_EXECUTION_GUIDANCE
+# (execution discipline: tool persistence, mandatory tool use for arithmetic,
+# external-write read-back, count reconciliation, literal preservation,
+# verification-gated completion) when agent.execution_guidance is "auto".
+#
+# gpt/codex/grok are the historical set; deepseek/kimi/qwen/glm/minimax/
+# mimo/mistral were added after Composio agentic-eval traces showed the same
+# failure modes on those families (financial math in prose, no read-back after
+# external writes, identifier "repair", completeness claims despite count
+# mismatches). GLM's tool-calls-as-plain-text stall (#53847) and MiMo (#41874)
+# are covered here too. Gemini/Gemma are excluded — they get the more specific
+# GOOGLE_MODEL_OPERATIONAL_GUIDANCE block instead. Claude is excluded because
+# it does not exhibit these failure modes; users can opt any model in via
+# config.yaml `agent.execution_guidance: true` or a substring list.
+EXECUTION_GUIDANCE_MODELS = (
+ "gpt", "codex", "grok",
+ "deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral",
+)
+
# Universal "finish the job" guidance — applied to ALL models, not gated
# by model family. Addresses two cross-model failure modes:
# 1. Stopping after a stub: writing a tiny file or running one command
@@ -438,13 +457,22 @@ PARALLEL_TOOL_CALL_GUIDANCE = (
# without tool calls, suggests workarounds instead of using existing tools,
# replies with plans/suggestions instead of executing). The body is
# family-agnostic; the OPENAI_ prefix reflects origin, not exclusivity.
+#
+# As of the Composio agentic-eval follow-up, the block is no longer fenced to
+# gpt/codex/grok: eval traces showed DeepSeek/Kimi doing financial math in
+# prose, skipping read-back verification after external writes, "repairing"
+# malformed identifiers, and claiming completeness despite count mismatches —
+# exactly the failure modes this block targets. The injection gate lives in
+# agent/system_prompt.py and is controlled by config.yaml
+# ``agent.execution_guidance`` (auto/true/false/list); "auto" matches the
+# EXECUTION_GUIDANCE_MODELS substring tuple below.
OPENAI_MODEL_EXECUTION_GUIDANCE = (
"# Execution discipline\n"
"\n"
"- Use tools whenever they improve correctness, completeness, or grounding.\n"
"- Do not stop early when another tool call would materially improve the result.\n"
- "- If a tool returns empty or partial results, retry with a different query or "
- "strategy before giving up.\n"
+ "- If a tool returns empty, partial, or suspiciously narrow results, retry "
+ "with a broader or different query or strategy before concluding.\n"
"- Keep calling tools until: (1) the task is complete, AND (2) you have verified "
"the result.\n"
"\n"
@@ -487,8 +515,30 @@ OPENAI_MODEL_EXECUTION_GUIDANCE = (
"- Formatting: does the output match the requested format or schema?\n"
"- Safety: if the next step has side effects (file writes, commands, API calls), "
"confirm scope before executing.\n"
+ "- Completion: 'done' means every named acceptance criterion is verified — "
+ "never a plausible subset. Completing your plan is not itself the answer; "
+ "the requested output must appear in your response.\n"
"\n"
"\n"
+ "\n"
+ "- After any state-changing write to an external system (API call, message "
+ "post, record update), verify the effect by reading back the exact target "
+ "before claiming success — a successful tool call is not a successful task. "
+ "Do NOT re-verify internal file edits a tool already confirmed.\n"
+ "- Declared totals in responses (total, reply_count, has_more, '...N more') "
+ "are hard assertions. If your enumerated count disagrees, re-fetch or parse "
+ "programmatically — never finalize on 'go with what I have'.\n"
+ "- When building write payloads, set fields explicitly rather than relying "
+ "on provider defaults that could contradict intent.\n"
+ "\n"
+ "\n"
+ "\n"
+ "- Preserve identifiers, commands, and values exactly as given — never "
+ "'repair' or normalize a token that fails a stated format. A successful "
+ "lookup does not validate a malformed source token; validate format first, "
+ "then look up.\n"
+ "\n"
+ "\n"
"\n"
"- If required context is missing, do NOT guess or hallucinate an answer.\n"
"- Use the appropriate lookup tool when missing information is retrievable "
diff --git a/agent/system_prompt.py b/agent/system_prompt.py
index 2e2579a79c..7d5bba76b3 100644
--- a/agent/system_prompt.py
+++ b/agent/system_prompt.py
@@ -33,6 +33,7 @@ from typing import Any, Dict, List, Optional
from agent.prompt_builder import (
DEFAULT_AGENT_IDENTITY,
+ EXECUTION_GUIDANCE_MODELS,
GOOGLE_MODEL_OPERATIONAL_GUIDANCE,
HERMES_AGENT_HELP_GUIDANCE,
KANBAN_GUIDANCE,
@@ -477,13 +478,36 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
# paths, parallel tool calls, verify-before-edit, etc.)
if "gemini" in _model_lower or "gemma" in _model_lower:
stable_parts.append(GOOGLE_MODEL_OPERATIONAL_GUIDANCE)
- # OpenAI GPT/Codex execution discipline (tool persistence,
- # prerequisite checks, verification, anti-hallucination).
- # Also applied to xAI Grok — same failure modes (claims completion
- # without tool calls, suggests workarounds instead of using
- # existing tools, replies with plans instead of executing).
- if "gpt" in _model_lower or "codex" in _model_lower or "grok" in _model_lower:
- stable_parts.append(OPENAI_MODEL_EXECUTION_GUIDANCE)
+
+ # Execution-discipline guidance (tool persistence, mandatory tool use
+ # for arithmetic, external-write read-back, count reconciliation,
+ # literal preservation, verification-gated completion). Historically
+ # nested inside the tool-use-enforcement branch and fenced to
+ # gpt/codex/grok; now an independent gate so DeepSeek/Kimi/Qwen-class
+ # models receive it even when tool_use_enforcement is off. Controlled
+ # by config.yaml agent.execution_guidance:
+ # "auto" (default) — matches EXECUTION_GUIDANCE_MODELS
+ # true — always inject (all models)
+ # false — never inject
+ # list — custom model-name substrings to match
+ # Resolved once at session start keyed on the (fixed) model name, so
+ # the system prompt stays byte-stable for the life of the conversation.
+ if agent.valid_tool_names:
+ _exec_guidance = getattr(agent, "_execution_guidance", "auto")
+ _exec_inject = False
+ if _exec_guidance is True or (isinstance(_exec_guidance, str) and _exec_guidance.lower() in {"true", "always", "yes", "on"}):
+ _exec_inject = True
+ elif _exec_guidance is False or (isinstance(_exec_guidance, str) and _exec_guidance.lower() in {"false", "never", "no", "off"}):
+ _exec_inject = False
+ elif isinstance(_exec_guidance, list):
+ model_lower = (agent.model or "").lower()
+ _exec_inject = any(p.lower() in model_lower for p in _exec_guidance if isinstance(p, str))
+ else:
+ # "auto" or any unrecognised value — use hardcoded defaults
+ model_lower = (agent.model or "").lower()
+ _exec_inject = any(p in model_lower for p in EXECUTION_GUIDANCE_MODELS)
+ if _exec_inject:
+ stable_parts.append(OPENAI_MODEL_EXECUTION_GUIDANCE)
has_skills_tools = any(name in agent.valid_tool_names for name in ['skills_list', 'skill_view', 'skill_manage'])
if has_skills_tools:
diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py
index c1dd275401..5a6173d6d2 100644
--- a/hermes_cli/config_defaults.py
+++ b/hermes_cli/config_defaults.py
@@ -146,6 +146,15 @@ DEFAULT_CONFIG = {
# (force on/off for all models), or a list of model-name substrings
# to match (e.g. ["gpt", "codex", "gemini", "qwen"]).
"tool_use_enforcement": "auto",
+ # Execution-discipline guidance: injects a system prompt block covering
+ # tool persistence, mandatory tool use for arithmetic/system facts,
+ # external-write read-back, count reconciliation, literal preservation
+ # of identifiers, and verification-gated completion. Chosen once at
+ # session start keyed on model name (prompt stays byte-stable).
+ # Values: "auto" (default — applies to gpt/codex/grok/deepseek/kimi/
+ # qwen/glm/minimax/mimo/mistral models), true/false (force on/off for
+ # all models), or a list of model-name substrings to match.
+ "execution_guidance": "auto",
# Intent-ack continuation: when the model opens a turn by narrating an
# action it will take ("I'll go check the logs...") but emits no tool
# call, intercept the turn-end, inject a "continue now, execute the
diff --git a/hermes_cli/dump.py b/hermes_cli/dump.py
index a8d3e992bd..e29675ce13 100644
--- a/hermes_cli/dump.py
+++ b/hermes_cli/dump.py
@@ -238,6 +238,7 @@ def _config_overrides(config: dict) -> dict[str, str]:
("agent", "gateway_timeout"),
("agent", "session_stall_timeout"),
("agent", "tool_use_enforcement"),
+ ("agent", "execution_guidance"),
("terminal", "backend"),
("terminal", "docker_image"),
("terminal", "persistent_shell"),
diff --git a/tests/agent/test_prompt_builder.py b/tests/agent/test_prompt_builder.py
index 38bc36efba..4d94272023 100644
--- a/tests/agent/test_prompt_builder.py
+++ b/tests/agent/test_prompt_builder.py
@@ -986,6 +986,51 @@ class TestOpenAIModelExecutionGuidance:
assert isinstance(OPENAI_MODEL_EXECUTION_GUIDANCE, str)
assert len(OPENAI_MODEL_EXECUTION_GUIDANCE) > 100
+ def test_guidance_covers_external_write_readback(self):
+ text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower()
+ assert "read" in text and "back" in text
+ assert "successful tool call is not a successful task" in text
+
+ def test_guidance_covers_count_reconciliation(self):
+ text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower()
+ assert "has_more" in text
+ assert "hard assertions" in text
+
+ def test_guidance_covers_literal_preservation(self):
+ text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower()
+ assert "normalize" in text
+ assert "malformed" in text
+
+ def test_guidance_covers_retry_differently(self):
+ text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower()
+ assert "suspiciously narrow" in text
+ assert "retry" in text
+
+ def test_guidance_gates_completion_on_verification(self):
+ text = OPENAI_MODEL_EXECUTION_GUIDANCE.lower()
+ assert "plausible subset" in text
+
+
+class TestExecutionGuidanceModels:
+ """Behavior contracts for the default auto-match model list."""
+
+ def test_includes_historical_families(self):
+ from agent.prompt_builder import EXECUTION_GUIDANCE_MODELS
+ for fam in ("gpt", "codex", "grok"):
+ assert fam in EXECUTION_GUIDANCE_MODELS
+
+ def test_includes_composio_eval_families(self):
+ from agent.prompt_builder import EXECUTION_GUIDANCE_MODELS
+ for fam in ("deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral"):
+ assert fam in EXECUTION_GUIDANCE_MODELS
+
+ def test_excludes_google_and_claude(self):
+ # Gemini/Gemma get GOOGLE_MODEL_OPERATIONAL_GUIDANCE instead;
+ # Claude doesn't exhibit the targeted failure modes.
+ from agent.prompt_builder import EXECUTION_GUIDANCE_MODELS
+ for fam in ("gemini", "gemma", "claude"):
+ assert fam not in EXECUTION_GUIDANCE_MODELS
+
class TestParallelToolCallGuidance:
"""Behavior contracts for the universal parallel-tool-call guidance block.
diff --git a/tests/agent/test_system_prompt.py b/tests/agent/test_system_prompt.py
index aac357bd75..56dc09f1a1 100644
--- a/tests/agent/test_system_prompt.py
+++ b/tests/agent/test_system_prompt.py
@@ -116,6 +116,87 @@ class TestCodingContextBlock:
assert "coding agent" not in _stable_prompt(agent)
+class TestExecutionGuidanceInjection:
+ """Injection gate for OPENAI_MODEL_EXECUTION_GUIDANCE via
+ ``agent.execution_guidance`` (auto/true/false/list).
+
+ Background — Composio agentic-eval traces (2026-08): the block was
+ historically fenced to gpt/codex/grok AND nested inside the
+ tool-use-enforcement branch, so DeepSeek/Kimi/Qwen-class models
+ received no execution discipline at all. The gate is now independent
+ of tool_use_enforcement and defaults to a broader family list.
+ """
+
+ def _prompt(self, model, execution_guidance="auto", *,
+ tool_use_enforcement=False,
+ valid_tool_names=("terminal", "read_file")):
+ agent = _make_agent(
+ valid_tool_names=list(valid_tool_names),
+ model=model,
+ _tool_use_enforcement=tool_use_enforcement,
+ _execution_guidance=execution_guidance,
+ )
+ return _stable_prompt(agent)
+
+ def test_deepseek_gets_guidance_by_default(self):
+ stable = self._prompt("deepseek/deepseek-v4-pro")
+ assert "Execution discipline" in stable
+ assert "" in stable
+
+ def test_kimi_gets_guidance_by_default(self):
+ assert "Execution discipline" in self._prompt("moonshotai/kimi-k3")
+
+ def test_qwen_glm_minimax_mimo_mistral_get_guidance_by_default(self):
+ for model in ("qwen/qwen-3-max", "z-ai/glm-5.2",
+ "minimax/minimax-m2", "xiaomi/mimo-v2",
+ "mistralai/mistral-large-3"):
+ assert "Execution discipline" in self._prompt(model), model
+
+ def test_gpt_still_gets_guidance(self):
+ assert "Execution discipline" in self._prompt("openai/gpt-5.5")
+
+ def test_grok_still_gets_guidance(self):
+ assert "Execution discipline" in self._prompt("xai/grok-4")
+
+ def test_independent_of_tool_use_enforcement(self):
+ # The gate must not require tool-use enforcement to be on.
+ stable = self._prompt("deepseek/deepseek-v4-flash",
+ tool_use_enforcement=False)
+ assert "Execution discipline" in stable
+ assert "Tool-use enforcement" not in stable
+
+ def test_claude_does_not_get_guidance_by_default(self):
+ assert "Execution discipline" not in self._prompt(
+ "anthropic/claude-opus-4.8")
+
+ def test_gemini_does_not_get_guidance_by_default(self):
+ assert "Execution discipline" not in self._prompt(
+ "google/gemini-2.5-pro")
+
+ def test_config_false_suppresses(self):
+ assert "Execution discipline" not in self._prompt(
+ "openai/gpt-5.5", execution_guidance=False)
+ assert "Execution discipline" not in self._prompt(
+ "deepseek/deepseek-v4-pro", execution_guidance="off")
+
+ def test_config_true_forces_for_any_model(self):
+ assert "Execution discipline" in self._prompt(
+ "anthropic/claude-opus-4.8", execution_guidance=True)
+
+ def test_config_list_matches_substring(self):
+ stable = self._prompt("mycorp/custom-llm-7b",
+ execution_guidance=["custom-llm", "gpt"])
+ assert "Execution discipline" in stable
+
+ def test_config_list_non_match_suppresses(self):
+ assert "Execution discipline" not in self._prompt(
+ "openai/gpt-5.5", execution_guidance=["deepseek"])
+
+ def test_no_tools_no_guidance(self):
+ assert "Execution discipline" not in self._prompt(
+ "deepseek/deepseek-v4-pro", valid_tool_names=())
+
+
class TestNamedProfileHintIntegration:
"""The same defect through the REAL resolution chain (#72894).
diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py
index 135e37e83f..cd4836c575 100644
--- a/tests/run_agent/test_run_agent.py
+++ b/tests/run_agent/test_run_agent.py
@@ -1092,6 +1092,72 @@ class TestToolUseEnforcementConfig:
assert TOOL_USE_ENFORCEMENT_GUIDANCE not in prompt
+class TestExecutionGuidanceConfig:
+ """End-to-end tests for the agent.execution_guidance config option —
+ from config.yaml through agent_init to the built system prompt."""
+
+ def _make_agent(self, model="deepseek/deepseek-v4-pro", execution_guidance=None):
+ agent_cfg = {"tool_use_enforcement": False}
+ if execution_guidance is not None:
+ agent_cfg["execution_guidance"] = execution_guidance
+ with (
+ patch(
+ "run_agent.get_tool_definitions",
+ return_value=_make_tool_defs("terminal", "web_search"),
+ ),
+ patch("run_agent.check_toolset_requirements", return_value={}),
+ patch("run_agent.OpenAI"),
+ patch(
+ "hermes_cli.config.load_config",
+ return_value={"agent": agent_cfg},
+ ), patch(
+ "hermes_cli.config.load_config_readonly",
+ return_value={"agent": agent_cfg},
+ ),
+ ):
+ a = AIAgent(
+ model=model,
+ api_key="test-key-1234567890",
+ base_url="https://openrouter.ai/api/v1",
+ quiet_mode=True,
+ skip_context_files=True,
+ skip_memory=True,
+ )
+ a.client = MagicMock()
+ return a
+
+ def test_deepseek_gets_guidance_by_default(self):
+ from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE
+ agent = self._make_agent(model="deepseek/deepseek-v4-pro")
+ assert OPENAI_MODEL_EXECUTION_GUIDANCE in agent._build_system_prompt()
+
+ def test_gpt_still_gets_guidance(self):
+ from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE
+ agent = self._make_agent(model="openai/gpt-4.1")
+ assert OPENAI_MODEL_EXECUTION_GUIDANCE in agent._build_system_prompt()
+
+ def test_config_false_suppresses(self):
+ from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE
+ agent = self._make_agent(
+ model="deepseek/deepseek-v4-pro", execution_guidance=False
+ )
+ assert OPENAI_MODEL_EXECUTION_GUIDANCE not in agent._build_system_prompt()
+
+ def test_config_list_matches(self):
+ from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE
+ agent = self._make_agent(
+ model="moonshotai/kimi-k3", execution_guidance=["kimi"]
+ )
+ assert OPENAI_MODEL_EXECUTION_GUIDANCE in agent._build_system_prompt()
+
+ def test_config_list_non_match_suppresses(self):
+ from agent.prompt_builder import OPENAI_MODEL_EXECUTION_GUIDANCE
+ agent = self._make_agent(
+ model="openai/gpt-4.1", execution_guidance=["kimi"]
+ )
+ assert OPENAI_MODEL_EXECUTION_GUIDANCE not in agent._build_system_prompt()
+
+
class TestTaskCompletionGuidance:
"""Tests for the universal task-completion / no-fabrication guidance
(config.yaml ``agent.task_completion_guidance``).
diff --git a/tools/todo_tool.py b/tools/todo_tool.py
index 1eea334f43..cc80fb92fb 100644
--- a/tools/todo_tool.py
+++ b/tools/todo_tool.py
@@ -296,6 +296,8 @@ TODO_SCHEMA = {
"description": (
"Manage your task list for the current session. Use for complex tasks "
"with 3+ steps or when the user provides multiple tasks. "
+ "For 'all N items' tasks, enumerate every instance as its own checklist "
+ "item so none are silently dropped. "
"Call with no parameters to read the current list.\n\n"
"Writing:\n"
"- Provide 'todos' array to create/update items\n"
@@ -304,7 +306,8 @@ TODO_SCHEMA = {
"Each item: {id: string, content: string, "
"status: pending|in_progress|completed|cancelled}\n"
"List order is priority. Only ONE item in_progress at a time.\n"
- "Mark items completed immediately when done. If something fails, "
+ "Mark an item completed only after the work is verified done, never "
+ "based on intent. If something fails, "
"cancel it and add a revised item.\n\n"
"Always returns the full current list."
),
diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md
index 0eccf2d5aa..e031cb639e 100644
--- a/website/docs/user-guide/configuration.md
+++ b/website/docs/user-guide/configuration.md
@@ -1629,13 +1629,11 @@ agent:
### What it injects
-When enabled, three layers of guidance may be added to the system prompt:
+When enabled, two layers of guidance may be added to the system prompt:
1. **General tool-use enforcement** (all matched models) — instructs the model to make tool calls immediately instead of describing intentions, keep working until the task is complete, and never end a turn with a promise of future action.
-2. **OpenAI execution discipline** (GPT, Codex, and Grok models) — additional guidance addressing GPT-specific failure modes: abandoning work on partial results, skipping prerequisite lookups, hallucinating instead of using tools, and declaring "done" without verification.
-
-3. **Google operational guidance** (Gemini and Gemma models only) — conciseness, absolute paths, parallel tool calls, and verify-before-edit patterns.
+2. **Google operational guidance** (Gemini and Gemma models only) — conciseness, absolute paths, parallel tool calls, and verify-before-edit patterns.
These are transparent to the user and only affect the system prompt. Models that already use tools reliably (like Claude) don't need this guidance, which is why `"auto"` excludes them.
@@ -1648,6 +1646,33 @@ agent:
tool_use_enforcement: ["gpt", "codex", "gemini", "grok", "my-custom-model"]
```
+## Execution-Discipline Guidance
+
+Separately from tool-use enforcement, Hermes injects an **execution-discipline** block for model families that share a set of agentic failure modes observed in eval traces: doing arithmetic in prose instead of code, skipping read-back verification after external writes, "repairing" malformed identifiers, claiming completeness despite count mismatches, and declaring "done" without verifying every acceptance criterion.
+
+```yaml
+agent:
+ execution_guidance: "auto" # "auto" | true | false | ["model-substring", ...]
+```
+
+| Value | Behavior |
+|-------|----------|
+| `"auto"` (default) | Enabled for models matching: `gpt`, `codex`, `grok`, `deepseek`, `kimi`, `qwen`, `glm`, `minimax`, `mimo`, `mistral`. |
+| `true` | Always enabled, regardless of model. |
+| `false` | Always disabled, regardless of model. |
+| `["deepseek", "my-custom-model"]` | Enabled only when the model name contains one of the listed substrings (case-insensitive). |
+
+The injected block covers:
+
+- **Tool persistence** — keep calling tools until the task is complete *and* verified; retry empty, partial, or suspiciously narrow lookup results with a broader or different query before concluding.
+- **Mandatory tool use** — arithmetic, hashes, dates, system state, and file facts always come from a tool, never from mental computation.
+- **External-write read-back** — after any state-changing write to an external system, read back the exact target before claiming success (internal file edits a tool already confirmed are not re-verified).
+- **Count reconciliation** — declared totals (`total`, `reply_count`, `has_more`) are hard assertions; on mismatch, re-fetch or parse programmatically.
+- **Literal preservation** — never normalize or "repair" identifiers that fail a stated format; a successful lookup does not validate a malformed source token.
+- **Verification-gated completion** — "done" means every named acceptance criterion is verified, never a plausible subset.
+
+The gate is independent of `tool_use_enforcement` — either can be on without the other. The guidance is chosen once at session start keyed on the model name, so the system prompt stays byte-stable (and prompt-cache-friendly) for the life of the conversation. Gemini/Gemma are excluded from the auto list because they receive the more specific Google operational guidance; Claude is excluded because it doesn't exhibit these failure modes — opt any model in with `true` or a substring list.
+
## Tool-Loop Guardrails
Hermes detects when the agent is stuck in an unproductive tool-calling loop — the same tool call failing repeatedly, the same tool failing over and over, or an idempotent call returning the same result with no progress. By default it injects a **warning** into the tool result so the model self-corrects; it does not hard-stop, since a person watching the CLI/TUI can intervene.