diff --git a/agent/background_review.py b/agent/background_review.py index bc58ac59d9..ed6df7e3ed 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -284,6 +284,15 @@ _SKILL_REVIEW_PROMPT = ( " • One-off task narratives. A user asking 'summarize today's " "market' or 'analyze this PR' is not a class of work that warrants " "a skill.\n\n" + " • Unresolved failures: if the session ended WITHOUT actually " + "finding a working method — you tried several things, none worked, " + "and told the user to check manually — do NOT write those attempts " + "up as a 'reliable workflow' or 'recommended approach'. That presents " + "an untested sequence of failures as validated guidance a future " + "session will trust and repeat. Either say 'Nothing to save', or, " + "only if you are independently confident of a real working alternative " + "(not something you are merely guessing might work), capture ONLY that " + "alternative — never the dead ends, and never dressed up as best practice.\n\n" "If a tool failed because of setup state, capture the FIX (install " "command, config step, env var to set) under an existing setup or " "troubleshooting skill — never 'this tool does not work' as a " @@ -377,6 +386,15 @@ _COMBINED_REVIEW_PROMPT = ( " • One-off task narratives. A user asking 'summarize today's " "market' or 'analyze this PR' is not a class of work that warrants " "a skill.\n\n" + " • Unresolved failures: if the session ended WITHOUT actually " + "finding a working method — you tried several things, none worked, " + "and told the user to check manually — do NOT write those attempts " + "up as a 'reliable workflow' or 'recommended approach'. That presents " + "an untested sequence of failures as validated guidance a future " + "session will trust and repeat. Either say 'Nothing to save', or, " + "only if you are independently confident of a real working alternative " + "(not something you are merely guessing might work), capture ONLY that " + "alternative — never the dead ends, and never dressed up as best practice.\n\n" "If a tool failed because of setup state, capture the FIX (install " "command, config step, env var to set) under an existing setup or " "troubleshooting skill — never 'this tool does not work' as a " diff --git a/tests/run_agent/test_review_prompt_class_first.py b/tests/run_agent/test_review_prompt_class_first.py index 8bde525a86..bece747838 100644 --- a/tests/run_agent/test_review_prompt_class_first.py +++ b/tests/run_agent/test_review_prompt_class_first.py @@ -88,6 +88,28 @@ def test_combined_review_prompt_has_memory_section(): +def _assert_unresolved_failure_guidance(prompt: str, label: str) -> None: + """Unresolved task attempts must not become persistent skill guidance.""" + lower = prompt.lower() + assert "unresolved failures" in lower, f"{label}: must identify unresolved failures" + assert "working method" in lower, f"{label}: must require a working method" + assert "told the user to check manually" in lower, ( + f"{label}: must recognize an explicitly unresolved session" + ) + assert "never the dead ends" in lower, f"{label}: must exclude failed attempts" + assert "independently confident" in lower, ( + f"{label}: must limit exceptions to verified alternatives" + ) + + +def test_skill_review_prompt_rejects_unresolved_failures(): + _assert_unresolved_failure_guidance(AIAgent._SKILL_REVIEW_PROMPT, "_SKILL_REVIEW_PROMPT") + + +def test_combined_review_prompt_rejects_unresolved_failures(): + _assert_unresolved_failure_guidance(AIAgent._COMBINED_REVIEW_PROMPT, "_COMBINED_REVIEW_PROMPT") + + # --------------------------------------------------------------------------- # Anti-pattern guidance — see issue #6051. The reviewer was learning transient # environment failures (e.g. "browser tools do not work" from a fresh-install