fix(background-review): reject unresolved failures as skills

This commit is contained in:
Zeraphim
2026-07-27 08:03:44 +08:00
committed by Teknium
parent 3b1dfca207
commit 2ace68ad37
2 changed files with 40 additions and 0 deletions
+18
View File
@@ -284,6 +284,15 @@ _SKILL_REVIEW_PROMPT = (
" • One-off task narratives. A user asking 'summarize today's "
"market' or 'analyze this PR' is not a class of work that warrants "
"a skill.\n\n"
" • Unresolved failures: if the session ended WITHOUT actually "
"finding a working method — you tried several things, none worked, "
"and told the user to check manually — do NOT write those attempts "
"up as a 'reliable workflow' or 'recommended approach'. That presents "
"an untested sequence of failures as validated guidance a future "
"session will trust and repeat. Either say 'Nothing to save', or, "
"only if you are independently confident of a real working alternative "
"(not something you are merely guessing might work), capture ONLY that "
"alternative — never the dead ends, and never dressed up as best practice.\n\n"
"If a tool failed because of setup state, capture the FIX (install "
"command, config step, env var to set) under an existing setup or "
"troubleshooting skill — never 'this tool does not work' as a "
@@ -377,6 +386,15 @@ _COMBINED_REVIEW_PROMPT = (
" • One-off task narratives. A user asking 'summarize today's "
"market' or 'analyze this PR' is not a class of work that warrants "
"a skill.\n\n"
" • Unresolved failures: if the session ended WITHOUT actually "
"finding a working method — you tried several things, none worked, "
"and told the user to check manually — do NOT write those attempts "
"up as a 'reliable workflow' or 'recommended approach'. That presents "
"an untested sequence of failures as validated guidance a future "
"session will trust and repeat. Either say 'Nothing to save', or, "
"only if you are independently confident of a real working alternative "
"(not something you are merely guessing might work), capture ONLY that "
"alternative — never the dead ends, and never dressed up as best practice.\n\n"
"If a tool failed because of setup state, capture the FIX (install "
"command, config step, env var to set) under an existing setup or "
"troubleshooting skill — never 'this tool does not work' as a "
@@ -88,6 +88,28 @@ def test_combined_review_prompt_has_memory_section():
def _assert_unresolved_failure_guidance(prompt: str, label: str) -> None:
"""Unresolved task attempts must not become persistent skill guidance."""
lower = prompt.lower()
assert "unresolved failures" in lower, f"{label}: must identify unresolved failures"
assert "working method" in lower, f"{label}: must require a working method"
assert "told the user to check manually" in lower, (
f"{label}: must recognize an explicitly unresolved session"
)
assert "never the dead ends" in lower, f"{label}: must exclude failed attempts"
assert "independently confident" in lower, (
f"{label}: must limit exceptions to verified alternatives"
)
def test_skill_review_prompt_rejects_unresolved_failures():
_assert_unresolved_failure_guidance(AIAgent._SKILL_REVIEW_PROMPT, "_SKILL_REVIEW_PROMPT")
def test_combined_review_prompt_rejects_unresolved_failures():
_assert_unresolved_failure_guidance(AIAgent._COMBINED_REVIEW_PROMPT, "_COMBINED_REVIEW_PROMPT")
# ---------------------------------------------------------------------------
# Anti-pattern guidance — see issue #6051. The reviewer was learning transient
# environment failures (e.g. "browser tools do not work" from a fresh-install