fix(loops): pause /loop --until on a blocked verdict; trim redundant gate condition and duplicate test

The goal judge now returns 'blocked' for unachievable goals, but the
/loop --until gate only checked == 'done', so an impossible stop
condition would re-fire every tick until loops.max_ticks. Pause the
loop with the judge's reason instead. Also collapse the kanban gate
callers' 'gate_verdict == "continue" or rejection is not None' to
'rejection is not None' (rejection is None iff verdict == done), drop
the duplicate blocked-verdict goal test, and document the verdict.
This commit is contained in:
Teknium
2026-09-02 04:36:05 -07:00
parent ae63db2304
commit 5c6cbbc1be
8 changed files with 36 additions and 27 deletions
+2 -2
View File
@@ -2398,7 +2398,7 @@ def _cmd_complete(args: argparse.Namespace) -> int:
)
failed.append(tid)
continue
if gate_verdict == "continue" or rejection is not None:
if rejection is not None:
print(
f"kanban: goal completion of {tid} rejected by judge: {rejection}. "
f"Provide evidence matching the task's acceptance criteria.",
@@ -2559,7 +2559,7 @@ def _cmd_request_review(args: argparse.Namespace) -> int:
file=sys.stderr,
)
return 1
if gate_verdict == "continue" or rejection is not None:
if rejection is not None:
print(
f"kanban: goal review handoff of {tid} rejected by judge: "
f"{rejection}. Provide acceptance evidence matching the task.",
+12
View File
@@ -768,6 +768,18 @@ class LoopManager:
"reason": s.last_stop_reason,
"message": f"✓ Loop finished after {s.ticks_fired} tick{'s' if s.ticks_fired != 1 else ''} — {reason}",
}
if verdict == "blocked":
# Judge ruled the stop condition unachievable — don't spin
# until the tick budget; pause so the user can re-scope.
s.status = "paused"
s.paused_reason = f"stop condition judged unachievable: {reason}"
save_loop(self.session_id, s)
return {
"status": "paused",
"stopped": True,
"reason": s.paused_reason,
"message": f"⏸ Loop paused — {s.paused_reason}. /loop resume to keep going, /loop stop to end it.",
}
# 3. --times user cap.
if s.times and s.ticks_fired >= s.times:
-17
View File
@@ -833,20 +833,3 @@ class TestBlockedVerdict:
assert mgr.state is not None
assert mgr.state.status == "paused"
assert "unachievable" in (mgr.state.paused_reason or "").lower()
def test_blocked_verdict_never_records_done(self, hermes_home):
from unittest.mock import patch
from hermes_cli.goals import GoalManager
mgr = GoalManager(session_id="blocked-sid-2")
mgr.set("square the circle")
with patch(
"hermes_cli.goals.judge_goal",
return_value=("blocked", "mathematically impossible", False, None, False),
):
decision = mgr.evaluate_after_turn("This cannot be done.")
assert decision["status"] != "done"
assert mgr.state is not None
assert mgr.state.status != "done"
assert mgr.state.last_verdict == "blocked"
+14
View File
@@ -443,6 +443,20 @@ class TestTickLifecycle:
decision = mgr.complete_tick("3 tests still failing")
assert decision["stopped"] is False
def test_until_judge_blocked_pauses(self, hermes_home):
"""An unachievable stop condition pauses the loop instead of spinning to the tick budget."""
from hermes_cli.loops import LoopManager
mgr = LoopManager(session_id="t11b")
state = mgr.set("poll", interval_seconds=300, until="the deleted repo's CI is green")
state.next_due_at = time.time() - 1
mgr.fire_tick()
with patch("hermes_cli.goals.judge_goal", return_value=("blocked", "repo no longer exists", False, None, False)):
decision = mgr.complete_tick("The repository was deleted; there is no CI to watch.")
assert decision["stopped"] is True
assert decision["status"] == "paused"
assert "unachievable" in decision["message"]
def test_until_judge_error_fails_open(self, hermes_home):
from hermes_cli.loops import LoopManager
+2 -2
View File
@@ -770,7 +770,7 @@ def _handle_complete(args: dict, **kw) -> str:
f"or record the block with kanban_block and hand the "
f"decision to a human / reviewer."
)
if gate_verdict == "continue" or rejection is not None:
if rejection is not None:
return tool_error(
f"Goal completion rejected by judge: {rejection}. "
f"To proceed, either: (1) provide explicit acceptance "
@@ -958,7 +958,7 @@ def _handle_request_review(args: dict, **kw) -> str:
f"unachievable — {rejection}. Record the block with "
f"kanban_block instead of requesting review."
)
if gate_verdict == "continue" or rejection is not None:
if rejection is not None:
return tool_error(
f"Goal review handoff rejected by judge: {rejection}. "
"Provide acceptance evidence matching the card before "
+4 -4
View File
@@ -49,7 +49,7 @@ What you'll see:
1. **Goal accepted** — `⊙ Goal set (20-turn budget): <your goal>`
2. **Turn 1 runs** — Hermes starts working as if you'd sent the goal as a normal message.
3. **Judge runs** — after the turn, the judge model decides `done` or `continue`.
3. **Judge runs** — after the turn, the judge model decides `done`, `continue`, or `blocked`.
4. **Loop fires if needed** — if `continue`, you'll see `↻ Continuing toward goal (1/20): <judge's reason>` and Hermes takes the next step automatically.
5. **Terminates** — eventually you see either `✓ Goal achieved: <reason>` or `⏸ Goal paused — N/20 turns used`.
@@ -140,7 +140,7 @@ A completion contract makes the judge stricter, but the judge is still an LLM re
How it works, each turn:
1. **Gates run before the judge.** If any gate fails, the judge is *not called* — a red gate is deterministic evidence the goal isn't done. The gate's exit code and output tail (last ~3 KB) become the continuation prompt, so the agent iterates against the actual failure instead of a vibe.
2. **All gates pass → normal judging.** The LLM judge then decides done/continue/wait exactly as before.
2. **All gates pass → normal judging.** The LLM judge then decides done/blocked/continue/wait exactly as before.
3. **Unchanged workspace → no re-run.** If a gate failed and nothing changed in the workspace since (tracked via a git fingerprint of HEAD + working-tree status), the gate is not re-run — the recorded failure is replayed and the attempt count advances. A stuck agent can't burn wall-clock re-running an identical red suite. Outside a git repo, gates simply always re-run.
4. **Retries are bounded.** Each gate defaults to 3 retries and a 5-minute timeout. When a gate exhausts its retries the goal auto-pauses (like the turn budget) with a message telling you to fix it manually, remove the gate, or `/goal resume`.
@@ -179,9 +179,9 @@ After every turn, Hermes calls an auxiliary model with:
- The standing goal text
- The agent's most recent final response (last ~4 KB of text)
- A system prompt telling the judge to reply with strict one-line JSON: `{"verdict": "done" | "continue" | "wait", "reason": "<one-sentence rationale>"}` (wait verdicts add `wait_on_session` / `wait_on_pid` / `wait_for_seconds`; the legacy `{"done": <bool>, "reason": "..."}` shape is still accepted)
- A system prompt telling the judge to reply with strict one-line JSON: `{"verdict": "done" | "blocked" | "continue" | "wait", "reason": "<one-sentence rationale>"}` (wait verdicts add `wait_on_session` / `wait_on_pid` / `wait_for_seconds`; the legacy `{"done": <bool>, "reason": "..."}` shape is still accepted)
The judge is deliberately conservative: it marks a goal `done` only when the response **explicitly** confirms the goal is complete, when the final deliverable is clearly produced, or when the goal is unachievable/blocked (treated as DONE with a block reason so we don't burn budget on impossible tasks).
The judge is deliberately conservative: it marks a goal `done` only when the response **explicitly** confirms the goal is complete, when the final deliverable is clearly produced. A goal the agent explains is **unachievable** (impossible, out of scope, needs user input) gets a `blocked` verdict instead — never `done`: the goal **pauses** with the judge's reason (`🚫 Goal judged unachievable — paused`), so you can re-scope it with `/goal <text>` or override with `/goal resume` rather than burning budget or having an impossible task waved through as complete.
### Fail-open semantics
+1 -1
View File
@@ -512,7 +512,7 @@ def register(ctx):
### Goal-mode cards (`--goal`)
By default each worker gets **one shot** at its card — do the work, call `kanban_complete`/`kanban_block`, exit. Pass `--goal` (CLI) or `goal_mode=True` (the `kanban_create` tool / dashboard) to instead run that worker in a **goal loop**, the same Ralph-style engine behind the `/goal` slash command: after every turn an auxiliary judge checks the worker's output against the card's title + body (treated as the acceptance criteria), and if the work isn't done — and the turn budget remains — the worker keeps going **in the same session** until the judge agrees, the worker terminates the task itself, or the budget runs out (which **blocks** the card for human review rather than exiting silently).
By default each worker gets **one shot** at its card — do the work, call `kanban_complete`/`kanban_block`, exit. Pass `--goal` (CLI) or `goal_mode=True` (the `kanban_create` tool / dashboard) to instead run that worker in a **goal loop**, the same Ralph-style engine behind the `/goal` slash command: after every turn an auxiliary judge checks the worker's output against the card's title + body (treated as the acceptance criteria), and if the work isn't done — and the turn budget remains — the worker keeps going **in the same session** until the judge agrees, the worker terminates the task itself, or the budget runs out (which **blocks** the card for human review rather than exiting silently). If the judge rules the goal **unachievable** as written, the card is blocked immediately with the judge's reason — an impossible card is never marked done, and `kanban complete` / `kanban request-review` on such a card are rejected with a pointer to `kanban block` or re-scoping.
```bash
hermes kanban create "Translate the docs site to French" \
+1 -1
View File
@@ -61,7 +61,7 @@ A loop ends when any of these fires:
|---|---|
| The agent decides it's done | The wakeup prompt teaches the agent to end its reply with `LOOP_COMPLETE` on its own line when the task is finished or moot. |
| A run cap | `--times N` — stop after N wakeups. |
| An evidence-based condition | `--until <condition>` — after each wakeup, the same auxiliary judge that powers `/goal` checks the reply against your condition (fail-open: a broken judge never wedges the loop). |
| An evidence-based condition | `--until <condition>` — after each wakeup, the same auxiliary judge that powers `/goal` checks the reply against your condition. If the judge rules the condition unachievable, the loop **pauses** with the reason instead of re-firing until the tick budget (fail-open: a broken judge never wedges the loop). |
| You | `/loop stop` (or `/loop pause` to keep it around). |
| The backstop budget | `loops.max_ticks` (default 100) pauses the loop so an unattended session can't burn tokens forever. `0` = unlimited. |