fix(loops): pause /loop --until on a blocked verdict; trim redundant gate condition and duplicate test
The goal judge now returns 'blocked' for unachievable goals, but the /loop --until gate only checked == 'done', so an impossible stop condition would re-fire every tick until loops.max_ticks. Pause the loop with the judge's reason instead. Also collapse the kanban gate callers' 'gate_verdict == "continue" or rejection is not None' to 'rejection is not None' (rejection is None iff verdict == done), drop the duplicate blocked-verdict goal test, and document the verdict.
This commit is contained in:
@@ -2398,7 +2398,7 @@ def _cmd_complete(args: argparse.Namespace) -> int:
|
|||||||
)
|
)
|
||||||
failed.append(tid)
|
failed.append(tid)
|
||||||
continue
|
continue
|
||||||
if gate_verdict == "continue" or rejection is not None:
|
if rejection is not None:
|
||||||
print(
|
print(
|
||||||
f"kanban: goal completion of {tid} rejected by judge: {rejection}. "
|
f"kanban: goal completion of {tid} rejected by judge: {rejection}. "
|
||||||
f"Provide evidence matching the task's acceptance criteria.",
|
f"Provide evidence matching the task's acceptance criteria.",
|
||||||
@@ -2559,7 +2559,7 @@ def _cmd_request_review(args: argparse.Namespace) -> int:
|
|||||||
file=sys.stderr,
|
file=sys.stderr,
|
||||||
)
|
)
|
||||||
return 1
|
return 1
|
||||||
if gate_verdict == "continue" or rejection is not None:
|
if rejection is not None:
|
||||||
print(
|
print(
|
||||||
f"kanban: goal review handoff of {tid} rejected by judge: "
|
f"kanban: goal review handoff of {tid} rejected by judge: "
|
||||||
f"{rejection}. Provide acceptance evidence matching the task.",
|
f"{rejection}. Provide acceptance evidence matching the task.",
|
||||||
|
|||||||
@@ -768,6 +768,18 @@ class LoopManager:
|
|||||||
"reason": s.last_stop_reason,
|
"reason": s.last_stop_reason,
|
||||||
"message": f"✓ Loop finished after {s.ticks_fired} tick{'s' if s.ticks_fired != 1 else ''} — {reason}",
|
"message": f"✓ Loop finished after {s.ticks_fired} tick{'s' if s.ticks_fired != 1 else ''} — {reason}",
|
||||||
}
|
}
|
||||||
|
if verdict == "blocked":
|
||||||
|
# Judge ruled the stop condition unachievable — don't spin
|
||||||
|
# until the tick budget; pause so the user can re-scope.
|
||||||
|
s.status = "paused"
|
||||||
|
s.paused_reason = f"stop condition judged unachievable: {reason}"
|
||||||
|
save_loop(self.session_id, s)
|
||||||
|
return {
|
||||||
|
"status": "paused",
|
||||||
|
"stopped": True,
|
||||||
|
"reason": s.paused_reason,
|
||||||
|
"message": f"⏸ Loop paused — {s.paused_reason}. /loop resume to keep going, /loop stop to end it.",
|
||||||
|
}
|
||||||
|
|
||||||
# 3. --times user cap.
|
# 3. --times user cap.
|
||||||
if s.times and s.ticks_fired >= s.times:
|
if s.times and s.ticks_fired >= s.times:
|
||||||
|
|||||||
@@ -833,20 +833,3 @@ class TestBlockedVerdict:
|
|||||||
assert mgr.state is not None
|
assert mgr.state is not None
|
||||||
assert mgr.state.status == "paused"
|
assert mgr.state.status == "paused"
|
||||||
assert "unachievable" in (mgr.state.paused_reason or "").lower()
|
assert "unachievable" in (mgr.state.paused_reason or "").lower()
|
||||||
|
|
||||||
def test_blocked_verdict_never_records_done(self, hermes_home):
|
|
||||||
from unittest.mock import patch
|
|
||||||
from hermes_cli.goals import GoalManager
|
|
||||||
|
|
||||||
mgr = GoalManager(session_id="blocked-sid-2")
|
|
||||||
mgr.set("square the circle")
|
|
||||||
with patch(
|
|
||||||
"hermes_cli.goals.judge_goal",
|
|
||||||
return_value=("blocked", "mathematically impossible", False, None, False),
|
|
||||||
):
|
|
||||||
decision = mgr.evaluate_after_turn("This cannot be done.")
|
|
||||||
assert decision["status"] != "done"
|
|
||||||
assert mgr.state is not None
|
|
||||||
assert mgr.state.status != "done"
|
|
||||||
assert mgr.state.last_verdict == "blocked"
|
|
||||||
|
|
||||||
|
|||||||
@@ -443,6 +443,20 @@ class TestTickLifecycle:
|
|||||||
decision = mgr.complete_tick("3 tests still failing")
|
decision = mgr.complete_tick("3 tests still failing")
|
||||||
assert decision["stopped"] is False
|
assert decision["stopped"] is False
|
||||||
|
|
||||||
|
def test_until_judge_blocked_pauses(self, hermes_home):
|
||||||
|
"""An unachievable stop condition pauses the loop instead of spinning to the tick budget."""
|
||||||
|
from hermes_cli.loops import LoopManager
|
||||||
|
|
||||||
|
mgr = LoopManager(session_id="t11b")
|
||||||
|
state = mgr.set("poll", interval_seconds=300, until="the deleted repo's CI is green")
|
||||||
|
state.next_due_at = time.time() - 1
|
||||||
|
mgr.fire_tick()
|
||||||
|
with patch("hermes_cli.goals.judge_goal", return_value=("blocked", "repo no longer exists", False, None, False)):
|
||||||
|
decision = mgr.complete_tick("The repository was deleted; there is no CI to watch.")
|
||||||
|
assert decision["stopped"] is True
|
||||||
|
assert decision["status"] == "paused"
|
||||||
|
assert "unachievable" in decision["message"]
|
||||||
|
|
||||||
def test_until_judge_error_fails_open(self, hermes_home):
|
def test_until_judge_error_fails_open(self, hermes_home):
|
||||||
from hermes_cli.loops import LoopManager
|
from hermes_cli.loops import LoopManager
|
||||||
|
|
||||||
|
|||||||
@@ -770,7 +770,7 @@ def _handle_complete(args: dict, **kw) -> str:
|
|||||||
f"or record the block with kanban_block and hand the "
|
f"or record the block with kanban_block and hand the "
|
||||||
f"decision to a human / reviewer."
|
f"decision to a human / reviewer."
|
||||||
)
|
)
|
||||||
if gate_verdict == "continue" or rejection is not None:
|
if rejection is not None:
|
||||||
return tool_error(
|
return tool_error(
|
||||||
f"Goal completion rejected by judge: {rejection}. "
|
f"Goal completion rejected by judge: {rejection}. "
|
||||||
f"To proceed, either: (1) provide explicit acceptance "
|
f"To proceed, either: (1) provide explicit acceptance "
|
||||||
@@ -958,7 +958,7 @@ def _handle_request_review(args: dict, **kw) -> str:
|
|||||||
f"unachievable — {rejection}. Record the block with "
|
f"unachievable — {rejection}. Record the block with "
|
||||||
f"kanban_block instead of requesting review."
|
f"kanban_block instead of requesting review."
|
||||||
)
|
)
|
||||||
if gate_verdict == "continue" or rejection is not None:
|
if rejection is not None:
|
||||||
return tool_error(
|
return tool_error(
|
||||||
f"Goal review handoff rejected by judge: {rejection}. "
|
f"Goal review handoff rejected by judge: {rejection}. "
|
||||||
"Provide acceptance evidence matching the card before "
|
"Provide acceptance evidence matching the card before "
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ What you'll see:
|
|||||||
|
|
||||||
1. **Goal accepted** — `⊙ Goal set (20-turn budget): <your goal>`
|
1. **Goal accepted** — `⊙ Goal set (20-turn budget): <your goal>`
|
||||||
2. **Turn 1 runs** — Hermes starts working as if you'd sent the goal as a normal message.
|
2. **Turn 1 runs** — Hermes starts working as if you'd sent the goal as a normal message.
|
||||||
3. **Judge runs** — after the turn, the judge model decides `done` or `continue`.
|
3. **Judge runs** — after the turn, the judge model decides `done`, `continue`, or `blocked`.
|
||||||
4. **Loop fires if needed** — if `continue`, you'll see `↻ Continuing toward goal (1/20): <judge's reason>` and Hermes takes the next step automatically.
|
4. **Loop fires if needed** — if `continue`, you'll see `↻ Continuing toward goal (1/20): <judge's reason>` and Hermes takes the next step automatically.
|
||||||
5. **Terminates** — eventually you see either `✓ Goal achieved: <reason>` or `⏸ Goal paused — N/20 turns used`.
|
5. **Terminates** — eventually you see either `✓ Goal achieved: <reason>` or `⏸ Goal paused — N/20 turns used`.
|
||||||
|
|
||||||
@@ -140,7 +140,7 @@ A completion contract makes the judge stricter, but the judge is still an LLM re
|
|||||||
How it works, each turn:
|
How it works, each turn:
|
||||||
|
|
||||||
1. **Gates run before the judge.** If any gate fails, the judge is *not called* — a red gate is deterministic evidence the goal isn't done. The gate's exit code and output tail (last ~3 KB) become the continuation prompt, so the agent iterates against the actual failure instead of a vibe.
|
1. **Gates run before the judge.** If any gate fails, the judge is *not called* — a red gate is deterministic evidence the goal isn't done. The gate's exit code and output tail (last ~3 KB) become the continuation prompt, so the agent iterates against the actual failure instead of a vibe.
|
||||||
2. **All gates pass → normal judging.** The LLM judge then decides done/continue/wait exactly as before.
|
2. **All gates pass → normal judging.** The LLM judge then decides done/blocked/continue/wait exactly as before.
|
||||||
3. **Unchanged workspace → no re-run.** If a gate failed and nothing changed in the workspace since (tracked via a git fingerprint of HEAD + working-tree status), the gate is not re-run — the recorded failure is replayed and the attempt count advances. A stuck agent can't burn wall-clock re-running an identical red suite. Outside a git repo, gates simply always re-run.
|
3. **Unchanged workspace → no re-run.** If a gate failed and nothing changed in the workspace since (tracked via a git fingerprint of HEAD + working-tree status), the gate is not re-run — the recorded failure is replayed and the attempt count advances. A stuck agent can't burn wall-clock re-running an identical red suite. Outside a git repo, gates simply always re-run.
|
||||||
4. **Retries are bounded.** Each gate defaults to 3 retries and a 5-minute timeout. When a gate exhausts its retries the goal auto-pauses (like the turn budget) with a message telling you to fix it manually, remove the gate, or `/goal resume`.
|
4. **Retries are bounded.** Each gate defaults to 3 retries and a 5-minute timeout. When a gate exhausts its retries the goal auto-pauses (like the turn budget) with a message telling you to fix it manually, remove the gate, or `/goal resume`.
|
||||||
|
|
||||||
@@ -179,9 +179,9 @@ After every turn, Hermes calls an auxiliary model with:
|
|||||||
|
|
||||||
- The standing goal text
|
- The standing goal text
|
||||||
- The agent's most recent final response (last ~4 KB of text)
|
- The agent's most recent final response (last ~4 KB of text)
|
||||||
- A system prompt telling the judge to reply with strict one-line JSON: `{"verdict": "done" | "continue" | "wait", "reason": "<one-sentence rationale>"}` (wait verdicts add `wait_on_session` / `wait_on_pid` / `wait_for_seconds`; the legacy `{"done": <bool>, "reason": "..."}` shape is still accepted)
|
- A system prompt telling the judge to reply with strict one-line JSON: `{"verdict": "done" | "blocked" | "continue" | "wait", "reason": "<one-sentence rationale>"}` (wait verdicts add `wait_on_session` / `wait_on_pid` / `wait_for_seconds`; the legacy `{"done": <bool>, "reason": "..."}` shape is still accepted)
|
||||||
|
|
||||||
The judge is deliberately conservative: it marks a goal `done` only when the response **explicitly** confirms the goal is complete, when the final deliverable is clearly produced, or when the goal is unachievable/blocked (treated as DONE with a block reason so we don't burn budget on impossible tasks).
|
The judge is deliberately conservative: it marks a goal `done` only when the response **explicitly** confirms the goal is complete, when the final deliverable is clearly produced. A goal the agent explains is **unachievable** (impossible, out of scope, needs user input) gets a `blocked` verdict instead — never `done`: the goal **pauses** with the judge's reason (`🚫 Goal judged unachievable — paused`), so you can re-scope it with `/goal <text>` or override with `/goal resume` rather than burning budget or having an impossible task waved through as complete.
|
||||||
|
|
||||||
### Fail-open semantics
|
### Fail-open semantics
|
||||||
|
|
||||||
|
|||||||
@@ -512,7 +512,7 @@ def register(ctx):
|
|||||||
|
|
||||||
### Goal-mode cards (`--goal`)
|
### Goal-mode cards (`--goal`)
|
||||||
|
|
||||||
By default each worker gets **one shot** at its card — do the work, call `kanban_complete`/`kanban_block`, exit. Pass `--goal` (CLI) or `goal_mode=True` (the `kanban_create` tool / dashboard) to instead run that worker in a **goal loop**, the same Ralph-style engine behind the `/goal` slash command: after every turn an auxiliary judge checks the worker's output against the card's title + body (treated as the acceptance criteria), and if the work isn't done — and the turn budget remains — the worker keeps going **in the same session** until the judge agrees, the worker terminates the task itself, or the budget runs out (which **blocks** the card for human review rather than exiting silently).
|
By default each worker gets **one shot** at its card — do the work, call `kanban_complete`/`kanban_block`, exit. Pass `--goal` (CLI) or `goal_mode=True` (the `kanban_create` tool / dashboard) to instead run that worker in a **goal loop**, the same Ralph-style engine behind the `/goal` slash command: after every turn an auxiliary judge checks the worker's output against the card's title + body (treated as the acceptance criteria), and if the work isn't done — and the turn budget remains — the worker keeps going **in the same session** until the judge agrees, the worker terminates the task itself, or the budget runs out (which **blocks** the card for human review rather than exiting silently). If the judge rules the goal **unachievable** as written, the card is blocked immediately with the judge's reason — an impossible card is never marked done, and `kanban complete` / `kanban request-review` on such a card are rejected with a pointer to `kanban block` or re-scoping.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
hermes kanban create "Translate the docs site to French" \
|
hermes kanban create "Translate the docs site to French" \
|
||||||
|
|||||||
@@ -61,7 +61,7 @@ A loop ends when any of these fires:
|
|||||||
|---|---|
|
|---|---|
|
||||||
| The agent decides it's done | The wakeup prompt teaches the agent to end its reply with `LOOP_COMPLETE` on its own line when the task is finished or moot. |
|
| The agent decides it's done | The wakeup prompt teaches the agent to end its reply with `LOOP_COMPLETE` on its own line when the task is finished or moot. |
|
||||||
| A run cap | `--times N` — stop after N wakeups. |
|
| A run cap | `--times N` — stop after N wakeups. |
|
||||||
| An evidence-based condition | `--until <condition>` — after each wakeup, the same auxiliary judge that powers `/goal` checks the reply against your condition (fail-open: a broken judge never wedges the loop). |
|
| An evidence-based condition | `--until <condition>` — after each wakeup, the same auxiliary judge that powers `/goal` checks the reply against your condition. If the judge rules the condition unachievable, the loop **pauses** with the reason instead of re-firing until the tick budget (fail-open: a broken judge never wedges the loop). |
|
||||||
| You | `/loop stop` (or `/loop pause` to keep it around). |
|
| You | `/loop stop` (or `/loop pause` to keep it around). |
|
||||||
| The backstop budget | `loops.max_ticks` (default 100) pauses the loop so an unattended session can't burn tokens forever. `0` = unlimited. |
|
| The backstop budget | `loops.max_ticks` (default 100) pauses the loop so an unattended session can't burn tokens forever. `0` = unlimited. |
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user