fix(kanban): 5xx and timeouts requeue the worker instead of spending its retry budget
`server_error` and `timeout` join the transient-provider set that makes a Kanban worker exit 75 (EX_TEMPFAIL). A provider outage or a hung connection says nothing about the task, so the dispatcher requeues without a failure tick rather than counting toward the circuit breaker (#91206 proposed the same set).
This commit is contained in:
@@ -4089,9 +4089,12 @@ def _sync_cli_session_id_from_agent(cli) -> None:
|
||||
cli.session_id = cli.agent.session_id
|
||||
|
||||
|
||||
# ``failure_reason`` values that say nothing about the task itself: the provider or the
|
||||
# account is walled, so a Kanban worker signals "try later" instead of "I failed".
|
||||
_QUOTA_WALL_REASONS = frozenset({"rate_limit", "upstream_rate_limit", "billing", "overloaded"})
|
||||
# ``failure_reason`` values that say nothing about the task itself: the provider is walled,
|
||||
# down or unreachable, or the account is out of credit, so a Kanban worker signals "try
|
||||
# later" instead of "I failed" and the dispatcher does not spend the task's retry budget on it.
|
||||
_TRANSIENT_PROVIDER_REASONS = frozenset({
|
||||
"rate_limit", "upstream_rate_limit", "billing", "overloaded", "server_error", "timeout",
|
||||
})
|
||||
|
||||
|
||||
def _single_query_exit_code(result) -> int:
|
||||
@@ -4102,7 +4105,7 @@ def _single_query_exit_code(result) -> int:
|
||||
failed, so ``result`` is not a dict). A Kanban worker (``HERMES_KANBAN_TASK`` set) that
|
||||
failed purely on a provider rate-limit / billing wall exits ``KANBAN_RATE_LIMIT_EXIT_CODE``
|
||||
(EX_TEMPFAIL): the dispatcher books that run ``rate_limited`` and requeues the task
|
||||
WITHOUT counting a failure, so a quota window cannot trip the circuit breaker.
|
||||
WITHOUT counting a failure, so a quota window or a provider outage cannot trip the breaker.
|
||||
"""
|
||||
if not isinstance(result, dict):
|
||||
return 1
|
||||
@@ -4110,7 +4113,7 @@ def _single_query_exit_code(result) -> int:
|
||||
return 130
|
||||
if not (result.get("failed") or result.get("partial") or result.get("completed") is False):
|
||||
return 0
|
||||
if os.environ.get("HERMES_KANBAN_TASK") and result.get("failure_reason") in _QUOTA_WALL_REASONS:
|
||||
if os.environ.get("HERMES_KANBAN_TASK") and result.get("failure_reason") in _TRANSIENT_PROVIDER_REASONS:
|
||||
from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE
|
||||
return KANBAN_RATE_LIMIT_EXIT_CODE
|
||||
return 1
|
||||
|
||||
@@ -44,8 +44,10 @@ def _run_non_quiet(monkeypatch, turn_result):
|
||||
return None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reason", ["rate_limit", "upstream_rate_limit", "billing", "overloaded"])
|
||||
def test_dispatcher_spawned_worker_signals_a_quota_wall_not_a_protocol_violation(monkeypatch, reason):
|
||||
@pytest.mark.parametrize(
|
||||
"reason", ["rate_limit", "upstream_rate_limit", "billing", "overloaded", "server_error", "timeout"]
|
||||
)
|
||||
def test_dispatcher_spawned_worker_signals_a_provider_outage_not_a_protocol_violation(monkeypatch, reason):
|
||||
monkeypatch.setenv("HERMES_KANBAN_TASK", "t_abc123")
|
||||
code = _run_non_quiet(monkeypatch, {"failed": True, "failure_reason": reason})
|
||||
assert code == KANBAN_RATE_LIMIT_EXIT_CODE
|
||||
|
||||
@@ -181,8 +181,9 @@ stdio) the process exit code reports the turn's outcome, on both the quiet and
|
||||
the non-quiet path: `0` the turn completed; `1` it failed, stopped partway
|
||||
(`partial`), hit the iteration budget, or never ran (credentials / agent init
|
||||
failed); `130` it was interrupted. A Kanban dispatcher-spawned worker
|
||||
(`HERMES_KANBAN_TASK` set) whose turn failed only because the provider
|
||||
rate-limited or overloaded it, or the account hit a billing/quota wall, exits
|
||||
(`HERMES_KANBAN_TASK` set) whose turn failed only because the provider was
|
||||
rate-limited, overloaded, returning 5xx, timing out, or the account hit a
|
||||
billing/quota wall, exits
|
||||
`75` (`EX_TEMPFAIL`) so the dispatcher requeues the task without counting a
|
||||
failure. With `--format stream-json` the terminal `result` record carries the
|
||||
same `exit_code`.
|
||||
|
||||
@@ -505,8 +505,8 @@ protocol. If the worker process exits with status 0 while the task is still
|
||||
`running`, the dispatcher treats that as a protocol violation and emits a
|
||||
`protocol_violation` event. A dispatcher-spawned worker whose turn failed
|
||||
therefore exits non-zero: `1` for an ordinary failure, and `75`
|
||||
(`EX_TEMPFAIL`) when the provider rate-limited or overloaded it, or the
|
||||
account hit a billing/quota wall — the dispatcher records that run as `rate_limited` and
|
||||
(`EX_TEMPFAIL`) when the provider was rate-limited, overloaded, returning
|
||||
5xx or timing out, or the account hit a billing/quota wall — the dispatcher records that run as `rate_limited` and
|
||||
requeues the task without counting a failure, so a quota window is never
|
||||
booked as a protocol violation.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user