fix(gateway): clear needs_attention/retrying_since on every transition to connected
`_flag_reconnect_needs_attention` stamps needs_attention=True + retrying_since when a reconnect loop passes the attention threshold, but only `_install_reconnected_adapter` (the watcher's own success path) cleared them. Every OTHER writer of `connected` — the startup stamp in `_start_connect_pending`, `BasePlatformAdapter._mark_connected`, Telegram's in-place polling recovery (`send_path_degraded` -> healthy) — left the flags in place. A gateway restarted after an escalation therefore reported telegram as `connected, needs_attention: true, retrying_since: 2026-08-30` for two weeks on a healthy bot. Fix at the single seam: `write_runtime_status(platform_state="connected")` now defaults needs_attention=False / retrying_since=None unless the caller passed them explicitly, so all four writers agree without each growing a copy of the clear. Discord's connect() (cherry-picked from #102557 by @sudhirpatil) calls `_mark_connected()` instead of a bare `_running = True` so its stale `fatal` stamp clears through the same seam (#102554).
This commit is contained in:
@@ -842,6 +842,13 @@ def write_runtime_status(
|
||||
))
|
||||
if platform is not _UNSET:
|
||||
platform_payload = payload["platforms"].get(platform, {})
|
||||
if platform_state == "connected":
|
||||
# Every writer that publishes ``connected`` (startup stamp, adapter ``_mark_connected``,
|
||||
# Telegram's in-place polling recovery) ends the retry episode; only the watcher's
|
||||
# reconnect path used to say so, and a restart after a NEEDS_ATTENTION escalation
|
||||
# carried the flag into a healthy record for weeks.
|
||||
needs_attention = False if needs_attention is _UNSET else needs_attention
|
||||
retrying_since = None if retrying_since is _UNSET else retrying_since
|
||||
_apply_set_fields(platform_payload, (
|
||||
("state", platform_state, None), ("error_code", error_code, None),
|
||||
("error_message", error_message, None),
|
||||
|
||||
@@ -412,6 +412,33 @@ class TestGatewayRuntimeStatus:
|
||||
assert payload["platforms"]["discord"]["error_code"] is None
|
||||
assert payload["platforms"]["discord"]["error_message"] is None
|
||||
|
||||
@pytest.mark.parametrize("via", ["startup_stamp", "adapter_mark_connected"])
|
||||
def test_connected_clears_needs_attention_from_any_writer(self, tmp_path, monkeypatch, via):
|
||||
"""The reconnect-loop escalation (needs_attention + retrying_since) must end on EVERY
|
||||
``connected`` write, not only the watcher's. A gateway restart after an escalation stamps
|
||||
``connected`` from the startup path / adapter, which left the flag sticky for weeks on a
|
||||
healthy Telegram record."""
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
status.write_runtime_status(
|
||||
platform="telegram", platform_state="retrying", needs_attention=True,
|
||||
retrying_since="2026-08-30T07:53:47+00:00",
|
||||
)
|
||||
if via == "startup_stamp":
|
||||
status.write_runtime_status(
|
||||
platform="telegram", platform_state="connected", error_code=None, error_message=None)
|
||||
else:
|
||||
from gateway.platforms.base import BasePlatformAdapter
|
||||
adapter = object.__new__(type("_Adapter", (BasePlatformAdapter,), {
|
||||
m: (lambda *a, **k: None) for m in ("connect", "disconnect", "get_chat_info", "send")}))
|
||||
adapter._runtime_status_platform_key = "telegram"
|
||||
adapter._fatal_error_code = adapter._fatal_error_message = None
|
||||
adapter._fatal_error_retryable = True
|
||||
adapter._mark_connected()
|
||||
entry = status.read_runtime_status()["platforms"]["telegram"]
|
||||
assert entry["state"] == "connected"
|
||||
assert entry["needs_attention"] is False
|
||||
assert entry["retrying_since"] is None
|
||||
|
||||
|
||||
class TestGetProcessStartTime:
|
||||
"""Start-time fingerprint backing the PID-reuse guard (#43846 / #50468).
|
||||
|
||||
Reference in New Issue
Block a user