fix(gateway): clear needs_attention/retrying_since on every transition to connected

`_flag_reconnect_needs_attention` stamps needs_attention=True +
retrying_since when a reconnect loop passes the attention threshold, but
only `_install_reconnected_adapter` (the watcher's own success path)
cleared them. Every OTHER writer of `connected` — the startup stamp in
`_start_connect_pending`, `BasePlatformAdapter._mark_connected`, Telegram's
in-place polling recovery (`send_path_degraded` -> healthy) — left the
flags in place. A gateway restarted after an escalation therefore reported
telegram as `connected, needs_attention: true, retrying_since: 2026-08-30`
for two weeks on a healthy bot.

Fix at the single seam: `write_runtime_status(platform_state="connected")`
now defaults needs_attention=False / retrying_since=None unless the caller
passed them explicitly, so all four writers agree without each growing a
copy of the clear. Discord's connect() (cherry-picked from #102557 by
@sudhirpatil) calls `_mark_connected()` instead of a bare `_running = True`
so its stale `fatal` stamp clears through the same seam (#102554).
This commit is contained in:
teknium1
2026-09-14 21:35:56 -07:00
committed by Teknium
parent 395e4248d0
commit c77b9e8093
2 changed files with 34 additions and 0 deletions
+7
View File
@@ -842,6 +842,13 @@ def write_runtime_status(
))
if platform is not _UNSET:
platform_payload = payload["platforms"].get(platform, {})
if platform_state == "connected":
# Every writer that publishes ``connected`` (startup stamp, adapter ``_mark_connected``,
# Telegram's in-place polling recovery) ends the retry episode; only the watcher's
# reconnect path used to say so, and a restart after a NEEDS_ATTENTION escalation
# carried the flag into a healthy record for weeks.
needs_attention = False if needs_attention is _UNSET else needs_attention
retrying_since = None if retrying_since is _UNSET else retrying_since
_apply_set_fields(platform_payload, (
("state", platform_state, None), ("error_code", error_code, None),
("error_message", error_message, None),
+27
View File
@@ -412,6 +412,33 @@ class TestGatewayRuntimeStatus:
assert payload["platforms"]["discord"]["error_code"] is None
assert payload["platforms"]["discord"]["error_message"] is None
@pytest.mark.parametrize("via", ["startup_stamp", "adapter_mark_connected"])
def test_connected_clears_needs_attention_from_any_writer(self, tmp_path, monkeypatch, via):
"""The reconnect-loop escalation (needs_attention + retrying_since) must end on EVERY
``connected`` write, not only the watcher's. A gateway restart after an escalation stamps
``connected`` from the startup path / adapter, which left the flag sticky for weeks on a
healthy Telegram record."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
status.write_runtime_status(
platform="telegram", platform_state="retrying", needs_attention=True,
retrying_since="2026-08-30T07:53:47+00:00",
)
if via == "startup_stamp":
status.write_runtime_status(
platform="telegram", platform_state="connected", error_code=None, error_message=None)
else:
from gateway.platforms.base import BasePlatformAdapter
adapter = object.__new__(type("_Adapter", (BasePlatformAdapter,), {
m: (lambda *a, **k: None) for m in ("connect", "disconnect", "get_chat_info", "send")}))
adapter._runtime_status_platform_key = "telegram"
adapter._fatal_error_code = adapter._fatal_error_message = None
adapter._fatal_error_retryable = True
adapter._mark_connected()
entry = status.read_runtime_status()["platforms"]["telegram"]
assert entry["state"] == "connected"
assert entry["needs_attention"] is False
assert entry["retrying_since"] is None
class TestGetProcessStartTime:
"""Start-time fingerprint backing the PID-reuse guard (#43846 / #50468).