From 2fcb50c50ebd09cc269cbeed2e7b964a66547165 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Mon, 14 Sep 2026 18:05:50 -0700 Subject: [PATCH] test(update): trim marker-discharge tests to two invariants; document the identity rule The five #105417 tests collapse into one positive (fleet current at expected_sha discharges the marker) and one parametrized negative (stale row / empty probe / unknown identity / checkout moved past the marker keep the warning). Both are red on origin/main. collect_fleet_versions' docstring now names why a live PID alone is not a `current` row (#110420): write_runtime_status re-stamps pid/code_sha for whatever process writes it, so the state-file fallback must pass live_gateway_pid_for_home like the inventory already does (#109680). --- hermes_cli/update_receipt.py | 5 +- .../test_update_fleet_restart_pending.py | 97 ++++--------------- 2 files changed, 25 insertions(+), 77 deletions(-) diff --git a/hermes_cli/update_receipt.py b/hermes_cli/update_receipt.py index bef4a88372..399e0407f3 100644 --- a/hermes_cli/update_receipt.py +++ b/hermes_cli/update_receipt.py @@ -287,7 +287,10 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l ``stale`` — gateway stamped a code_sha that differs from the updated checkout's HEAD (it is still serving pre-update modules). ``unknown`` — gateway predates the code-identity stamp (started before this - feature landed) or identity could not be resolved. ``down`` — the gateway was ALIVE when this update + feature landed), identity could not be resolved, or the state file's live PID is not the + verified gateway for that home (``live_gateway_pid_for_home``): ``write_runtime_status`` re-stamps + ``pid``/``code_sha`` for whatever process writes it, so a foreign writer must never read as + ``current`` (#110420, sibling of #109680). ``down`` — the gateway was ALIVE when this update started (``pre_restart_pids``), its runtime status still says running, but the PID is dead and no successor rewrote the record: the restart phase stopped it and nothing came back. Without this row a killed-and-never-replaced gateway produced NO entry at all and the matrix passed silently (Phase-1 diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py index 95c6758690..cb303d03e6 100644 --- a/tests/hermes_cli/test_update_fleet_restart_pending.py +++ b/tests/hermes_cli/test_update_fleet_restart_pending.py @@ -611,12 +611,12 @@ def test_startup_warn_silent_when_nothing_pending(capsys): assert captured.out == "" -# ── Self-heal: marker left behind by a supervisor-level restart (TRA-1180 class) ── +# ── Self-heal: marker left behind by a supervisor-level restart (#105417 / #111272) ── # -# `systemctl --user restart hermes-gateway` (weekly-update fallback, manual ops) never -# runs this module's clear path, so the marker survives a restart that DID bring the -# fleet to the pulled code — and every later CLI call warns forever. The marker must be -# discharged when (and only when) the fleet provably serves expected_sha. +# `systemctl --user restart hermes-gateway` never runs this module's clear path, and an update +# whose fleet probe answered empty exits before clearing — so the marker survives a restart +# that DID bring the fleet to the pulled code, and every later CLI call warns forever. The +# marker is discharged when (and only when) the fleet provably serves expected_sha. def _patch_marker_sha(monkeypatch, disk_sha): @@ -628,11 +628,6 @@ def test_startup_warn_discharged_when_fleet_current(monkeypatch, capsys): disk_sha = "e" * 40 update_cmd._write_fleet_restart_pending_marker(expected_sha=disk_sha) _patch_marker_sha(monkeypatch, disk_sha) - monkeypatch.setattr( - update_cmd_fleet, - "_marker_only_restart_obsolete", - update_cmd_fleet._marker_only_restart_obsolete, - ) monkeypatch.setattr( "hermes_cli.update_receipt.collect_fleet_versions", lambda **kwargs: [ @@ -642,77 +637,27 @@ def test_startup_warn_discharged_when_fleet_current(monkeypatch, capsys): update_cmd._warn_pending_fleet_restart_on_startup() - captured = capsys.readouterr() - assert captured.err == "" + assert capsys.readouterr().err == "" assert not update_cmd._fleet_restart_pending_marker_path().exists() -def test_startup_warn_kept_when_fleet_stale(monkeypatch, capsys): - disk_sha = "e" * 40 - update_cmd._write_fleet_restart_pending_marker(expected_sha=disk_sha) +@pytest.mark.parametrize( + "disk_sha, fleet", + [ + ("e" * 40, [{"profile": "default", "pid": 42, "code_sha": "7" * 40, "code_version": "0.20.0", "state": "stale"}]), + ("e" * 40, []), # probe answered empty: no proof either way + ("e" * 40, [{"profile": "default", "pid": 42, "code_sha": None, "code_version": None, "state": "unknown"}]), + # checkout advanced past the marker: a newer pull owns a fresh obligation + ("f" * 40, [{"profile": "default", "pid": 42, "code_sha": "e" * 40, "code_version": None, "state": "current"}]), + ], + ids=["stale-row", "empty-probe", "unknown-identity", "checkout-moved"], +) +def test_startup_warn_kept_without_positive_evidence(monkeypatch, capsys, disk_sha, fleet): + update_cmd._write_fleet_restart_pending_marker(expected_sha="e" * 40) _patch_marker_sha(monkeypatch, disk_sha) - monkeypatch.setattr( - "hermes_cli.update_receipt.collect_fleet_versions", - lambda **kwargs: [ - {"profile": "default", "pid": 42, "code_sha": "7" * 40, "code_version": "0.20.0", "state": "stale"} - ], - ) + monkeypatch.setattr("hermes_cli.update_receipt.collect_fleet_versions", lambda **kwargs: fleet) update_cmd._warn_pending_fleet_restart_on_startup() - err = capsys.readouterr().err - assert "did not restart running gateways" in err + assert "did not restart running gateways" in capsys.readouterr().err assert update_cmd._fleet_restart_pending_marker_path().exists() - - -def test_startup_warn_kept_when_fleet_probe_empty(monkeypatch, capsys): - disk_sha = "e" * 40 - update_cmd._write_fleet_restart_pending_marker(expected_sha=disk_sha) - _patch_marker_sha(monkeypatch, disk_sha) - monkeypatch.setattr( - "hermes_cli.update_receipt.collect_fleet_versions", - lambda **kwargs: [], - ) - - update_cmd._warn_pending_fleet_restart_on_startup() - - err = capsys.readouterr().err - assert "did not restart running gateways" in err - assert update_cmd._fleet_restart_pending_marker_path().exists() - - -def test_startup_warn_kept_when_fleet_identity_unknown(monkeypatch, capsys): - disk_sha = "e" * 40 - update_cmd._write_fleet_restart_pending_marker(expected_sha=disk_sha) - _patch_marker_sha(monkeypatch, disk_sha) - monkeypatch.setattr( - "hermes_cli.update_receipt.collect_fleet_versions", - lambda **kwargs: [ - {"profile": "default", "pid": 42, "code_sha": None, "code_version": None, "state": "unknown"} - ], - ) - - update_cmd._warn_pending_fleet_restart_on_startup() - - err = capsys.readouterr().err - assert "did not restart running gateways" in err - assert update_cmd._fleet_restart_pending_marker_path().exists() - - -def test_marker_kept_when_newer_pull_moved_checkout_past_marker(monkeypatch, capsys): - update_cmd._write_fleet_restart_pending_marker(expected_sha="d" * 40) - _patch_marker_sha(monkeypatch, "e" * 40) # checkout advanced after the marker - seen = {"called": False} - - def _collect(**kwargs): - seen["called"] = True - return [{"profile": "default", "pid": 42, "code_sha": "d" * 40, "code_version": None, "state": "current"}] - - monkeypatch.setattr("hermes_cli.update_receipt.collect_fleet_versions", _collect) - - update_cmd._warn_pending_fleet_restart_on_startup() - - err = capsys.readouterr().err - assert "did not restart running gateways" in err - assert update_cmd._fleet_restart_pending_marker_path().exists() - assert seen["called"] is False # short-circuits before probing the fleet