feat(gateway): disk-usage telemetry + dashboard disk-pressure banner (NS-656)
Extends the NS-656 memory-pressure surface to cover disk exhaustion (OOF-2 / OOF-107 lineage: agents fill their data volume — SQLite writes fail, sessions stop persisting — while every dashboard looks healthy). - gateway/disk_status.py (new): collect_disk_status() samples shutil.disk_usage(HERMES_HOME) and classifies pressure (critical: <256 MB free or >=95% used; elevated: <512 MB free). Never raises — degrades to pressure="unknown" with null telemetry, same contract as collect_memory_status(). - /api/status: sibling `disk` block next to `memory`, advisory only — not folded into component/overall health. - web: DiskPressureStatus type; MemoryPressureBanner generalized to a resource banner with worst-first triggers (disk critical > memory critical > OOM restart > disk elevated > memory elevated) and cascading dismissals — hiding the top trigger surfaces the next one instead of silencing everything. All dismissals stay boot_id-scoped. - i18n: diskCriticalBanner / diskElevatedBanner (en, optional fields with English fallback per existing pattern). Tests: gateway/test_disk_status.py (14), web_server disk-block presence/degradation, banner disk trigger/priority/dismissal-cascade suite (21 total).
This commit is contained in:
@@ -0,0 +1,117 @@
|
||||
"""Disk-usage rollup for ``/api/status`` (NS-656).
|
||||
|
||||
Companion to :mod:`gateway.memory_status`, closing the same class of gap
|
||||
for storage: a hosted agent can fill its data volume completely — SQLite
|
||||
writes failing, session persistence dead, config saves lost — while its
|
||||
dashboard and the NAS agent card both look perfectly healthy. Fleet
|
||||
incidents OOF-2 (unrecoverable disk-full) and OOF-107 (fleet-wide disk
|
||||
exhaustion, remediated by hand) are exactly this failure mode.
|
||||
|
||||
The readiness endpoint already probes disk (``gateway/readiness.py::
|
||||
_probe_disk``), but readiness is a component verdict, not user-facing
|
||||
telemetry — nothing renders it. This module produces the public block
|
||||
the dashboard SPA and the NAS availability sweep actually consume.
|
||||
|
||||
Unlike the memory block (which distills already-persisted heartbeat
|
||||
files), disk is sampled live via :func:`shutil.disk_usage` — a single
|
||||
``statvfs`` call, the same thing the readiness probe does per request.
|
||||
There is no meaningful "staleness" dimension, so no ``sampled_at``.
|
||||
|
||||
Public-safety note: ``/api/status`` is an unauthenticated liveness probe
|
||||
(``PUBLIC_API_PATHS``). This block carries only coarse numbers (MB
|
||||
granularity, whole-percent usage) and an enum — the same disclosure
|
||||
class as the ``memory`` block.
|
||||
|
||||
Everything is best-effort and read-only: an unreadable filesystem
|
||||
degrades to ``pressure="unknown"`` rather than raising into the status
|
||||
endpoint.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Disk-pressure thresholds. Percent alone misleads in both directions:
|
||||
# 90% used on a 100 GB volume leaves a comfortable 10 GB, while 50% used
|
||||
# on a tiny volume can be one image download from write failures. So the
|
||||
# percent triggers are gated on absolute headroom also being low, and a
|
||||
# hard absolute floor applies regardless of size — below it, SQLite
|
||||
# journaling and config writes are at genuine risk on any volume.
|
||||
_CRITICAL_FREE_MB = 256 # < 256 MB free: critical on any volume
|
||||
_CRITICAL_PERCENT = 95.0 # >= 95% used AND < 1 GB free: critical
|
||||
_CRITICAL_HEADROOM_MB = 1024
|
||||
_ELEVATED_FREE_MB = 512 # < 512 MB free: elevated on any volume
|
||||
_ELEVATED_PERCENT = 85.0 # >= 85% used AND < 4 GB free: elevated
|
||||
_ELEVATED_HEADROOM_MB = 4096
|
||||
|
||||
_BYTES_PER_MB = 1024 * 1024
|
||||
|
||||
|
||||
def _coerce_mb(value: Any) -> Optional[int]:
|
||||
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def classify_disk_pressure(free_mb: Any, total_mb: Any) -> str:
|
||||
"""Map free/total MB to ``ok``/``elevated``/``critical``.
|
||||
|
||||
``unknown`` when the sample is missing or malformed — the caller must
|
||||
not treat "we could not read it" as "disk is fine".
|
||||
"""
|
||||
free = _coerce_mb(free_mb)
|
||||
total = _coerce_mb(total_mb)
|
||||
if free is None or total is None or total <= 0:
|
||||
return "unknown"
|
||||
used_percent = (1 - free / total) * 100.0
|
||||
if free < _CRITICAL_FREE_MB or (
|
||||
used_percent >= _CRITICAL_PERCENT and free < _CRITICAL_HEADROOM_MB
|
||||
):
|
||||
return "critical"
|
||||
if free < _ELEVATED_FREE_MB or (
|
||||
used_percent >= _ELEVATED_PERCENT and free < _ELEVATED_HEADROOM_MB
|
||||
):
|
||||
return "elevated"
|
||||
return "ok"
|
||||
|
||||
|
||||
def collect_disk_status(home: Optional[Path] = None) -> Dict[str, Any]:
|
||||
"""Build the ``disk`` block for ``/api/status``.
|
||||
|
||||
``home`` scopes the sample to a profile's HERMES_HOME (the status
|
||||
endpoint's ``?profile=`` handling passes it through); on hosted
|
||||
images every profile shares the ``/opt/data`` volume, so the answer
|
||||
is the same — but scoping keeps the contract identical to the
|
||||
``memory`` block's.
|
||||
|
||||
Always returns a dict — an unreadable/unmounted filesystem yields
|
||||
``{"pressure": "unknown", ...}``. Never raises.
|
||||
"""
|
||||
status: Dict[str, Any] = {
|
||||
"pressure": "unknown",
|
||||
"total_mb": None,
|
||||
"free_mb": None,
|
||||
"used_percent": None,
|
||||
}
|
||||
try:
|
||||
if home is None:
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
home = get_hermes_home()
|
||||
usage = shutil.disk_usage(home)
|
||||
except Exception:
|
||||
return status
|
||||
if usage.total <= 0:
|
||||
return status
|
||||
total_mb = usage.total // _BYTES_PER_MB
|
||||
free_mb = usage.free // _BYTES_PER_MB
|
||||
status["total_mb"] = total_mb
|
||||
status["free_mb"] = free_mb
|
||||
status["used_percent"] = round((usage.used / usage.total) * 100, 1)
|
||||
status["pressure"] = classify_disk_pressure(free_mb, total_mb)
|
||||
return status
|
||||
@@ -3354,6 +3354,24 @@ async def get_status(profile: Optional[str] = None):
|
||||
except Exception:
|
||||
status["memory"] = {"pressure": "unknown"}
|
||||
|
||||
# Disk-usage rollup (NS-656, same lineage as OOF-2/OOF-107 fleet
|
||||
# disk-exhaustion incidents). One statvfs call on HERMES_HOME's
|
||||
# filesystem — coarse MB numbers + enum, same public disclosure
|
||||
# class as the memory block, and equally advisory: not folded
|
||||
# into components/overall.
|
||||
try:
|
||||
from gateway.disk_status import collect_disk_status
|
||||
|
||||
status["disk"] = await asyncio.get_running_loop().run_in_executor(
|
||||
None,
|
||||
functools.partial(
|
||||
collect_disk_status,
|
||||
profile_dir if profile_dir else get_hermes_home(),
|
||||
),
|
||||
)
|
||||
except Exception:
|
||||
status["disk"] = {"pressure": "unknown"}
|
||||
|
||||
# Deferred FTS rebuild progress (schema v23): lets the desktop /
|
||||
# dashboard render a "search index rebuilding: N%" indicator instead
|
||||
# of users wondering why old-message search is slower after an
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Tests for gateway.disk_status — the /api/status disk rollup (NS-656)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway import disk_status
|
||||
from gateway.disk_status import classify_disk_pressure, collect_disk_status
|
||||
|
||||
_GB = 1024 # MB per GB, for readable fixtures
|
||||
|
||||
|
||||
class TestClassifyDiskPressure:
|
||||
def test_plentiful_disk_is_ok(self) -> None:
|
||||
# 40 GB free of 100 GB.
|
||||
assert classify_disk_pressure(40 * _GB, 100 * _GB) == "ok"
|
||||
|
||||
def test_absolute_floor_is_critical_on_any_volume(self) -> None:
|
||||
# 200 MB free — below the 256 MB floor even on a huge, low-percent
|
||||
# volume would be impossible, so use a big volume mostly full.
|
||||
assert classify_disk_pressure(200, 500 * _GB) == "critical"
|
||||
|
||||
def test_high_percent_with_low_headroom_is_critical(self) -> None:
|
||||
# 96% used, 800 MB free (< 1 GB headroom) on a 20 GB volume.
|
||||
assert classify_disk_pressure(800, 20 * _GB) == "critical"
|
||||
|
||||
def test_high_percent_with_ample_headroom_is_not_critical(self) -> None:
|
||||
# 96% used but 20 GB free on a 500 GB volume — percent alone must
|
||||
# not trigger critical when absolute headroom is comfortable.
|
||||
assert classify_disk_pressure(20 * _GB, 500 * _GB) == "ok"
|
||||
|
||||
def test_low_free_is_elevated(self) -> None:
|
||||
# 400 MB free of 4 GB (~90% used): below the 512 MB elevated floor,
|
||||
# above the 256 MB critical floor, and under the 95% critical
|
||||
# percent gate.
|
||||
assert classify_disk_pressure(400, 4 * _GB) == "elevated"
|
||||
|
||||
def test_elevated_percent_band(self) -> None:
|
||||
# 88% used, 2.4 GB free of 20 GB — elevated percent gate with
|
||||
# headroom under 4 GB.
|
||||
assert classify_disk_pressure(2400, 20 * _GB) == "elevated"
|
||||
|
||||
def test_elevated_percent_with_ample_headroom_is_ok(self) -> None:
|
||||
# 90% used but 50 GB free of 500 GB.
|
||||
assert classify_disk_pressure(50 * _GB, 500 * _GB) == "ok"
|
||||
|
||||
def test_missing_sample_is_unknown(self) -> None:
|
||||
assert classify_disk_pressure(None, None) == "unknown"
|
||||
|
||||
def test_malformed_sample_is_unknown(self) -> None:
|
||||
assert classify_disk_pressure("lots", 100) == "unknown"
|
||||
assert classify_disk_pressure(True, 100) == "unknown"
|
||||
assert classify_disk_pressure(-5, 100) == "unknown"
|
||||
assert classify_disk_pressure(100, 0) == "unknown"
|
||||
|
||||
|
||||
class TestCollectDiskStatus:
|
||||
def test_reports_real_usage(self, tmp_path: Path) -> None:
|
||||
status = collect_disk_status(tmp_path)
|
||||
assert status["pressure"] in {"ok", "elevated", "critical"}
|
||||
assert isinstance(status["total_mb"], int) and status["total_mb"] > 0
|
||||
assert isinstance(status["free_mb"], int) and status["free_mb"] >= 0
|
||||
assert isinstance(status["used_percent"], float)
|
||||
assert 0.0 <= status["used_percent"] <= 100.0
|
||||
|
||||
def test_unreadable_filesystem_degrades_to_unknown(
|
||||
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
def _boom(_path): # noqa: ANN001, ANN202
|
||||
raise OSError("statvfs failed")
|
||||
|
||||
monkeypatch.setattr(shutil, "disk_usage", _boom)
|
||||
status = collect_disk_status(tmp_path)
|
||||
assert status == {
|
||||
"pressure": "unknown",
|
||||
"total_mb": None,
|
||||
"free_mb": None,
|
||||
"used_percent": None,
|
||||
}
|
||||
|
||||
def test_zero_total_degrades_to_unknown(
|
||||
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
fake = shutil._ntuple_diskusage(total=0, used=0, free=0) # type: ignore[attr-defined]
|
||||
monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
|
||||
status = collect_disk_status(tmp_path)
|
||||
assert status["pressure"] == "unknown"
|
||||
assert status["total_mb"] is None
|
||||
|
||||
def test_synthetic_full_volume_is_critical(
|
||||
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# 10 GB volume with 100 MB free — the OOF-2/OOF-107 state.
|
||||
total = 10 * 1024**3
|
||||
free = 100 * 1024**2
|
||||
fake = shutil._ntuple_diskusage( # type: ignore[attr-defined]
|
||||
total=total, used=total - free, free=free
|
||||
)
|
||||
monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
|
||||
status = collect_disk_status(tmp_path)
|
||||
assert status["pressure"] == "critical"
|
||||
assert status["total_mb"] == 10 * 1024
|
||||
assert status["free_mb"] == 100
|
||||
assert status["used_percent"] == 99.0
|
||||
|
||||
def test_never_raises_even_without_home(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# Default-home resolution failing must degrade, not raise.
|
||||
monkeypatch.setattr(
|
||||
disk_status.shutil,
|
||||
"disk_usage",
|
||||
lambda _p: (_ for _ in ()).throw(PermissionError("nope")),
|
||||
)
|
||||
assert collect_disk_status(None)["pressure"] == "unknown"
|
||||
@@ -3028,6 +3028,26 @@ class TestStatusMemoryBlock:
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["memory"] == {"pressure": "unknown"}
|
||||
|
||||
def test_disk_block_present_with_pressure_field(self):
|
||||
data = self.client.get("/api/status").json()
|
||||
assert "disk" in data
|
||||
assert data["disk"]["pressure"] in {
|
||||
"ok", "elevated", "critical", "unknown",
|
||||
}
|
||||
|
||||
def test_disk_block_degrades_when_collector_raises(self, monkeypatch):
|
||||
"""Same contract as the memory block: a broken collector must never
|
||||
take down the status endpoint."""
|
||||
import gateway.disk_status as ds
|
||||
|
||||
def _boom(*_a, **_k):
|
||||
raise RuntimeError("collector exploded")
|
||||
|
||||
monkeypatch.setattr(ds, "collect_disk_status", _boom)
|
||||
resp = self.client.get("/api/status")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["disk"] == {"pressure": "unknown"}
|
||||
|
||||
|
||||
class TestGatewayUpdatedAtContract:
|
||||
"""Contract tests for /api/status ``gateway_updated_at``.
|
||||
|
||||
@@ -9,7 +9,11 @@ import type { ReactNode } from "react";
|
||||
|
||||
import { I18nProvider } from "@/i18n";
|
||||
import { MemoryPressureBanner } from "./MemoryPressureBanner";
|
||||
import type { StatusResponse, MemoryPressureStatus } from "@/lib/api";
|
||||
import type {
|
||||
StatusResponse,
|
||||
MemoryPressureStatus,
|
||||
DiskPressureStatus,
|
||||
} from "@/lib/api";
|
||||
|
||||
let container: HTMLDivElement;
|
||||
let root: Root;
|
||||
@@ -38,6 +42,13 @@ function statusWith(memory: MemoryPressureStatus | undefined): StatusResponse {
|
||||
return { memory } as StatusResponse;
|
||||
}
|
||||
|
||||
function statusWithDisk(
|
||||
disk: DiskPressureStatus | undefined,
|
||||
memory?: MemoryPressureStatus,
|
||||
): StatusResponse {
|
||||
return { memory, disk } as StatusResponse;
|
||||
}
|
||||
|
||||
function banner(): HTMLElement | null {
|
||||
return container.querySelector('[data-testid="memory-pressure-banner"]');
|
||||
}
|
||||
@@ -240,4 +251,188 @@ describe("MemoryPressureBanner", () => {
|
||||
);
|
||||
expect(banner()).toBeNull();
|
||||
});
|
||||
|
||||
it("renders nothing for healthy or unknown disk", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "ok", free_mb: 5000 })}
|
||||
/>,
|
||||
);
|
||||
expect(banner()).toBeNull();
|
||||
await rerender(
|
||||
<MemoryPressureBanner status={statusWithDisk({ pressure: "unknown" })} />,
|
||||
);
|
||||
expect(banner()).toBeNull();
|
||||
});
|
||||
|
||||
it("shows the disk-critical warning with free-space detail", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "critical", free_mb: 120 })}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("disk is almost full");
|
||||
expect(banner()?.textContent).toContain("(120 MB free)");
|
||||
});
|
||||
|
||||
it("shows the disk-elevated warning", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("disk is filling up");
|
||||
});
|
||||
|
||||
it("disk critical outranks memory critical", async () => {
|
||||
// Imminent data loss beats imminent restart.
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "critical", free_mb: 100 },
|
||||
{ pressure: "critical" },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("disk is almost full");
|
||||
});
|
||||
|
||||
it("memory OOM notice outranks disk elevated", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "elevated", free_mb: 900 },
|
||||
{ pressure: "ok", last_boot_suspected_oom: true },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("restarted unexpectedly");
|
||||
});
|
||||
|
||||
it("dismissing a disk warning does not mask a later memory warning", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
|
||||
/>,
|
||||
);
|
||||
const dismiss = container.querySelector(
|
||||
'[data-testid="memory-pressure-banner"] button',
|
||||
) as HTMLButtonElement;
|
||||
await act(async () => dismiss.click());
|
||||
expect(banner()).toBeNull();
|
||||
await rerender(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "elevated", free_mb: 900 },
|
||||
{ pressure: "elevated" },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("running low on memory");
|
||||
});
|
||||
|
||||
it("disk escalation to critical re-opens a dismissed disk banner", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
|
||||
/>,
|
||||
);
|
||||
const dismiss = container.querySelector(
|
||||
'[data-testid="memory-pressure-banner"] button',
|
||||
) as HTMLButtonElement;
|
||||
await act(async () => dismiss.click());
|
||||
expect(banner()).toBeNull();
|
||||
await rerender(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("disk is almost full");
|
||||
});
|
||||
|
||||
it("disk recovery to ok resets disk dismissals for the next episode", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
|
||||
/>,
|
||||
);
|
||||
const dismiss = container.querySelector(
|
||||
'[data-testid="memory-pressure-banner"] button',
|
||||
) as HTMLButtonElement;
|
||||
await act(async () => dismiss.click());
|
||||
// User frees space (or grows the disk)...
|
||||
await rerender(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "ok", free_mb: 8000 })}
|
||||
/>,
|
||||
);
|
||||
expect(banner()).toBeNull();
|
||||
// ...then the disk fills again: new episode surfaces.
|
||||
await rerender(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("disk is almost full");
|
||||
});
|
||||
|
||||
it("disk recovery does NOT reset memory dismissals (and vice versa)", async () => {
|
||||
// Dismiss a memory warning while disk is also elevated.
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "elevated", free_mb: 900 },
|
||||
{ pressure: "critical" },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
const dismiss = container.querySelector(
|
||||
'[data-testid="memory-pressure-banner"] button',
|
||||
) as HTMLButtonElement;
|
||||
// Trigger shown is memory critical (outranks disk elevated) — dismiss it.
|
||||
await act(async () => dismiss.click());
|
||||
// Disk warning is next in line and has its own key, so it surfaces...
|
||||
expect(banner()?.textContent).toContain("disk is filling up");
|
||||
const dismissDisk = container.querySelector(
|
||||
'[data-testid="memory-pressure-banner"] button',
|
||||
) as HTMLButtonElement;
|
||||
await act(async () => dismissDisk.click());
|
||||
expect(banner()).toBeNull();
|
||||
// Disk recovers; memory still critical — its dismissal must survive.
|
||||
await rerender(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "ok", free_mb: 8000 },
|
||||
{ pressure: "critical" },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
expect(banner()).toBeNull();
|
||||
});
|
||||
|
||||
it("a gateway reboot (boot_id change) re-opens a dismissed disk banner", async () => {
|
||||
await render(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "critical", free_mb: 150 },
|
||||
{ pressure: "ok", boot_id: "2026-08-13T01:00:00+00:00" },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
const dismiss = container.querySelector(
|
||||
'[data-testid="memory-pressure-banner"] button',
|
||||
) as HTMLButtonElement;
|
||||
await act(async () => dismiss.click());
|
||||
expect(banner()).toBeNull();
|
||||
// Restart with the disk still full: new boot, warning returns.
|
||||
await rerender(
|
||||
<MemoryPressureBanner
|
||||
status={statusWithDisk(
|
||||
{ pressure: "critical", free_mb: 150 },
|
||||
{ pressure: "ok", boot_id: "2026-08-13T02:00:00+00:00" },
|
||||
)}
|
||||
/>,
|
||||
);
|
||||
expect(banner()?.textContent).toContain("disk is almost full");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -4,31 +4,41 @@ import type { StatusResponse } from "@/lib/api";
|
||||
import { useI18n } from "@/i18n";
|
||||
|
||||
/**
|
||||
* App-wide warning banner for memory trouble (NS-656).
|
||||
* App-wide warning banner for resource trouble (NS-656): memory pressure
|
||||
* and disk exhaustion.
|
||||
*
|
||||
* Two independent triggers, worst-first:
|
||||
* 1. Live pressure — the gateway's heartbeat shows system memory in the
|
||||
* `elevated`/`critical` band right now.
|
||||
* 2. Post-mortem — the previous gateway life died uncleanly and its last
|
||||
* Triggers, worst-first:
|
||||
* 1. Disk critical — the HERMES_HOME volume is nearly full. Worst because
|
||||
* the failure mode is silent data loss (SQLite writes failing, sessions
|
||||
* and config not persisting), not just a restart (OOF-2/OOF-107).
|
||||
* 2. Memory critical — the gateway's heartbeat shows system memory in the
|
||||
* `critical` band right now.
|
||||
* 3. Post-mortem — the previous gateway life died uncleanly and its last
|
||||
* heartbeat showed near-exhausted memory (`last_boot_suspected_oom`).
|
||||
* This is a heuristic, not proof the OOM killer acted — copy says so.
|
||||
* 4. Disk elevated / 5. memory elevated — early warnings.
|
||||
*
|
||||
* Both previously died in server-side log files; a hosted agent could be
|
||||
* OOM-killed hourly while the dashboard looked healthy.
|
||||
* All of this previously died in server-side log files; a hosted agent
|
||||
* could be OOM-killed hourly or fill its disk completely while the
|
||||
* dashboard looked healthy.
|
||||
*
|
||||
* Dismissal semantics (session-scoped, sessionStorage):
|
||||
* - EVERY dismissal key embeds the reporting boot (`boot_id`), so a gateway
|
||||
* restart invalidates all of them. Without this, dismissing `critical`,
|
||||
* rebooting, and coming back still-critical would hide the NEW incident —
|
||||
* and mask the OOM notice too, since critical takes precedence.
|
||||
* - Within one boot, live-pressure dismissal masks only the dismissed
|
||||
* severity; escalation (elevated → critical) re-opens immediately, and a
|
||||
* confirmed recovery (pressure back to "ok", not "unknown") clears live
|
||||
* dismissals so the NEXT episode in the same boot surfaces again.
|
||||
* and mask the OOM notice too, since critical takes precedence. Disk
|
||||
* entries share the scheme: disk state has no boot relationship, but
|
||||
* re-surfacing a still-full disk after a restart is the desired behavior.
|
||||
* - Within one boot, dismissal masks only the dismissed trigger; escalation
|
||||
* (elevated → critical, in either domain) re-opens immediately, and a
|
||||
* confirmed recovery (pressure back to "ok", not "unknown") clears that
|
||||
* domain's live dismissals so the NEXT episode in the same boot surfaces
|
||||
* again.
|
||||
*/
|
||||
|
||||
const STORAGE_KEY = "memoryBannerDismissed";
|
||||
const LIVE_TRIGGERS = ["critical", "elevated"];
|
||||
const MEMORY_LIVE_TRIGGERS = ["critical", "elevated"];
|
||||
const DISK_LIVE_TRIGGERS = ["disk_critical", "disk_elevated"];
|
||||
|
||||
function readDismissed(): string[] {
|
||||
try {
|
||||
@@ -53,6 +63,11 @@ function writeDismissed(entries: string[]) {
|
||||
}
|
||||
}
|
||||
|
||||
function entryMatches(triggers: string[]) {
|
||||
return (entry: string) =>
|
||||
triggers.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
|
||||
}
|
||||
|
||||
export function MemoryPressureBanner({
|
||||
status,
|
||||
}: {
|
||||
@@ -60,49 +75,60 @@ export function MemoryPressureBanner({
|
||||
}) {
|
||||
const { t } = useI18n();
|
||||
const memory = status?.memory;
|
||||
const disk = status?.disk;
|
||||
const pressure = memory?.pressure;
|
||||
const diskPressure = disk?.pressure;
|
||||
|
||||
const [dismissed, setDismissed] = useState<string[]>(readDismissed);
|
||||
|
||||
// Recovery reset (render-time state adjustment — the sanctioned React
|
||||
// pattern for reacting to prop changes without an effect): once live
|
||||
// pressure is demonstrably back to "ok", any dismissed live-pressure
|
||||
// entries describe a PAST episode — drop them so the next one isn't
|
||||
// silently hidden. "unknown" (stale/absent heartbeat) is absence of
|
||||
// evidence, not recovery, and clears nothing. Cross-boot invalidation
|
||||
// doesn't need handling here: boot_id is part of every dismissal key.
|
||||
// pattern for reacting to prop changes without an effect): once a
|
||||
// domain's live pressure is demonstrably back to "ok", any dismissed
|
||||
// live entries for that domain describe a PAST episode — drop them so
|
||||
// the next one isn't silently hidden. "unknown" (stale/absent sample)
|
||||
// is absence of evidence, not recovery, and clears nothing. Each domain
|
||||
// recovers independently: a fixed disk must not un-dismiss a memory
|
||||
// warning or vice versa. Cross-boot invalidation doesn't need handling
|
||||
// here: boot_id is part of every dismissal key.
|
||||
const [prevPressure, setPrevPressure] = useState(pressure);
|
||||
if (pressure !== prevPressure) {
|
||||
const [prevDiskPressure, setPrevDiskPressure] = useState(diskPressure);
|
||||
if (pressure !== prevPressure || diskPressure !== prevDiskPressure) {
|
||||
setPrevPressure(pressure);
|
||||
const isLiveEntry = (entry: string) =>
|
||||
LIVE_TRIGGERS.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
|
||||
if (pressure === "ok" && dismissed.some(isLiveEntry)) {
|
||||
const next = dismissed.filter((entry) => !isLiveEntry(entry));
|
||||
writeDismissed(next);
|
||||
setDismissed(next);
|
||||
setPrevDiskPressure(diskPressure);
|
||||
const recovered: Array<(entry: string) => boolean> = [];
|
||||
if (pressure === "ok") recovered.push(entryMatches(MEMORY_LIVE_TRIGGERS));
|
||||
if (diskPressure === "ok") recovered.push(entryMatches(DISK_LIVE_TRIGGERS));
|
||||
if (recovered.length > 0) {
|
||||
const isRecovered = (entry: string) =>
|
||||
recovered.some((match) => match(entry));
|
||||
if (dismissed.some(isRecovered)) {
|
||||
const next = dismissed.filter((entry) => !isRecovered(entry));
|
||||
writeDismissed(next);
|
||||
setDismissed(next);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Highest-severity active trigger, or null.
|
||||
const trigger = !memory
|
||||
? null
|
||||
: memory.pressure === "critical"
|
||||
? "critical"
|
||||
: memory.last_boot_suspected_oom
|
||||
? "oom_restart"
|
||||
: memory.pressure === "elevated"
|
||||
? "elevated"
|
||||
: null;
|
||||
// Active triggers, worst-first. Disk critical outranks memory critical:
|
||||
// imminent data loss beats imminent restart. Dismissal cascades — hiding
|
||||
// the top trigger surfaces the next one rather than silencing everything.
|
||||
const activeTriggers: string[] = [];
|
||||
if (diskPressure === "critical") activeTriggers.push("disk_critical");
|
||||
if (memory?.pressure === "critical") activeTriggers.push("critical");
|
||||
if (memory?.last_boot_suspected_oom) activeTriggers.push("oom_restart");
|
||||
if (diskPressure === "elevated") activeTriggers.push("disk_elevated");
|
||||
if (memory?.pressure === "elevated") activeTriggers.push("elevated");
|
||||
|
||||
// Every dismissal is scoped to the reporting boot: `boot_id` changes on
|
||||
// each gateway life, so restarts invalidate prior dismissals of ANY kind.
|
||||
// A missing boot_id (degraded payload / pre-NS-656 image) degrades to a
|
||||
// shared per-severity bucket — old behavior, never a crash.
|
||||
const dismissKey = trigger
|
||||
? `${trigger}:${memory?.boot_id ?? "unknown"}`
|
||||
: null;
|
||||
const keyFor = (trig: string) => `${trig}:${memory?.boot_id ?? "unknown"}`;
|
||||
const trigger =
|
||||
activeTriggers.find((trig) => !dismissed.includes(keyFor(trig))) ?? null;
|
||||
const dismissKey = trigger ? keyFor(trigger) : null;
|
||||
|
||||
if (!trigger || !dismissKey || dismissed.includes(dismissKey)) return null;
|
||||
if (!trigger || !dismissKey) return null;
|
||||
|
||||
const dismiss = () => {
|
||||
setDismissed((prev) => {
|
||||
@@ -112,16 +138,28 @@ export function MemoryPressureBanner({
|
||||
});
|
||||
};
|
||||
|
||||
const critical = trigger === "critical";
|
||||
const critical = trigger === "critical" || trigger === "disk_critical";
|
||||
const diskFreeLabel =
|
||||
disk?.free_mb != null ? ` (${Math.round(disk.free_mb)} MB free)` : "";
|
||||
const message =
|
||||
trigger === "oom_restart"
|
||||
? (t.app.memoryOomRestartBanner ??
|
||||
"Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
|
||||
: critical
|
||||
? (t.app.memoryCriticalBanner ??
|
||||
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
|
||||
: (t.app.memoryElevatedBanner ??
|
||||
"Your agent is running low on memory.");
|
||||
trigger === "disk_critical"
|
||||
? `${
|
||||
t.app.diskCriticalBanner ??
|
||||
"Your agent's disk is almost full. New messages, memories, and settings may fail to save."
|
||||
}${diskFreeLabel}`
|
||||
: trigger === "disk_elevated"
|
||||
? `${
|
||||
t.app.diskElevatedBanner ??
|
||||
"Your agent's disk is filling up. Consider clearing old sessions or expanding its storage."
|
||||
}${diskFreeLabel}`
|
||||
: trigger === "oom_restart"
|
||||
? (t.app.memoryOomRestartBanner ??
|
||||
"Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
|
||||
: critical
|
||||
? (t.app.memoryCriticalBanner ??
|
||||
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
|
||||
: (t.app.memoryElevatedBanner ??
|
||||
"Your agent is running low on memory.");
|
||||
|
||||
return (
|
||||
<div
|
||||
|
||||
@@ -102,6 +102,10 @@ export const en: Translations = {
|
||||
memoryCriticalBanner:
|
||||
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.",
|
||||
memoryElevatedBanner: "Your agent is running low on memory.",
|
||||
diskCriticalBanner:
|
||||
"Your agent's disk is almost full. New messages, memories, and settings may fail to save.",
|
||||
diskElevatedBanner:
|
||||
"Your agent's disk is filling up. Consider clearing old sessions or expanding its storage.",
|
||||
dismiss: "Dismiss",
|
||||
},
|
||||
|
||||
|
||||
@@ -119,6 +119,9 @@ export interface Translations {
|
||||
memoryOomRestartBanner?: string;
|
||||
memoryCriticalBanner?: string;
|
||||
memoryElevatedBanner?: string;
|
||||
/** NS-656 disk-usage banner — optional, English fallback. */
|
||||
diskCriticalBanner?: string;
|
||||
diskElevatedBanner?: string;
|
||||
dismiss?: string;
|
||||
};
|
||||
|
||||
|
||||
@@ -1885,6 +1885,9 @@ export interface StatusResponse {
|
||||
/** NS-656: memory-pressure rollup from the gateway heartbeat +
|
||||
* lifecycle ledger. Absent on older gateways. */
|
||||
memory?: MemoryPressureStatus;
|
||||
/** NS-656: disk-usage rollup for the HERMES_HOME volume. Absent on
|
||||
* older gateways. */
|
||||
disk?: DiskPressureStatus;
|
||||
release_date: string;
|
||||
version: string;
|
||||
}
|
||||
@@ -1907,6 +1910,16 @@ export interface MemoryPressureStatus {
|
||||
boot_id?: string | null;
|
||||
}
|
||||
|
||||
/** NS-656: coarse disk telemetry served by /api/status. Live statvfs
|
||||
* sample of the HERMES_HOME volume — no staleness dimension, so no
|
||||
* sampled_at. */
|
||||
export interface DiskPressureStatus {
|
||||
pressure: "ok" | "elevated" | "critical" | "unknown";
|
||||
total_mb?: number | null;
|
||||
free_mb?: number | null;
|
||||
used_percent?: number | null;
|
||||
}
|
||||
|
||||
export interface SessionInfo {
|
||||
id: string;
|
||||
source: string | null;
|
||||
|
||||
Reference in New Issue
Block a user