feat(gateway): disk-usage telemetry + dashboard disk-pressure banner (NS-656)

Extends the NS-656 memory-pressure surface to cover disk exhaustion
(OOF-2 / OOF-107 lineage: agents fill their data volume — SQLite writes
fail, sessions stop persisting — while every dashboard looks healthy).

- gateway/disk_status.py (new): collect_disk_status() samples
  shutil.disk_usage(HERMES_HOME) and classifies pressure
  (critical: <256 MB free or >=95% used; elevated: <512 MB free).
  Never raises — degrades to pressure="unknown" with null telemetry,
  same contract as collect_memory_status().
- /api/status: sibling `disk` block next to `memory`, advisory only —
  not folded into component/overall health.
- web: DiskPressureStatus type; MemoryPressureBanner generalized to a
  resource banner with worst-first triggers (disk critical > memory
  critical > OOM restart > disk elevated > memory elevated) and
  cascading dismissals — hiding the top trigger surfaces the next one
  instead of silencing everything. All dismissals stay boot_id-scoped.
- i18n: diskCriticalBanner / diskElevatedBanner (en, optional fields
  with English fallback per existing pattern).

Tests: gateway/test_disk_status.py (14), web_server disk-block
presence/degradation, banner disk trigger/priority/dismissal-cascade
suite (21 total).
This commit is contained in:
Shannon Sands
2026-08-13 18:58:39 +10:00
committed by Teknium
parent f5a26b1575
commit 6977d21fa7
9 changed files with 576 additions and 50 deletions
+117
View File
@@ -0,0 +1,117 @@
"""Disk-usage rollup for ``/api/status`` (NS-656).
Companion to :mod:`gateway.memory_status`, closing the same class of gap
for storage: a hosted agent can fill its data volume completely — SQLite
writes failing, session persistence dead, config saves lost — while its
dashboard and the NAS agent card both look perfectly healthy. Fleet
incidents OOF-2 (unrecoverable disk-full) and OOF-107 (fleet-wide disk
exhaustion, remediated by hand) are exactly this failure mode.
The readiness endpoint already probes disk (``gateway/readiness.py::
_probe_disk``), but readiness is a component verdict, not user-facing
telemetry — nothing renders it. This module produces the public block
the dashboard SPA and the NAS availability sweep actually consume.
Unlike the memory block (which distills already-persisted heartbeat
files), disk is sampled live via :func:`shutil.disk_usage` — a single
``statvfs`` call, the same thing the readiness probe does per request.
There is no meaningful "staleness" dimension, so no ``sampled_at``.
Public-safety note: ``/api/status`` is an unauthenticated liveness probe
(``PUBLIC_API_PATHS``). This block carries only coarse numbers (MB
granularity, whole-percent usage) and an enum — the same disclosure
class as the ``memory`` block.
Everything is best-effort and read-only: an unreadable filesystem
degrades to ``pressure="unknown"`` rather than raising into the status
endpoint.
"""
from __future__ import annotations
import logging
import shutil
from pathlib import Path
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
# Disk-pressure thresholds. Percent alone misleads in both directions:
# 90% used on a 100 GB volume leaves a comfortable 10 GB, while 50% used
# on a tiny volume can be one image download from write failures. So the
# percent triggers are gated on absolute headroom also being low, and a
# hard absolute floor applies regardless of size — below it, SQLite
# journaling and config writes are at genuine risk on any volume.
_CRITICAL_FREE_MB = 256 # < 256 MB free: critical on any volume
_CRITICAL_PERCENT = 95.0 # >= 95% used AND < 1 GB free: critical
_CRITICAL_HEADROOM_MB = 1024
_ELEVATED_FREE_MB = 512 # < 512 MB free: elevated on any volume
_ELEVATED_PERCENT = 85.0 # >= 85% used AND < 4 GB free: elevated
_ELEVATED_HEADROOM_MB = 4096
_BYTES_PER_MB = 1024 * 1024
def _coerce_mb(value: Any) -> Optional[int]:
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
return None
return value
def classify_disk_pressure(free_mb: Any, total_mb: Any) -> str:
"""Map free/total MB to ``ok``/``elevated``/``critical``.
``unknown`` when the sample is missing or malformed — the caller must
not treat "we could not read it" as "disk is fine".
"""
free = _coerce_mb(free_mb)
total = _coerce_mb(total_mb)
if free is None or total is None or total <= 0:
return "unknown"
used_percent = (1 - free / total) * 100.0
if free < _CRITICAL_FREE_MB or (
used_percent >= _CRITICAL_PERCENT and free < _CRITICAL_HEADROOM_MB
):
return "critical"
if free < _ELEVATED_FREE_MB or (
used_percent >= _ELEVATED_PERCENT and free < _ELEVATED_HEADROOM_MB
):
return "elevated"
return "ok"
def collect_disk_status(home: Optional[Path] = None) -> Dict[str, Any]:
"""Build the ``disk`` block for ``/api/status``.
``home`` scopes the sample to a profile's HERMES_HOME (the status
endpoint's ``?profile=`` handling passes it through); on hosted
images every profile shares the ``/opt/data`` volume, so the answer
is the same — but scoping keeps the contract identical to the
``memory`` block's.
Always returns a dict — an unreadable/unmounted filesystem yields
``{"pressure": "unknown", ...}``. Never raises.
"""
status: Dict[str, Any] = {
"pressure": "unknown",
"total_mb": None,
"free_mb": None,
"used_percent": None,
}
try:
if home is None:
from hermes_constants import get_hermes_home
home = get_hermes_home()
usage = shutil.disk_usage(home)
except Exception:
return status
if usage.total <= 0:
return status
total_mb = usage.total // _BYTES_PER_MB
free_mb = usage.free // _BYTES_PER_MB
status["total_mb"] = total_mb
status["free_mb"] = free_mb
status["used_percent"] = round((usage.used / usage.total) * 100, 1)
status["pressure"] = classify_disk_pressure(free_mb, total_mb)
return status
+18
View File
@@ -3354,6 +3354,24 @@ async def get_status(profile: Optional[str] = None):
except Exception: except Exception:
status["memory"] = {"pressure": "unknown"} status["memory"] = {"pressure": "unknown"}
# Disk-usage rollup (NS-656, same lineage as OOF-2/OOF-107 fleet
# disk-exhaustion incidents). One statvfs call on HERMES_HOME's
# filesystem — coarse MB numbers + enum, same public disclosure
# class as the memory block, and equally advisory: not folded
# into components/overall.
try:
from gateway.disk_status import collect_disk_status
status["disk"] = await asyncio.get_running_loop().run_in_executor(
None,
functools.partial(
collect_disk_status,
profile_dir if profile_dir else get_hermes_home(),
),
)
except Exception:
status["disk"] = {"pressure": "unknown"}
# Deferred FTS rebuild progress (schema v23): lets the desktop / # Deferred FTS rebuild progress (schema v23): lets the desktop /
# dashboard render a "search index rebuilding: N%" indicator instead # dashboard render a "search index rebuilding: N%" indicator instead
# of users wondering why old-message search is slower after an # of users wondering why old-message search is slower after an
+118
View File
@@ -0,0 +1,118 @@
"""Tests for gateway.disk_status — the /api/status disk rollup (NS-656)."""
from __future__ import annotations
import shutil
from pathlib import Path
import pytest
from gateway import disk_status
from gateway.disk_status import classify_disk_pressure, collect_disk_status
_GB = 1024 # MB per GB, for readable fixtures
class TestClassifyDiskPressure:
def test_plentiful_disk_is_ok(self) -> None:
# 40 GB free of 100 GB.
assert classify_disk_pressure(40 * _GB, 100 * _GB) == "ok"
def test_absolute_floor_is_critical_on_any_volume(self) -> None:
# 200 MB free — below the 256 MB floor even on a huge, low-percent
# volume would be impossible, so use a big volume mostly full.
assert classify_disk_pressure(200, 500 * _GB) == "critical"
def test_high_percent_with_low_headroom_is_critical(self) -> None:
# 96% used, 800 MB free (< 1 GB headroom) on a 20 GB volume.
assert classify_disk_pressure(800, 20 * _GB) == "critical"
def test_high_percent_with_ample_headroom_is_not_critical(self) -> None:
# 96% used but 20 GB free on a 500 GB volume — percent alone must
# not trigger critical when absolute headroom is comfortable.
assert classify_disk_pressure(20 * _GB, 500 * _GB) == "ok"
def test_low_free_is_elevated(self) -> None:
# 400 MB free of 4 GB (~90% used): below the 512 MB elevated floor,
# above the 256 MB critical floor, and under the 95% critical
# percent gate.
assert classify_disk_pressure(400, 4 * _GB) == "elevated"
def test_elevated_percent_band(self) -> None:
# 88% used, 2.4 GB free of 20 GB — elevated percent gate with
# headroom under 4 GB.
assert classify_disk_pressure(2400, 20 * _GB) == "elevated"
def test_elevated_percent_with_ample_headroom_is_ok(self) -> None:
# 90% used but 50 GB free of 500 GB.
assert classify_disk_pressure(50 * _GB, 500 * _GB) == "ok"
def test_missing_sample_is_unknown(self) -> None:
assert classify_disk_pressure(None, None) == "unknown"
def test_malformed_sample_is_unknown(self) -> None:
assert classify_disk_pressure("lots", 100) == "unknown"
assert classify_disk_pressure(True, 100) == "unknown"
assert classify_disk_pressure(-5, 100) == "unknown"
assert classify_disk_pressure(100, 0) == "unknown"
class TestCollectDiskStatus:
def test_reports_real_usage(self, tmp_path: Path) -> None:
status = collect_disk_status(tmp_path)
assert status["pressure"] in {"ok", "elevated", "critical"}
assert isinstance(status["total_mb"], int) and status["total_mb"] > 0
assert isinstance(status["free_mb"], int) and status["free_mb"] >= 0
assert isinstance(status["used_percent"], float)
assert 0.0 <= status["used_percent"] <= 100.0
def test_unreadable_filesystem_degrades_to_unknown(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
def _boom(_path): # noqa: ANN001, ANN202
raise OSError("statvfs failed")
monkeypatch.setattr(shutil, "disk_usage", _boom)
status = collect_disk_status(tmp_path)
assert status == {
"pressure": "unknown",
"total_mb": None,
"free_mb": None,
"used_percent": None,
}
def test_zero_total_degrades_to_unknown(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
fake = shutil._ntuple_diskusage(total=0, used=0, free=0) # type: ignore[attr-defined]
monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
status = collect_disk_status(tmp_path)
assert status["pressure"] == "unknown"
assert status["total_mb"] is None
def test_synthetic_full_volume_is_critical(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
# 10 GB volume with 100 MB free — the OOF-2/OOF-107 state.
total = 10 * 1024**3
free = 100 * 1024**2
fake = shutil._ntuple_diskusage( # type: ignore[attr-defined]
total=total, used=total - free, free=free
)
monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
status = collect_disk_status(tmp_path)
assert status["pressure"] == "critical"
assert status["total_mb"] == 10 * 1024
assert status["free_mb"] == 100
assert status["used_percent"] == 99.0
def test_never_raises_even_without_home(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
# Default-home resolution failing must degrade, not raise.
monkeypatch.setattr(
disk_status.shutil,
"disk_usage",
lambda _p: (_ for _ in ()).throw(PermissionError("nope")),
)
assert collect_disk_status(None)["pressure"] == "unknown"
+20
View File
@@ -3028,6 +3028,26 @@ class TestStatusMemoryBlock:
assert resp.status_code == 200 assert resp.status_code == 200
assert resp.json()["memory"] == {"pressure": "unknown"} assert resp.json()["memory"] == {"pressure": "unknown"}
def test_disk_block_present_with_pressure_field(self):
data = self.client.get("/api/status").json()
assert "disk" in data
assert data["disk"]["pressure"] in {
"ok", "elevated", "critical", "unknown",
}
def test_disk_block_degrades_when_collector_raises(self, monkeypatch):
"""Same contract as the memory block: a broken collector must never
take down the status endpoint."""
import gateway.disk_status as ds
def _boom(*_a, **_k):
raise RuntimeError("collector exploded")
monkeypatch.setattr(ds, "collect_disk_status", _boom)
resp = self.client.get("/api/status")
assert resp.status_code == 200
assert resp.json()["disk"] == {"pressure": "unknown"}
class TestGatewayUpdatedAtContract: class TestGatewayUpdatedAtContract:
"""Contract tests for /api/status ``gateway_updated_at``. """Contract tests for /api/status ``gateway_updated_at``.
@@ -9,7 +9,11 @@ import type { ReactNode } from "react";
import { I18nProvider } from "@/i18n"; import { I18nProvider } from "@/i18n";
import { MemoryPressureBanner } from "./MemoryPressureBanner"; import { MemoryPressureBanner } from "./MemoryPressureBanner";
import type { StatusResponse, MemoryPressureStatus } from "@/lib/api"; import type {
StatusResponse,
MemoryPressureStatus,
DiskPressureStatus,
} from "@/lib/api";
let container: HTMLDivElement; let container: HTMLDivElement;
let root: Root; let root: Root;
@@ -38,6 +42,13 @@ function statusWith(memory: MemoryPressureStatus | undefined): StatusResponse {
return { memory } as StatusResponse; return { memory } as StatusResponse;
} }
function statusWithDisk(
disk: DiskPressureStatus | undefined,
memory?: MemoryPressureStatus,
): StatusResponse {
return { memory, disk } as StatusResponse;
}
function banner(): HTMLElement | null { function banner(): HTMLElement | null {
return container.querySelector('[data-testid="memory-pressure-banner"]'); return container.querySelector('[data-testid="memory-pressure-banner"]');
} }
@@ -240,4 +251,188 @@ describe("MemoryPressureBanner", () => {
); );
expect(banner()).toBeNull(); expect(banner()).toBeNull();
}); });
it("renders nothing for healthy or unknown disk", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "ok", free_mb: 5000 })}
/>,
);
expect(banner()).toBeNull();
await rerender(
<MemoryPressureBanner status={statusWithDisk({ pressure: "unknown" })} />,
);
expect(banner()).toBeNull();
});
it("shows the disk-critical warning with free-space detail", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 120 })}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
expect(banner()?.textContent).toContain("(120 MB free)");
});
it("shows the disk-elevated warning", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
/>,
);
expect(banner()?.textContent).toContain("disk is filling up");
});
it("disk critical outranks memory critical", async () => {
// Imminent data loss beats imminent restart.
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "critical", free_mb: 100 },
{ pressure: "critical" },
)}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
it("memory OOM notice outranks disk elevated", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "elevated", free_mb: 900 },
{ pressure: "ok", last_boot_suspected_oom: true },
)}
/>,
);
expect(banner()?.textContent).toContain("restarted unexpectedly");
});
it("dismissing a disk warning does not mask a later memory warning", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
expect(banner()).toBeNull();
await rerender(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "elevated", free_mb: 900 },
{ pressure: "elevated" },
)}
/>,
);
expect(banner()?.textContent).toContain("running low on memory");
});
it("disk escalation to critical re-opens a dismissed disk banner", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
expect(banner()).toBeNull();
await rerender(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
it("disk recovery to ok resets disk dismissals for the next episode", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
// User frees space (or grows the disk)...
await rerender(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "ok", free_mb: 8000 })}
/>,
);
expect(banner()).toBeNull();
// ...then the disk fills again: new episode surfaces.
await rerender(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
it("disk recovery does NOT reset memory dismissals (and vice versa)", async () => {
// Dismiss a memory warning while disk is also elevated.
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "elevated", free_mb: 900 },
{ pressure: "critical" },
)}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
// Trigger shown is memory critical (outranks disk elevated) — dismiss it.
await act(async () => dismiss.click());
// Disk warning is next in line and has its own key, so it surfaces...
expect(banner()?.textContent).toContain("disk is filling up");
const dismissDisk = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismissDisk.click());
expect(banner()).toBeNull();
// Disk recovers; memory still critical — its dismissal must survive.
await rerender(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "ok", free_mb: 8000 },
{ pressure: "critical" },
)}
/>,
);
expect(banner()).toBeNull();
});
it("a gateway reboot (boot_id change) re-opens a dismissed disk banner", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "critical", free_mb: 150 },
{ pressure: "ok", boot_id: "2026-08-13T01:00:00+00:00" },
)}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
expect(banner()).toBeNull();
// Restart with the disk still full: new boot, warning returns.
await rerender(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "critical", free_mb: 150 },
{ pressure: "ok", boot_id: "2026-08-13T02:00:00+00:00" },
)}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
}); });
+87 -49
View File
@@ -4,31 +4,41 @@ import type { StatusResponse } from "@/lib/api";
import { useI18n } from "@/i18n"; import { useI18n } from "@/i18n";
/** /**
* App-wide warning banner for memory trouble (NS-656). * App-wide warning banner for resource trouble (NS-656): memory pressure
* and disk exhaustion.
* *
* Two independent triggers, worst-first: * Triggers, worst-first:
* 1. Live pressure — the gateway's heartbeat shows system memory in the * 1. Disk critical — the HERMES_HOME volume is nearly full. Worst because
* `elevated`/`critical` band right now. * the failure mode is silent data loss (SQLite writes failing, sessions
* 2. Post-mortem — the previous gateway life died uncleanly and its last * and config not persisting), not just a restart (OOF-2/OOF-107).
* 2. Memory critical — the gateway's heartbeat shows system memory in the
* `critical` band right now.
* 3. Post-mortem — the previous gateway life died uncleanly and its last
* heartbeat showed near-exhausted memory (`last_boot_suspected_oom`). * heartbeat showed near-exhausted memory (`last_boot_suspected_oom`).
* This is a heuristic, not proof the OOM killer acted — copy says so. * This is a heuristic, not proof the OOM killer acted — copy says so.
* 4. Disk elevated / 5. memory elevated — early warnings.
* *
* Both previously died in server-side log files; a hosted agent could be * All of this previously died in server-side log files; a hosted agent
* OOM-killed hourly while the dashboard looked healthy. * could be OOM-killed hourly or fill its disk completely while the
* dashboard looked healthy.
* *
* Dismissal semantics (session-scoped, sessionStorage): * Dismissal semantics (session-scoped, sessionStorage):
* - EVERY dismissal key embeds the reporting boot (`boot_id`), so a gateway * - EVERY dismissal key embeds the reporting boot (`boot_id`), so a gateway
* restart invalidates all of them. Without this, dismissing `critical`, * restart invalidates all of them. Without this, dismissing `critical`,
* rebooting, and coming back still-critical would hide the NEW incident — * rebooting, and coming back still-critical would hide the NEW incident —
* and mask the OOM notice too, since critical takes precedence. * and mask the OOM notice too, since critical takes precedence. Disk
* - Within one boot, live-pressure dismissal masks only the dismissed * entries share the scheme: disk state has no boot relationship, but
* severity; escalation (elevated → critical) re-opens immediately, and a * re-surfacing a still-full disk after a restart is the desired behavior.
* confirmed recovery (pressure back to "ok", not "unknown") clears live * - Within one boot, dismissal masks only the dismissed trigger; escalation
* dismissals so the NEXT episode in the same boot surfaces again. * (elevated → critical, in either domain) re-opens immediately, and a
* confirmed recovery (pressure back to "ok", not "unknown") clears that
* domain's live dismissals so the NEXT episode in the same boot surfaces
* again.
*/ */
const STORAGE_KEY = "memoryBannerDismissed"; const STORAGE_KEY = "memoryBannerDismissed";
const LIVE_TRIGGERS = ["critical", "elevated"]; const MEMORY_LIVE_TRIGGERS = ["critical", "elevated"];
const DISK_LIVE_TRIGGERS = ["disk_critical", "disk_elevated"];
function readDismissed(): string[] { function readDismissed(): string[] {
try { try {
@@ -53,6 +63,11 @@ function writeDismissed(entries: string[]) {
} }
} }
function entryMatches(triggers: string[]) {
return (entry: string) =>
triggers.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
}
export function MemoryPressureBanner({ export function MemoryPressureBanner({
status, status,
}: { }: {
@@ -60,49 +75,60 @@ export function MemoryPressureBanner({
}) { }) {
const { t } = useI18n(); const { t } = useI18n();
const memory = status?.memory; const memory = status?.memory;
const disk = status?.disk;
const pressure = memory?.pressure; const pressure = memory?.pressure;
const diskPressure = disk?.pressure;
const [dismissed, setDismissed] = useState<string[]>(readDismissed); const [dismissed, setDismissed] = useState<string[]>(readDismissed);
// Recovery reset (render-time state adjustment — the sanctioned React // Recovery reset (render-time state adjustment — the sanctioned React
// pattern for reacting to prop changes without an effect): once live // pattern for reacting to prop changes without an effect): once a
// pressure is demonstrably back to "ok", any dismissed live-pressure // domain's live pressure is demonstrably back to "ok", any dismissed
// entries describe a PAST episode — drop them so the next one isn't // live entries for that domain describe a PAST episode — drop them so
// silently hidden. "unknown" (stale/absent heartbeat) is absence of // the next one isn't silently hidden. "unknown" (stale/absent sample)
// evidence, not recovery, and clears nothing. Cross-boot invalidation // is absence of evidence, not recovery, and clears nothing. Each domain
// doesn't need handling here: boot_id is part of every dismissal key. // recovers independently: a fixed disk must not un-dismiss a memory
// warning or vice versa. Cross-boot invalidation doesn't need handling
// here: boot_id is part of every dismissal key.
const [prevPressure, setPrevPressure] = useState(pressure); const [prevPressure, setPrevPressure] = useState(pressure);
if (pressure !== prevPressure) { const [prevDiskPressure, setPrevDiskPressure] = useState(diskPressure);
if (pressure !== prevPressure || diskPressure !== prevDiskPressure) {
setPrevPressure(pressure); setPrevPressure(pressure);
const isLiveEntry = (entry: string) => setPrevDiskPressure(diskPressure);
LIVE_TRIGGERS.some((sev) => entry === sev || entry.startsWith(`${sev}:`)); const recovered: Array<(entry: string) => boolean> = [];
if (pressure === "ok" && dismissed.some(isLiveEntry)) { if (pressure === "ok") recovered.push(entryMatches(MEMORY_LIVE_TRIGGERS));
const next = dismissed.filter((entry) => !isLiveEntry(entry)); if (diskPressure === "ok") recovered.push(entryMatches(DISK_LIVE_TRIGGERS));
writeDismissed(next); if (recovered.length > 0) {
setDismissed(next); const isRecovered = (entry: string) =>
recovered.some((match) => match(entry));
if (dismissed.some(isRecovered)) {
const next = dismissed.filter((entry) => !isRecovered(entry));
writeDismissed(next);
setDismissed(next);
}
} }
} }
// Highest-severity active trigger, or null. // Active triggers, worst-first. Disk critical outranks memory critical:
const trigger = !memory // imminent data loss beats imminent restart. Dismissal cascades — hiding
? null // the top trigger surfaces the next one rather than silencing everything.
: memory.pressure === "critical" const activeTriggers: string[] = [];
? "critical" if (diskPressure === "critical") activeTriggers.push("disk_critical");
: memory.last_boot_suspected_oom if (memory?.pressure === "critical") activeTriggers.push("critical");
? "oom_restart" if (memory?.last_boot_suspected_oom) activeTriggers.push("oom_restart");
: memory.pressure === "elevated" if (diskPressure === "elevated") activeTriggers.push("disk_elevated");
? "elevated" if (memory?.pressure === "elevated") activeTriggers.push("elevated");
: null;
// Every dismissal is scoped to the reporting boot: `boot_id` changes on // Every dismissal is scoped to the reporting boot: `boot_id` changes on
// each gateway life, so restarts invalidate prior dismissals of ANY kind. // each gateway life, so restarts invalidate prior dismissals of ANY kind.
// A missing boot_id (degraded payload / pre-NS-656 image) degrades to a // A missing boot_id (degraded payload / pre-NS-656 image) degrades to a
// shared per-severity bucket — old behavior, never a crash. // shared per-severity bucket — old behavior, never a crash.
const dismissKey = trigger const keyFor = (trig: string) => `${trig}:${memory?.boot_id ?? "unknown"}`;
? `${trigger}:${memory?.boot_id ?? "unknown"}` const trigger =
: null; activeTriggers.find((trig) => !dismissed.includes(keyFor(trig))) ?? null;
const dismissKey = trigger ? keyFor(trigger) : null;
if (!trigger || !dismissKey || dismissed.includes(dismissKey)) return null; if (!trigger || !dismissKey) return null;
const dismiss = () => { const dismiss = () => {
setDismissed((prev) => { setDismissed((prev) => {
@@ -112,16 +138,28 @@ export function MemoryPressureBanner({
}); });
}; };
const critical = trigger === "critical"; const critical = trigger === "critical" || trigger === "disk_critical";
const diskFreeLabel =
disk?.free_mb != null ? ` (${Math.round(disk.free_mb)} MB free)` : "";
const message = const message =
trigger === "oom_restart" trigger === "disk_critical"
? (t.app.memoryOomRestartBanner ?? ? `${
"Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.") t.app.diskCriticalBanner ??
: critical "Your agent's disk is almost full. New messages, memories, and settings may fail to save."
? (t.app.memoryCriticalBanner ?? }${diskFreeLabel}`
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.") : trigger === "disk_elevated"
: (t.app.memoryElevatedBanner ?? ? `${
"Your agent is running low on memory."); t.app.diskElevatedBanner ??
"Your agent's disk is filling up. Consider clearing old sessions or expanding its storage."
}${diskFreeLabel}`
: trigger === "oom_restart"
? (t.app.memoryOomRestartBanner ??
"Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
: critical
? (t.app.memoryCriticalBanner ??
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
: (t.app.memoryElevatedBanner ??
"Your agent is running low on memory.");
return ( return (
<div <div
+4
View File
@@ -102,6 +102,10 @@ export const en: Translations = {
memoryCriticalBanner: memoryCriticalBanner:
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.", "Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.",
memoryElevatedBanner: "Your agent is running low on memory.", memoryElevatedBanner: "Your agent is running low on memory.",
diskCriticalBanner:
"Your agent's disk is almost full. New messages, memories, and settings may fail to save.",
diskElevatedBanner:
"Your agent's disk is filling up. Consider clearing old sessions or expanding its storage.",
dismiss: "Dismiss", dismiss: "Dismiss",
}, },
+3
View File
@@ -119,6 +119,9 @@ export interface Translations {
memoryOomRestartBanner?: string; memoryOomRestartBanner?: string;
memoryCriticalBanner?: string; memoryCriticalBanner?: string;
memoryElevatedBanner?: string; memoryElevatedBanner?: string;
/** NS-656 disk-usage banner — optional, English fallback. */
diskCriticalBanner?: string;
diskElevatedBanner?: string;
dismiss?: string; dismiss?: string;
}; };
+13
View File
@@ -1885,6 +1885,9 @@ export interface StatusResponse {
/** NS-656: memory-pressure rollup from the gateway heartbeat + /** NS-656: memory-pressure rollup from the gateway heartbeat +
* lifecycle ledger. Absent on older gateways. */ * lifecycle ledger. Absent on older gateways. */
memory?: MemoryPressureStatus; memory?: MemoryPressureStatus;
/** NS-656: disk-usage rollup for the HERMES_HOME volume. Absent on
* older gateways. */
disk?: DiskPressureStatus;
release_date: string; release_date: string;
version: string; version: string;
} }
@@ -1907,6 +1910,16 @@ export interface MemoryPressureStatus {
boot_id?: string | null; boot_id?: string | null;
} }
/** NS-656: coarse disk telemetry served by /api/status. Live statvfs
* sample of the HERMES_HOME volume — no staleness dimension, so no
* sampled_at. */
export interface DiskPressureStatus {
pressure: "ok" | "elevated" | "critical" | "unknown";
total_mb?: number | null;
free_mb?: number | null;
used_percent?: number | null;
}
export interface SessionInfo { export interface SessionInfo {
id: string; id: string;
source: string | null; source: string | null;