diff --git a/gateway/disk_status.py b/gateway/disk_status.py
new file mode 100644
index 0000000000..b762233cf3
--- /dev/null
+++ b/gateway/disk_status.py
@@ -0,0 +1,117 @@
+"""Disk-usage rollup for ``/api/status`` (NS-656).
+
+Companion to :mod:`gateway.memory_status`, closing the same class of gap
+for storage: a hosted agent can fill its data volume completely — SQLite
+writes failing, session persistence dead, config saves lost — while its
+dashboard and the NAS agent card both look perfectly healthy. Fleet
+incidents OOF-2 (unrecoverable disk-full) and OOF-107 (fleet-wide disk
+exhaustion, remediated by hand) are exactly this failure mode.
+
+The readiness endpoint already probes disk (``gateway/readiness.py::
+_probe_disk``), but readiness is a component verdict, not user-facing
+telemetry — nothing renders it. This module produces the public block
+the dashboard SPA and the NAS availability sweep actually consume.
+
+Unlike the memory block (which distills already-persisted heartbeat
+files), disk is sampled live via :func:`shutil.disk_usage` — a single
+``statvfs`` call, the same thing the readiness probe does per request.
+There is no meaningful "staleness" dimension, so no ``sampled_at``.
+
+Public-safety note: ``/api/status`` is an unauthenticated liveness probe
+(``PUBLIC_API_PATHS``). This block carries only coarse numbers (MB
+granularity, whole-percent usage) and an enum — the same disclosure
+class as the ``memory`` block.
+
+Everything is best-effort and read-only: an unreadable filesystem
+degrades to ``pressure="unknown"`` rather than raising into the status
+endpoint.
+"""
+
+from __future__ import annotations
+
+import logging
+import shutil
+from pathlib import Path
+from typing import Any, Dict, Optional
+
+logger = logging.getLogger(__name__)
+
+# Disk-pressure thresholds. Percent alone misleads in both directions:
+# 90% used on a 100 GB volume leaves a comfortable 10 GB, while 50% used
+# on a tiny volume can be one image download from write failures. So the
+# percent triggers are gated on absolute headroom also being low, and a
+# hard absolute floor applies regardless of size — below it, SQLite
+# journaling and config writes are at genuine risk on any volume.
+_CRITICAL_FREE_MB = 256 # < 256 MB free: critical on any volume
+_CRITICAL_PERCENT = 95.0 # >= 95% used AND < 1 GB free: critical
+_CRITICAL_HEADROOM_MB = 1024
+_ELEVATED_FREE_MB = 512 # < 512 MB free: elevated on any volume
+_ELEVATED_PERCENT = 85.0 # >= 85% used AND < 4 GB free: elevated
+_ELEVATED_HEADROOM_MB = 4096
+
+_BYTES_PER_MB = 1024 * 1024
+
+
+def _coerce_mb(value: Any) -> Optional[int]:
+ if isinstance(value, bool) or not isinstance(value, int) or value < 0:
+ return None
+ return value
+
+
+def classify_disk_pressure(free_mb: Any, total_mb: Any) -> str:
+ """Map free/total MB to ``ok``/``elevated``/``critical``.
+
+ ``unknown`` when the sample is missing or malformed — the caller must
+ not treat "we could not read it" as "disk is fine".
+ """
+ free = _coerce_mb(free_mb)
+ total = _coerce_mb(total_mb)
+ if free is None or total is None or total <= 0:
+ return "unknown"
+ used_percent = (1 - free / total) * 100.0
+ if free < _CRITICAL_FREE_MB or (
+ used_percent >= _CRITICAL_PERCENT and free < _CRITICAL_HEADROOM_MB
+ ):
+ return "critical"
+ if free < _ELEVATED_FREE_MB or (
+ used_percent >= _ELEVATED_PERCENT and free < _ELEVATED_HEADROOM_MB
+ ):
+ return "elevated"
+ return "ok"
+
+
+def collect_disk_status(home: Optional[Path] = None) -> Dict[str, Any]:
+ """Build the ``disk`` block for ``/api/status``.
+
+ ``home`` scopes the sample to a profile's HERMES_HOME (the status
+ endpoint's ``?profile=`` handling passes it through); on hosted
+ images every profile shares the ``/opt/data`` volume, so the answer
+ is the same — but scoping keeps the contract identical to the
+ ``memory`` block's.
+
+ Always returns a dict — an unreadable/unmounted filesystem yields
+ ``{"pressure": "unknown", ...}``. Never raises.
+ """
+ status: Dict[str, Any] = {
+ "pressure": "unknown",
+ "total_mb": None,
+ "free_mb": None,
+ "used_percent": None,
+ }
+ try:
+ if home is None:
+ from hermes_constants import get_hermes_home
+
+ home = get_hermes_home()
+ usage = shutil.disk_usage(home)
+ except Exception:
+ return status
+ if usage.total <= 0:
+ return status
+ total_mb = usage.total // _BYTES_PER_MB
+ free_mb = usage.free // _BYTES_PER_MB
+ status["total_mb"] = total_mb
+ status["free_mb"] = free_mb
+ status["used_percent"] = round((usage.used / usage.total) * 100, 1)
+ status["pressure"] = classify_disk_pressure(free_mb, total_mb)
+ return status
diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py
index 543e1c0b9a..365de042c3 100644
--- a/hermes_cli/web_server.py
+++ b/hermes_cli/web_server.py
@@ -3354,6 +3354,24 @@ async def get_status(profile: Optional[str] = None):
except Exception:
status["memory"] = {"pressure": "unknown"}
+ # Disk-usage rollup (NS-656, same lineage as OOF-2/OOF-107 fleet
+ # disk-exhaustion incidents). One statvfs call on HERMES_HOME's
+ # filesystem — coarse MB numbers + enum, same public disclosure
+ # class as the memory block, and equally advisory: not folded
+ # into components/overall.
+ try:
+ from gateway.disk_status import collect_disk_status
+
+ status["disk"] = await asyncio.get_running_loop().run_in_executor(
+ None,
+ functools.partial(
+ collect_disk_status,
+ profile_dir if profile_dir else get_hermes_home(),
+ ),
+ )
+ except Exception:
+ status["disk"] = {"pressure": "unknown"}
+
# Deferred FTS rebuild progress (schema v23): lets the desktop /
# dashboard render a "search index rebuilding: N%" indicator instead
# of users wondering why old-message search is slower after an
diff --git a/tests/gateway/test_disk_status.py b/tests/gateway/test_disk_status.py
new file mode 100644
index 0000000000..14ca98b4a2
--- /dev/null
+++ b/tests/gateway/test_disk_status.py
@@ -0,0 +1,118 @@
+"""Tests for gateway.disk_status — the /api/status disk rollup (NS-656)."""
+
+from __future__ import annotations
+
+import shutil
+from pathlib import Path
+
+import pytest
+
+from gateway import disk_status
+from gateway.disk_status import classify_disk_pressure, collect_disk_status
+
+_GB = 1024 # MB per GB, for readable fixtures
+
+
+class TestClassifyDiskPressure:
+ def test_plentiful_disk_is_ok(self) -> None:
+ # 40 GB free of 100 GB.
+ assert classify_disk_pressure(40 * _GB, 100 * _GB) == "ok"
+
+ def test_absolute_floor_is_critical_on_any_volume(self) -> None:
+ # 200 MB free — below the 256 MB floor even on a huge, low-percent
+ # volume would be impossible, so use a big volume mostly full.
+ assert classify_disk_pressure(200, 500 * _GB) == "critical"
+
+ def test_high_percent_with_low_headroom_is_critical(self) -> None:
+ # 96% used, 800 MB free (< 1 GB headroom) on a 20 GB volume.
+ assert classify_disk_pressure(800, 20 * _GB) == "critical"
+
+ def test_high_percent_with_ample_headroom_is_not_critical(self) -> None:
+ # 96% used but 20 GB free on a 500 GB volume — percent alone must
+ # not trigger critical when absolute headroom is comfortable.
+ assert classify_disk_pressure(20 * _GB, 500 * _GB) == "ok"
+
+ def test_low_free_is_elevated(self) -> None:
+ # 400 MB free of 4 GB (~90% used): below the 512 MB elevated floor,
+ # above the 256 MB critical floor, and under the 95% critical
+ # percent gate.
+ assert classify_disk_pressure(400, 4 * _GB) == "elevated"
+
+ def test_elevated_percent_band(self) -> None:
+ # 88% used, 2.4 GB free of 20 GB — elevated percent gate with
+ # headroom under 4 GB.
+ assert classify_disk_pressure(2400, 20 * _GB) == "elevated"
+
+ def test_elevated_percent_with_ample_headroom_is_ok(self) -> None:
+ # 90% used but 50 GB free of 500 GB.
+ assert classify_disk_pressure(50 * _GB, 500 * _GB) == "ok"
+
+ def test_missing_sample_is_unknown(self) -> None:
+ assert classify_disk_pressure(None, None) == "unknown"
+
+ def test_malformed_sample_is_unknown(self) -> None:
+ assert classify_disk_pressure("lots", 100) == "unknown"
+ assert classify_disk_pressure(True, 100) == "unknown"
+ assert classify_disk_pressure(-5, 100) == "unknown"
+ assert classify_disk_pressure(100, 0) == "unknown"
+
+
+class TestCollectDiskStatus:
+ def test_reports_real_usage(self, tmp_path: Path) -> None:
+ status = collect_disk_status(tmp_path)
+ assert status["pressure"] in {"ok", "elevated", "critical"}
+ assert isinstance(status["total_mb"], int) and status["total_mb"] > 0
+ assert isinstance(status["free_mb"], int) and status["free_mb"] >= 0
+ assert isinstance(status["used_percent"], float)
+ assert 0.0 <= status["used_percent"] <= 100.0
+
+ def test_unreadable_filesystem_degrades_to_unknown(
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+ ) -> None:
+ def _boom(_path): # noqa: ANN001, ANN202
+ raise OSError("statvfs failed")
+
+ monkeypatch.setattr(shutil, "disk_usage", _boom)
+ status = collect_disk_status(tmp_path)
+ assert status == {
+ "pressure": "unknown",
+ "total_mb": None,
+ "free_mb": None,
+ "used_percent": None,
+ }
+
+ def test_zero_total_degrades_to_unknown(
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+ ) -> None:
+ fake = shutil._ntuple_diskusage(total=0, used=0, free=0) # type: ignore[attr-defined]
+ monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
+ status = collect_disk_status(tmp_path)
+ assert status["pressure"] == "unknown"
+ assert status["total_mb"] is None
+
+ def test_synthetic_full_volume_is_critical(
+ self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+ ) -> None:
+ # 10 GB volume with 100 MB free — the OOF-2/OOF-107 state.
+ total = 10 * 1024**3
+ free = 100 * 1024**2
+ fake = shutil._ntuple_diskusage( # type: ignore[attr-defined]
+ total=total, used=total - free, free=free
+ )
+ monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
+ status = collect_disk_status(tmp_path)
+ assert status["pressure"] == "critical"
+ assert status["total_mb"] == 10 * 1024
+ assert status["free_mb"] == 100
+ assert status["used_percent"] == 99.0
+
+ def test_never_raises_even_without_home(
+ self, monkeypatch: pytest.MonkeyPatch
+ ) -> None:
+ # Default-home resolution failing must degrade, not raise.
+ monkeypatch.setattr(
+ disk_status.shutil,
+ "disk_usage",
+ lambda _p: (_ for _ in ()).throw(PermissionError("nope")),
+ )
+ assert collect_disk_status(None)["pressure"] == "unknown"
diff --git a/tests/hermes_cli/test_web_server.py b/tests/hermes_cli/test_web_server.py
index 5a06173807..ee655e0a3b 100644
--- a/tests/hermes_cli/test_web_server.py
+++ b/tests/hermes_cli/test_web_server.py
@@ -3028,6 +3028,26 @@ class TestStatusMemoryBlock:
assert resp.status_code == 200
assert resp.json()["memory"] == {"pressure": "unknown"}
+ def test_disk_block_present_with_pressure_field(self):
+ data = self.client.get("/api/status").json()
+ assert "disk" in data
+ assert data["disk"]["pressure"] in {
+ "ok", "elevated", "critical", "unknown",
+ }
+
+ def test_disk_block_degrades_when_collector_raises(self, monkeypatch):
+ """Same contract as the memory block: a broken collector must never
+ take down the status endpoint."""
+ import gateway.disk_status as ds
+
+ def _boom(*_a, **_k):
+ raise RuntimeError("collector exploded")
+
+ monkeypatch.setattr(ds, "collect_disk_status", _boom)
+ resp = self.client.get("/api/status")
+ assert resp.status_code == 200
+ assert resp.json()["disk"] == {"pressure": "unknown"}
+
class TestGatewayUpdatedAtContract:
"""Contract tests for /api/status ``gateway_updated_at``.
diff --git a/web/src/components/MemoryPressureBanner.test.tsx b/web/src/components/MemoryPressureBanner.test.tsx
index cf89247528..c1ee756e1b 100644
--- a/web/src/components/MemoryPressureBanner.test.tsx
+++ b/web/src/components/MemoryPressureBanner.test.tsx
@@ -9,7 +9,11 @@ import type { ReactNode } from "react";
import { I18nProvider } from "@/i18n";
import { MemoryPressureBanner } from "./MemoryPressureBanner";
-import type { StatusResponse, MemoryPressureStatus } from "@/lib/api";
+import type {
+ StatusResponse,
+ MemoryPressureStatus,
+ DiskPressureStatus,
+} from "@/lib/api";
let container: HTMLDivElement;
let root: Root;
@@ -38,6 +42,13 @@ function statusWith(memory: MemoryPressureStatus | undefined): StatusResponse {
return { memory } as StatusResponse;
}
+function statusWithDisk(
+ disk: DiskPressureStatus | undefined,
+ memory?: MemoryPressureStatus,
+): StatusResponse {
+ return { memory, disk } as StatusResponse;
+}
+
function banner(): HTMLElement | null {
return container.querySelector('[data-testid="memory-pressure-banner"]');
}
@@ -240,4 +251,188 @@ describe("MemoryPressureBanner", () => {
);
expect(banner()).toBeNull();
});
+
+ it("renders nothing for healthy or unknown disk", async () => {
+ await render(
+ ,
+ );
+ expect(banner()).toBeNull();
+ await rerender(
+ ,
+ );
+ expect(banner()).toBeNull();
+ });
+
+ it("shows the disk-critical warning with free-space detail", async () => {
+ await render(
+ ,
+ );
+ expect(banner()?.textContent).toContain("disk is almost full");
+ expect(banner()?.textContent).toContain("(120 MB free)");
+ });
+
+ it("shows the disk-elevated warning", async () => {
+ await render(
+ ,
+ );
+ expect(banner()?.textContent).toContain("disk is filling up");
+ });
+
+ it("disk critical outranks memory critical", async () => {
+ // Imminent data loss beats imminent restart.
+ await render(
+ ,
+ );
+ expect(banner()?.textContent).toContain("disk is almost full");
+ });
+
+ it("memory OOM notice outranks disk elevated", async () => {
+ await render(
+ ,
+ );
+ expect(banner()?.textContent).toContain("restarted unexpectedly");
+ });
+
+ it("dismissing a disk warning does not mask a later memory warning", async () => {
+ await render(
+ ,
+ );
+ const dismiss = container.querySelector(
+ '[data-testid="memory-pressure-banner"] button',
+ ) as HTMLButtonElement;
+ await act(async () => dismiss.click());
+ expect(banner()).toBeNull();
+ await rerender(
+ ,
+ );
+ expect(banner()?.textContent).toContain("running low on memory");
+ });
+
+ it("disk escalation to critical re-opens a dismissed disk banner", async () => {
+ await render(
+ ,
+ );
+ const dismiss = container.querySelector(
+ '[data-testid="memory-pressure-banner"] button',
+ ) as HTMLButtonElement;
+ await act(async () => dismiss.click());
+ expect(banner()).toBeNull();
+ await rerender(
+ ,
+ );
+ expect(banner()?.textContent).toContain("disk is almost full");
+ });
+
+ it("disk recovery to ok resets disk dismissals for the next episode", async () => {
+ await render(
+ ,
+ );
+ const dismiss = container.querySelector(
+ '[data-testid="memory-pressure-banner"] button',
+ ) as HTMLButtonElement;
+ await act(async () => dismiss.click());
+ // User frees space (or grows the disk)...
+ await rerender(
+ ,
+ );
+ expect(banner()).toBeNull();
+ // ...then the disk fills again: new episode surfaces.
+ await rerender(
+ ,
+ );
+ expect(banner()?.textContent).toContain("disk is almost full");
+ });
+
+ it("disk recovery does NOT reset memory dismissals (and vice versa)", async () => {
+ // Dismiss a memory warning while disk is also elevated.
+ await render(
+ ,
+ );
+ const dismiss = container.querySelector(
+ '[data-testid="memory-pressure-banner"] button',
+ ) as HTMLButtonElement;
+ // Trigger shown is memory critical (outranks disk elevated) — dismiss it.
+ await act(async () => dismiss.click());
+ // Disk warning is next in line and has its own key, so it surfaces...
+ expect(banner()?.textContent).toContain("disk is filling up");
+ const dismissDisk = container.querySelector(
+ '[data-testid="memory-pressure-banner"] button',
+ ) as HTMLButtonElement;
+ await act(async () => dismissDisk.click());
+ expect(banner()).toBeNull();
+ // Disk recovers; memory still critical — its dismissal must survive.
+ await rerender(
+ ,
+ );
+ expect(banner()).toBeNull();
+ });
+
+ it("a gateway reboot (boot_id change) re-opens a dismissed disk banner", async () => {
+ await render(
+ ,
+ );
+ const dismiss = container.querySelector(
+ '[data-testid="memory-pressure-banner"] button',
+ ) as HTMLButtonElement;
+ await act(async () => dismiss.click());
+ expect(banner()).toBeNull();
+ // Restart with the disk still full: new boot, warning returns.
+ await rerender(
+ ,
+ );
+ expect(banner()?.textContent).toContain("disk is almost full");
+ });
});
diff --git a/web/src/components/MemoryPressureBanner.tsx b/web/src/components/MemoryPressureBanner.tsx
index 851fe5991c..c7dd9d1850 100644
--- a/web/src/components/MemoryPressureBanner.tsx
+++ b/web/src/components/MemoryPressureBanner.tsx
@@ -4,31 +4,41 @@ import type { StatusResponse } from "@/lib/api";
import { useI18n } from "@/i18n";
/**
- * App-wide warning banner for memory trouble (NS-656).
+ * App-wide warning banner for resource trouble (NS-656): memory pressure
+ * and disk exhaustion.
*
- * Two independent triggers, worst-first:
- * 1. Live pressure — the gateway's heartbeat shows system memory in the
- * `elevated`/`critical` band right now.
- * 2. Post-mortem — the previous gateway life died uncleanly and its last
+ * Triggers, worst-first:
+ * 1. Disk critical — the HERMES_HOME volume is nearly full. Worst because
+ * the failure mode is silent data loss (SQLite writes failing, sessions
+ * and config not persisting), not just a restart (OOF-2/OOF-107).
+ * 2. Memory critical — the gateway's heartbeat shows system memory in the
+ * `critical` band right now.
+ * 3. Post-mortem — the previous gateway life died uncleanly and its last
* heartbeat showed near-exhausted memory (`last_boot_suspected_oom`).
* This is a heuristic, not proof the OOM killer acted — copy says so.
+ * 4. Disk elevated / 5. memory elevated — early warnings.
*
- * Both previously died in server-side log files; a hosted agent could be
- * OOM-killed hourly while the dashboard looked healthy.
+ * All of this previously died in server-side log files; a hosted agent
+ * could be OOM-killed hourly or fill its disk completely while the
+ * dashboard looked healthy.
*
* Dismissal semantics (session-scoped, sessionStorage):
* - EVERY dismissal key embeds the reporting boot (`boot_id`), so a gateway
* restart invalidates all of them. Without this, dismissing `critical`,
* rebooting, and coming back still-critical would hide the NEW incident —
- * and mask the OOM notice too, since critical takes precedence.
- * - Within one boot, live-pressure dismissal masks only the dismissed
- * severity; escalation (elevated → critical) re-opens immediately, and a
- * confirmed recovery (pressure back to "ok", not "unknown") clears live
- * dismissals so the NEXT episode in the same boot surfaces again.
+ * and mask the OOM notice too, since critical takes precedence. Disk
+ * entries share the scheme: disk state has no boot relationship, but
+ * re-surfacing a still-full disk after a restart is the desired behavior.
+ * - Within one boot, dismissal masks only the dismissed trigger; escalation
+ * (elevated → critical, in either domain) re-opens immediately, and a
+ * confirmed recovery (pressure back to "ok", not "unknown") clears that
+ * domain's live dismissals so the NEXT episode in the same boot surfaces
+ * again.
*/
const STORAGE_KEY = "memoryBannerDismissed";
-const LIVE_TRIGGERS = ["critical", "elevated"];
+const MEMORY_LIVE_TRIGGERS = ["critical", "elevated"];
+const DISK_LIVE_TRIGGERS = ["disk_critical", "disk_elevated"];
function readDismissed(): string[] {
try {
@@ -53,6 +63,11 @@ function writeDismissed(entries: string[]) {
}
}
+function entryMatches(triggers: string[]) {
+ return (entry: string) =>
+ triggers.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
+}
+
export function MemoryPressureBanner({
status,
}: {
@@ -60,49 +75,60 @@ export function MemoryPressureBanner({
}) {
const { t } = useI18n();
const memory = status?.memory;
+ const disk = status?.disk;
const pressure = memory?.pressure;
+ const diskPressure = disk?.pressure;
const [dismissed, setDismissed] = useState(readDismissed);
// Recovery reset (render-time state adjustment — the sanctioned React
- // pattern for reacting to prop changes without an effect): once live
- // pressure is demonstrably back to "ok", any dismissed live-pressure
- // entries describe a PAST episode — drop them so the next one isn't
- // silently hidden. "unknown" (stale/absent heartbeat) is absence of
- // evidence, not recovery, and clears nothing. Cross-boot invalidation
- // doesn't need handling here: boot_id is part of every dismissal key.
+ // pattern for reacting to prop changes without an effect): once a
+ // domain's live pressure is demonstrably back to "ok", any dismissed
+ // live entries for that domain describe a PAST episode — drop them so
+ // the next one isn't silently hidden. "unknown" (stale/absent sample)
+ // is absence of evidence, not recovery, and clears nothing. Each domain
+ // recovers independently: a fixed disk must not un-dismiss a memory
+ // warning or vice versa. Cross-boot invalidation doesn't need handling
+ // here: boot_id is part of every dismissal key.
const [prevPressure, setPrevPressure] = useState(pressure);
- if (pressure !== prevPressure) {
+ const [prevDiskPressure, setPrevDiskPressure] = useState(diskPressure);
+ if (pressure !== prevPressure || diskPressure !== prevDiskPressure) {
setPrevPressure(pressure);
- const isLiveEntry = (entry: string) =>
- LIVE_TRIGGERS.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
- if (pressure === "ok" && dismissed.some(isLiveEntry)) {
- const next = dismissed.filter((entry) => !isLiveEntry(entry));
- writeDismissed(next);
- setDismissed(next);
+ setPrevDiskPressure(diskPressure);
+ const recovered: Array<(entry: string) => boolean> = [];
+ if (pressure === "ok") recovered.push(entryMatches(MEMORY_LIVE_TRIGGERS));
+ if (diskPressure === "ok") recovered.push(entryMatches(DISK_LIVE_TRIGGERS));
+ if (recovered.length > 0) {
+ const isRecovered = (entry: string) =>
+ recovered.some((match) => match(entry));
+ if (dismissed.some(isRecovered)) {
+ const next = dismissed.filter((entry) => !isRecovered(entry));
+ writeDismissed(next);
+ setDismissed(next);
+ }
}
}
- // Highest-severity active trigger, or null.
- const trigger = !memory
- ? null
- : memory.pressure === "critical"
- ? "critical"
- : memory.last_boot_suspected_oom
- ? "oom_restart"
- : memory.pressure === "elevated"
- ? "elevated"
- : null;
+ // Active triggers, worst-first. Disk critical outranks memory critical:
+ // imminent data loss beats imminent restart. Dismissal cascades — hiding
+ // the top trigger surfaces the next one rather than silencing everything.
+ const activeTriggers: string[] = [];
+ if (diskPressure === "critical") activeTriggers.push("disk_critical");
+ if (memory?.pressure === "critical") activeTriggers.push("critical");
+ if (memory?.last_boot_suspected_oom) activeTriggers.push("oom_restart");
+ if (diskPressure === "elevated") activeTriggers.push("disk_elevated");
+ if (memory?.pressure === "elevated") activeTriggers.push("elevated");
// Every dismissal is scoped to the reporting boot: `boot_id` changes on
// each gateway life, so restarts invalidate prior dismissals of ANY kind.
// A missing boot_id (degraded payload / pre-NS-656 image) degrades to a
// shared per-severity bucket — old behavior, never a crash.
- const dismissKey = trigger
- ? `${trigger}:${memory?.boot_id ?? "unknown"}`
- : null;
+ const keyFor = (trig: string) => `${trig}:${memory?.boot_id ?? "unknown"}`;
+ const trigger =
+ activeTriggers.find((trig) => !dismissed.includes(keyFor(trig))) ?? null;
+ const dismissKey = trigger ? keyFor(trigger) : null;
- if (!trigger || !dismissKey || dismissed.includes(dismissKey)) return null;
+ if (!trigger || !dismissKey) return null;
const dismiss = () => {
setDismissed((prev) => {
@@ -112,16 +138,28 @@ export function MemoryPressureBanner({
});
};
- const critical = trigger === "critical";
+ const critical = trigger === "critical" || trigger === "disk_critical";
+ const diskFreeLabel =
+ disk?.free_mb != null ? ` (${Math.round(disk.free_mb)} MB free)` : "";
const message =
- trigger === "oom_restart"
- ? (t.app.memoryOomRestartBanner ??
- "Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
- : critical
- ? (t.app.memoryCriticalBanner ??
- "Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
- : (t.app.memoryElevatedBanner ??
- "Your agent is running low on memory.");
+ trigger === "disk_critical"
+ ? `${
+ t.app.diskCriticalBanner ??
+ "Your agent's disk is almost full. New messages, memories, and settings may fail to save."
+ }${diskFreeLabel}`
+ : trigger === "disk_elevated"
+ ? `${
+ t.app.diskElevatedBanner ??
+ "Your agent's disk is filling up. Consider clearing old sessions or expanding its storage."
+ }${diskFreeLabel}`
+ : trigger === "oom_restart"
+ ? (t.app.memoryOomRestartBanner ??
+ "Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
+ : critical
+ ? (t.app.memoryCriticalBanner ??
+ "Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
+ : (t.app.memoryElevatedBanner ??
+ "Your agent is running low on memory.");
return (