feat(gateway): disk-usage telemetry + dashboard disk-pressure banner (NS-656)

Extends the NS-656 memory-pressure surface to cover disk exhaustion
(OOF-2 / OOF-107 lineage: agents fill their data volume — SQLite writes
fail, sessions stop persisting — while every dashboard looks healthy).

- gateway/disk_status.py (new): collect_disk_status() samples
  shutil.disk_usage(HERMES_HOME) and classifies pressure
  (critical: <256 MB free or >=95% used; elevated: <512 MB free).
  Never raises — degrades to pressure="unknown" with null telemetry,
  same contract as collect_memory_status().
- /api/status: sibling `disk` block next to `memory`, advisory only —
  not folded into component/overall health.
- web: DiskPressureStatus type; MemoryPressureBanner generalized to a
  resource banner with worst-first triggers (disk critical > memory
  critical > OOM restart > disk elevated > memory elevated) and
  cascading dismissals — hiding the top trigger surfaces the next one
  instead of silencing everything. All dismissals stay boot_id-scoped.
- i18n: diskCriticalBanner / diskElevatedBanner (en, optional fields
  with English fallback per existing pattern).

Tests: gateway/test_disk_status.py (14), web_server disk-block
presence/degradation, banner disk trigger/priority/dismissal-cascade
suite (21 total).
This commit is contained in:
Shannon Sands
2026-08-13 18:58:39 +10:00
committed by Teknium
parent f5a26b1575
commit 6977d21fa7
9 changed files with 576 additions and 50 deletions
+117
View File
@@ -0,0 +1,117 @@
"""Disk-usage rollup for ``/api/status`` (NS-656).
Companion to :mod:`gateway.memory_status`, closing the same class of gap
for storage: a hosted agent can fill its data volume completely — SQLite
writes failing, session persistence dead, config saves lost — while its
dashboard and the NAS agent card both look perfectly healthy. Fleet
incidents OOF-2 (unrecoverable disk-full) and OOF-107 (fleet-wide disk
exhaustion, remediated by hand) are exactly this failure mode.
The readiness endpoint already probes disk (``gateway/readiness.py::
_probe_disk``), but readiness is a component verdict, not user-facing
telemetry — nothing renders it. This module produces the public block
the dashboard SPA and the NAS availability sweep actually consume.
Unlike the memory block (which distills already-persisted heartbeat
files), disk is sampled live via :func:`shutil.disk_usage` — a single
``statvfs`` call, the same thing the readiness probe does per request.
There is no meaningful "staleness" dimension, so no ``sampled_at``.
Public-safety note: ``/api/status`` is an unauthenticated liveness probe
(``PUBLIC_API_PATHS``). This block carries only coarse numbers (MB
granularity, whole-percent usage) and an enum — the same disclosure
class as the ``memory`` block.
Everything is best-effort and read-only: an unreadable filesystem
degrades to ``pressure="unknown"`` rather than raising into the status
endpoint.
"""
from __future__ import annotations
import logging
import shutil
from pathlib import Path
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
# Disk-pressure thresholds. Percent alone misleads in both directions:
# 90% used on a 100 GB volume leaves a comfortable 10 GB, while 50% used
# on a tiny volume can be one image download from write failures. So the
# percent triggers are gated on absolute headroom also being low, and a
# hard absolute floor applies regardless of size — below it, SQLite
# journaling and config writes are at genuine risk on any volume.
_CRITICAL_FREE_MB = 256 # < 256 MB free: critical on any volume
_CRITICAL_PERCENT = 95.0 # >= 95% used AND < 1 GB free: critical
_CRITICAL_HEADROOM_MB = 1024
_ELEVATED_FREE_MB = 512 # < 512 MB free: elevated on any volume
_ELEVATED_PERCENT = 85.0 # >= 85% used AND < 4 GB free: elevated
_ELEVATED_HEADROOM_MB = 4096
_BYTES_PER_MB = 1024 * 1024
def _coerce_mb(value: Any) -> Optional[int]:
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
return None
return value
def classify_disk_pressure(free_mb: Any, total_mb: Any) -> str:
"""Map free/total MB to ``ok``/``elevated``/``critical``.
``unknown`` when the sample is missing or malformed — the caller must
not treat "we could not read it" as "disk is fine".
"""
free = _coerce_mb(free_mb)
total = _coerce_mb(total_mb)
if free is None or total is None or total <= 0:
return "unknown"
used_percent = (1 - free / total) * 100.0
if free < _CRITICAL_FREE_MB or (
used_percent >= _CRITICAL_PERCENT and free < _CRITICAL_HEADROOM_MB
):
return "critical"
if free < _ELEVATED_FREE_MB or (
used_percent >= _ELEVATED_PERCENT and free < _ELEVATED_HEADROOM_MB
):
return "elevated"
return "ok"
def collect_disk_status(home: Optional[Path] = None) -> Dict[str, Any]:
"""Build the ``disk`` block for ``/api/status``.
``home`` scopes the sample to a profile's HERMES_HOME (the status
endpoint's ``?profile=`` handling passes it through); on hosted
images every profile shares the ``/opt/data`` volume, so the answer
is the same — but scoping keeps the contract identical to the
``memory`` block's.
Always returns a dict — an unreadable/unmounted filesystem yields
``{"pressure": "unknown", ...}``. Never raises.
"""
status: Dict[str, Any] = {
"pressure": "unknown",
"total_mb": None,
"free_mb": None,
"used_percent": None,
}
try:
if home is None:
from hermes_constants import get_hermes_home
home = get_hermes_home()
usage = shutil.disk_usage(home)
except Exception:
return status
if usage.total <= 0:
return status
total_mb = usage.total // _BYTES_PER_MB
free_mb = usage.free // _BYTES_PER_MB
status["total_mb"] = total_mb
status["free_mb"] = free_mb
status["used_percent"] = round((usage.used / usage.total) * 100, 1)
status["pressure"] = classify_disk_pressure(free_mb, total_mb)
return status
+18
View File
@@ -3354,6 +3354,24 @@ async def get_status(profile: Optional[str] = None):
except Exception:
status["memory"] = {"pressure": "unknown"}
# Disk-usage rollup (NS-656, same lineage as OOF-2/OOF-107 fleet
# disk-exhaustion incidents). One statvfs call on HERMES_HOME's
# filesystem — coarse MB numbers + enum, same public disclosure
# class as the memory block, and equally advisory: not folded
# into components/overall.
try:
from gateway.disk_status import collect_disk_status
status["disk"] = await asyncio.get_running_loop().run_in_executor(
None,
functools.partial(
collect_disk_status,
profile_dir if profile_dir else get_hermes_home(),
),
)
except Exception:
status["disk"] = {"pressure": "unknown"}
# Deferred FTS rebuild progress (schema v23): lets the desktop /
# dashboard render a "search index rebuilding: N%" indicator instead
# of users wondering why old-message search is slower after an
+118
View File
@@ -0,0 +1,118 @@
"""Tests for gateway.disk_status — the /api/status disk rollup (NS-656)."""
from __future__ import annotations
import shutil
from pathlib import Path
import pytest
from gateway import disk_status
from gateway.disk_status import classify_disk_pressure, collect_disk_status
_GB = 1024 # MB per GB, for readable fixtures
class TestClassifyDiskPressure:
def test_plentiful_disk_is_ok(self) -> None:
# 40 GB free of 100 GB.
assert classify_disk_pressure(40 * _GB, 100 * _GB) == "ok"
def test_absolute_floor_is_critical_on_any_volume(self) -> None:
# 200 MB free — below the 256 MB floor even on a huge, low-percent
# volume would be impossible, so use a big volume mostly full.
assert classify_disk_pressure(200, 500 * _GB) == "critical"
def test_high_percent_with_low_headroom_is_critical(self) -> None:
# 96% used, 800 MB free (< 1 GB headroom) on a 20 GB volume.
assert classify_disk_pressure(800, 20 * _GB) == "critical"
def test_high_percent_with_ample_headroom_is_not_critical(self) -> None:
# 96% used but 20 GB free on a 500 GB volume — percent alone must
# not trigger critical when absolute headroom is comfortable.
assert classify_disk_pressure(20 * _GB, 500 * _GB) == "ok"
def test_low_free_is_elevated(self) -> None:
# 400 MB free of 4 GB (~90% used): below the 512 MB elevated floor,
# above the 256 MB critical floor, and under the 95% critical
# percent gate.
assert classify_disk_pressure(400, 4 * _GB) == "elevated"
def test_elevated_percent_band(self) -> None:
# 88% used, 2.4 GB free of 20 GB — elevated percent gate with
# headroom under 4 GB.
assert classify_disk_pressure(2400, 20 * _GB) == "elevated"
def test_elevated_percent_with_ample_headroom_is_ok(self) -> None:
# 90% used but 50 GB free of 500 GB.
assert classify_disk_pressure(50 * _GB, 500 * _GB) == "ok"
def test_missing_sample_is_unknown(self) -> None:
assert classify_disk_pressure(None, None) == "unknown"
def test_malformed_sample_is_unknown(self) -> None:
assert classify_disk_pressure("lots", 100) == "unknown"
assert classify_disk_pressure(True, 100) == "unknown"
assert classify_disk_pressure(-5, 100) == "unknown"
assert classify_disk_pressure(100, 0) == "unknown"
class TestCollectDiskStatus:
def test_reports_real_usage(self, tmp_path: Path) -> None:
status = collect_disk_status(tmp_path)
assert status["pressure"] in {"ok", "elevated", "critical"}
assert isinstance(status["total_mb"], int) and status["total_mb"] > 0
assert isinstance(status["free_mb"], int) and status["free_mb"] >= 0
assert isinstance(status["used_percent"], float)
assert 0.0 <= status["used_percent"] <= 100.0
def test_unreadable_filesystem_degrades_to_unknown(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
def _boom(_path): # noqa: ANN001, ANN202
raise OSError("statvfs failed")
monkeypatch.setattr(shutil, "disk_usage", _boom)
status = collect_disk_status(tmp_path)
assert status == {
"pressure": "unknown",
"total_mb": None,
"free_mb": None,
"used_percent": None,
}
def test_zero_total_degrades_to_unknown(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
fake = shutil._ntuple_diskusage(total=0, used=0, free=0) # type: ignore[attr-defined]
monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
status = collect_disk_status(tmp_path)
assert status["pressure"] == "unknown"
assert status["total_mb"] is None
def test_synthetic_full_volume_is_critical(
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
# 10 GB volume with 100 MB free — the OOF-2/OOF-107 state.
total = 10 * 1024**3
free = 100 * 1024**2
fake = shutil._ntuple_diskusage( # type: ignore[attr-defined]
total=total, used=total - free, free=free
)
monkeypatch.setattr(shutil, "disk_usage", lambda _p: fake)
status = collect_disk_status(tmp_path)
assert status["pressure"] == "critical"
assert status["total_mb"] == 10 * 1024
assert status["free_mb"] == 100
assert status["used_percent"] == 99.0
def test_never_raises_even_without_home(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
# Default-home resolution failing must degrade, not raise.
monkeypatch.setattr(
disk_status.shutil,
"disk_usage",
lambda _p: (_ for _ in ()).throw(PermissionError("nope")),
)
assert collect_disk_status(None)["pressure"] == "unknown"
+20
View File
@@ -3028,6 +3028,26 @@ class TestStatusMemoryBlock:
assert resp.status_code == 200
assert resp.json()["memory"] == {"pressure": "unknown"}
def test_disk_block_present_with_pressure_field(self):
data = self.client.get("/api/status").json()
assert "disk" in data
assert data["disk"]["pressure"] in {
"ok", "elevated", "critical", "unknown",
}
def test_disk_block_degrades_when_collector_raises(self, monkeypatch):
"""Same contract as the memory block: a broken collector must never
take down the status endpoint."""
import gateway.disk_status as ds
def _boom(*_a, **_k):
raise RuntimeError("collector exploded")
monkeypatch.setattr(ds, "collect_disk_status", _boom)
resp = self.client.get("/api/status")
assert resp.status_code == 200
assert resp.json()["disk"] == {"pressure": "unknown"}
class TestGatewayUpdatedAtContract:
"""Contract tests for /api/status ``gateway_updated_at``.
@@ -9,7 +9,11 @@ import type { ReactNode } from "react";
import { I18nProvider } from "@/i18n";
import { MemoryPressureBanner } from "./MemoryPressureBanner";
import type { StatusResponse, MemoryPressureStatus } from "@/lib/api";
import type {
StatusResponse,
MemoryPressureStatus,
DiskPressureStatus,
} from "@/lib/api";
let container: HTMLDivElement;
let root: Root;
@@ -38,6 +42,13 @@ function statusWith(memory: MemoryPressureStatus | undefined): StatusResponse {
return { memory } as StatusResponse;
}
function statusWithDisk(
disk: DiskPressureStatus | undefined,
memory?: MemoryPressureStatus,
): StatusResponse {
return { memory, disk } as StatusResponse;
}
function banner(): HTMLElement | null {
return container.querySelector('[data-testid="memory-pressure-banner"]');
}
@@ -240,4 +251,188 @@ describe("MemoryPressureBanner", () => {
);
expect(banner()).toBeNull();
});
it("renders nothing for healthy or unknown disk", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "ok", free_mb: 5000 })}
/>,
);
expect(banner()).toBeNull();
await rerender(
<MemoryPressureBanner status={statusWithDisk({ pressure: "unknown" })} />,
);
expect(banner()).toBeNull();
});
it("shows the disk-critical warning with free-space detail", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 120 })}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
expect(banner()?.textContent).toContain("(120 MB free)");
});
it("shows the disk-elevated warning", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
/>,
);
expect(banner()?.textContent).toContain("disk is filling up");
});
it("disk critical outranks memory critical", async () => {
// Imminent data loss beats imminent restart.
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "critical", free_mb: 100 },
{ pressure: "critical" },
)}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
it("memory OOM notice outranks disk elevated", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "elevated", free_mb: 900 },
{ pressure: "ok", last_boot_suspected_oom: true },
)}
/>,
);
expect(banner()?.textContent).toContain("restarted unexpectedly");
});
it("dismissing a disk warning does not mask a later memory warning", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
expect(banner()).toBeNull();
await rerender(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "elevated", free_mb: 900 },
{ pressure: "elevated" },
)}
/>,
);
expect(banner()?.textContent).toContain("running low on memory");
});
it("disk escalation to critical re-opens a dismissed disk banner", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "elevated", free_mb: 900 })}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
expect(banner()).toBeNull();
await rerender(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
it("disk recovery to ok resets disk dismissals for the next episode", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
// User frees space (or grows the disk)...
await rerender(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "ok", free_mb: 8000 })}
/>,
);
expect(banner()).toBeNull();
// ...then the disk fills again: new episode surfaces.
await rerender(
<MemoryPressureBanner
status={statusWithDisk({ pressure: "critical", free_mb: 150 })}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
it("disk recovery does NOT reset memory dismissals (and vice versa)", async () => {
// Dismiss a memory warning while disk is also elevated.
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "elevated", free_mb: 900 },
{ pressure: "critical" },
)}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
// Trigger shown is memory critical (outranks disk elevated) — dismiss it.
await act(async () => dismiss.click());
// Disk warning is next in line and has its own key, so it surfaces...
expect(banner()?.textContent).toContain("disk is filling up");
const dismissDisk = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismissDisk.click());
expect(banner()).toBeNull();
// Disk recovers; memory still critical — its dismissal must survive.
await rerender(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "ok", free_mb: 8000 },
{ pressure: "critical" },
)}
/>,
);
expect(banner()).toBeNull();
});
it("a gateway reboot (boot_id change) re-opens a dismissed disk banner", async () => {
await render(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "critical", free_mb: 150 },
{ pressure: "ok", boot_id: "2026-08-13T01:00:00+00:00" },
)}
/>,
);
const dismiss = container.querySelector(
'[data-testid="memory-pressure-banner"] button',
) as HTMLButtonElement;
await act(async () => dismiss.click());
expect(banner()).toBeNull();
// Restart with the disk still full: new boot, warning returns.
await rerender(
<MemoryPressureBanner
status={statusWithDisk(
{ pressure: "critical", free_mb: 150 },
{ pressure: "ok", boot_id: "2026-08-13T02:00:00+00:00" },
)}
/>,
);
expect(banner()?.textContent).toContain("disk is almost full");
});
});
+87 -49
View File
@@ -4,31 +4,41 @@ import type { StatusResponse } from "@/lib/api";
import { useI18n } from "@/i18n";
/**
* App-wide warning banner for memory trouble (NS-656).
* App-wide warning banner for resource trouble (NS-656): memory pressure
* and disk exhaustion.
*
* Two independent triggers, worst-first:
* 1. Live pressure — the gateway's heartbeat shows system memory in the
* `elevated`/`critical` band right now.
* 2. Post-mortem — the previous gateway life died uncleanly and its last
* Triggers, worst-first:
* 1. Disk critical — the HERMES_HOME volume is nearly full. Worst because
* the failure mode is silent data loss (SQLite writes failing, sessions
* and config not persisting), not just a restart (OOF-2/OOF-107).
* 2. Memory critical — the gateway's heartbeat shows system memory in the
* `critical` band right now.
* 3. Post-mortem — the previous gateway life died uncleanly and its last
* heartbeat showed near-exhausted memory (`last_boot_suspected_oom`).
* This is a heuristic, not proof the OOM killer acted — copy says so.
* 4. Disk elevated / 5. memory elevated — early warnings.
*
* Both previously died in server-side log files; a hosted agent could be
* OOM-killed hourly while the dashboard looked healthy.
* All of this previously died in server-side log files; a hosted agent
* could be OOM-killed hourly or fill its disk completely while the
* dashboard looked healthy.
*
* Dismissal semantics (session-scoped, sessionStorage):
* - EVERY dismissal key embeds the reporting boot (`boot_id`), so a gateway
* restart invalidates all of them. Without this, dismissing `critical`,
* rebooting, and coming back still-critical would hide the NEW incident —
* and mask the OOM notice too, since critical takes precedence.
* - Within one boot, live-pressure dismissal masks only the dismissed
* severity; escalation (elevated → critical) re-opens immediately, and a
* confirmed recovery (pressure back to "ok", not "unknown") clears live
* dismissals so the NEXT episode in the same boot surfaces again.
* and mask the OOM notice too, since critical takes precedence. Disk
* entries share the scheme: disk state has no boot relationship, but
* re-surfacing a still-full disk after a restart is the desired behavior.
* - Within one boot, dismissal masks only the dismissed trigger; escalation
* (elevated → critical, in either domain) re-opens immediately, and a
* confirmed recovery (pressure back to "ok", not "unknown") clears that
* domain's live dismissals so the NEXT episode in the same boot surfaces
* again.
*/
const STORAGE_KEY = "memoryBannerDismissed";
const LIVE_TRIGGERS = ["critical", "elevated"];
const MEMORY_LIVE_TRIGGERS = ["critical", "elevated"];
const DISK_LIVE_TRIGGERS = ["disk_critical", "disk_elevated"];
function readDismissed(): string[] {
try {
@@ -53,6 +63,11 @@ function writeDismissed(entries: string[]) {
}
}
function entryMatches(triggers: string[]) {
return (entry: string) =>
triggers.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
}
export function MemoryPressureBanner({
status,
}: {
@@ -60,49 +75,60 @@ export function MemoryPressureBanner({
}) {
const { t } = useI18n();
const memory = status?.memory;
const disk = status?.disk;
const pressure = memory?.pressure;
const diskPressure = disk?.pressure;
const [dismissed, setDismissed] = useState<string[]>(readDismissed);
// Recovery reset (render-time state adjustment — the sanctioned React
// pattern for reacting to prop changes without an effect): once live
// pressure is demonstrably back to "ok", any dismissed live-pressure
// entries describe a PAST episode — drop them so the next one isn't
// silently hidden. "unknown" (stale/absent heartbeat) is absence of
// evidence, not recovery, and clears nothing. Cross-boot invalidation
// doesn't need handling here: boot_id is part of every dismissal key.
// pattern for reacting to prop changes without an effect): once a
// domain's live pressure is demonstrably back to "ok", any dismissed
// live entries for that domain describe a PAST episode — drop them so
// the next one isn't silently hidden. "unknown" (stale/absent sample)
// is absence of evidence, not recovery, and clears nothing. Each domain
// recovers independently: a fixed disk must not un-dismiss a memory
// warning or vice versa. Cross-boot invalidation doesn't need handling
// here: boot_id is part of every dismissal key.
const [prevPressure, setPrevPressure] = useState(pressure);
if (pressure !== prevPressure) {
const [prevDiskPressure, setPrevDiskPressure] = useState(diskPressure);
if (pressure !== prevPressure || diskPressure !== prevDiskPressure) {
setPrevPressure(pressure);
const isLiveEntry = (entry: string) =>
LIVE_TRIGGERS.some((sev) => entry === sev || entry.startsWith(`${sev}:`));
if (pressure === "ok" && dismissed.some(isLiveEntry)) {
const next = dismissed.filter((entry) => !isLiveEntry(entry));
writeDismissed(next);
setDismissed(next);
setPrevDiskPressure(diskPressure);
const recovered: Array<(entry: string) => boolean> = [];
if (pressure === "ok") recovered.push(entryMatches(MEMORY_LIVE_TRIGGERS));
if (diskPressure === "ok") recovered.push(entryMatches(DISK_LIVE_TRIGGERS));
if (recovered.length > 0) {
const isRecovered = (entry: string) =>
recovered.some((match) => match(entry));
if (dismissed.some(isRecovered)) {
const next = dismissed.filter((entry) => !isRecovered(entry));
writeDismissed(next);
setDismissed(next);
}
}
}
// Highest-severity active trigger, or null.
const trigger = !memory
? null
: memory.pressure === "critical"
? "critical"
: memory.last_boot_suspected_oom
? "oom_restart"
: memory.pressure === "elevated"
? "elevated"
: null;
// Active triggers, worst-first. Disk critical outranks memory critical:
// imminent data loss beats imminent restart. Dismissal cascades — hiding
// the top trigger surfaces the next one rather than silencing everything.
const activeTriggers: string[] = [];
if (diskPressure === "critical") activeTriggers.push("disk_critical");
if (memory?.pressure === "critical") activeTriggers.push("critical");
if (memory?.last_boot_suspected_oom) activeTriggers.push("oom_restart");
if (diskPressure === "elevated") activeTriggers.push("disk_elevated");
if (memory?.pressure === "elevated") activeTriggers.push("elevated");
// Every dismissal is scoped to the reporting boot: `boot_id` changes on
// each gateway life, so restarts invalidate prior dismissals of ANY kind.
// A missing boot_id (degraded payload / pre-NS-656 image) degrades to a
// shared per-severity bucket — old behavior, never a crash.
const dismissKey = trigger
? `${trigger}:${memory?.boot_id ?? "unknown"}`
: null;
const keyFor = (trig: string) => `${trig}:${memory?.boot_id ?? "unknown"}`;
const trigger =
activeTriggers.find((trig) => !dismissed.includes(keyFor(trig))) ?? null;
const dismissKey = trigger ? keyFor(trigger) : null;
if (!trigger || !dismissKey || dismissed.includes(dismissKey)) return null;
if (!trigger || !dismissKey) return null;
const dismiss = () => {
setDismissed((prev) => {
@@ -112,16 +138,28 @@ export function MemoryPressureBanner({
});
};
const critical = trigger === "critical";
const critical = trigger === "critical" || trigger === "disk_critical";
const diskFreeLabel =
disk?.free_mb != null ? ` (${Math.round(disk.free_mb)} MB free)` : "";
const message =
trigger === "oom_restart"
? (t.app.memoryOomRestartBanner ??
"Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
: critical
? (t.app.memoryCriticalBanner ??
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
: (t.app.memoryElevatedBanner ??
"Your agent is running low on memory.");
trigger === "disk_critical"
? `${
t.app.diskCriticalBanner ??
"Your agent's disk is almost full. New messages, memories, and settings may fail to save."
}${diskFreeLabel}`
: trigger === "disk_elevated"
? `${
t.app.diskElevatedBanner ??
"Your agent's disk is filling up. Consider clearing old sessions or expanding its storage."
}${diskFreeLabel}`
: trigger === "oom_restart"
? (t.app.memoryOomRestartBanner ??
"Your agent restarted unexpectedly, most likely because it ran out of memory. Long sessions and many concurrent tasks increase memory use.")
: critical
? (t.app.memoryCriticalBanner ??
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.")
: (t.app.memoryElevatedBanner ??
"Your agent is running low on memory.");
return (
<div
+4
View File
@@ -102,6 +102,10 @@ export const en: Translations = {
memoryCriticalBanner:
"Your agent is almost out of memory and may restart. Consider closing idle sessions or upgrading its memory.",
memoryElevatedBanner: "Your agent is running low on memory.",
diskCriticalBanner:
"Your agent's disk is almost full. New messages, memories, and settings may fail to save.",
diskElevatedBanner:
"Your agent's disk is filling up. Consider clearing old sessions or expanding its storage.",
dismiss: "Dismiss",
},
+3
View File
@@ -119,6 +119,9 @@ export interface Translations {
memoryOomRestartBanner?: string;
memoryCriticalBanner?: string;
memoryElevatedBanner?: string;
/** NS-656 disk-usage banner — optional, English fallback. */
diskCriticalBanner?: string;
diskElevatedBanner?: string;
dismiss?: string;
};
+13
View File
@@ -1885,6 +1885,9 @@ export interface StatusResponse {
/** NS-656: memory-pressure rollup from the gateway heartbeat +
* lifecycle ledger. Absent on older gateways. */
memory?: MemoryPressureStatus;
/** NS-656: disk-usage rollup for the HERMES_HOME volume. Absent on
* older gateways. */
disk?: DiskPressureStatus;
release_date: string;
version: string;
}
@@ -1907,6 +1910,16 @@ export interface MemoryPressureStatus {
boot_id?: string | null;
}
/** NS-656: coarse disk telemetry served by /api/status. Live statvfs
* sample of the HERMES_HOME volume — no staleness dimension, so no
* sampled_at. */
export interface DiskPressureStatus {
pressure: "ok" | "elevated" | "critical" | "unknown";
total_mb?: number | null;
free_mb?: number | null;
used_percent?: number | null;
}
export interface SessionInfo {
id: string;
source: string | null;