Files
hermes-agent/hermes_cli/local_runtime/growth.py
T
emozilla 0316d3d404 fix(local-runtime): unify memory accounting and effective-window MTP
Price weights, context, runtime, projector and batch overhead consistently
across catalog admission, initial launch, growth and restored windows.
Keep MTP and the larger window when lean batches avoid unnecessary spill.

Admit optional external drafts only when their complete footprint fits.
Use preset-only model discovery so refused files cannot autoload, and
preserve refusal/spill decisions atomically for desktop status read-back.

Add regression coverage for complete-footprint boundaries, MTP restarts,
growth admission, draft budgets and placement status transitions.

Builds on the overhead-accounting contribution in #102993 and the
restored-window MTP contribution in #106897. Does not adopt the 40%
host-RAM reserve or resolve the remaining requests in #102865/#106895.

Co-authored-by: infinitycrew39 <infinitycrew39@gmail.com>
Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
2026-09-11 01:40:43 -04:00

135 lines
5.3 KiB
Python

"""In-session context growth for the managed llama.cpp runtime.
Scope guard: only a server THIS process supervises grows. Detected external servers and
other-process supervisors keep their own policies.
"""
from __future__ import annotations
from contextlib import suppress
import json
import logging
logger = logging.getLogger(__name__)
def window_overrides_path():
from hermes_cli.local_runtime.binaries import runtimes_root
return runtimes_root() / "window_overrides.json"
def load_window_overrides() -> dict:
"""model_id -> granted window (int). Empty on any read problem."""
with suppress(Exception):
with open(window_overrides_path(), encoding="utf-8") as fh:
data = json.load(fh)
return {str(k): int(v) for k, v in data.items()}
return {}
def _write_overrides(overrides: dict) -> None:
path = window_overrides_path()
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(overrides, indent=1), encoding="utf-8")
def save_window_override(model_id: str, window: int) -> None:
overrides = load_window_overrides()
overrides[model_id] = int(window)
_write_overrides(overrides)
def clear_window_override(model_id: str) -> None:
"""Drop a model's growth state (delete/re-download paths)."""
overrides = load_window_overrides()
if model_id in overrides:
del overrides[model_id]
_write_overrides(overrides)
def is_managed_endpoint(base_url: str) -> bool:
"""True when base_url is the server this process's state file points at."""
with suppress(Exception):
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
return state is not None and (
(base_url or "").rstrip("/") == str(state.get("base_url", "")).rstrip("/"))
return False
def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
current_window: int,
measured_decode_tok_s: float | None = None) -> int | None:
"""One growth evaluation + execution. Returns the NEW window when the ladder granted a bigger
one, else None (hold / compress / not ours).
The caller sits at a request boundary by construction (the pre-API compression gate), so
re-prefill growth is safe at any call: the next request rebuilds server state in the larger
window — nothing rewinds.
"""
from hermes_cli.local_runtime.bootstrap import (
get_supervisor, refresh_local_runtime, staged_models)
from hermes_cli.local_runtime.context_policy import growth_decision
from hermes_cli.local_runtime.estimator import profile_from_gguf
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
from hermes_cli.local_runtime.hardware import probe_budget
from hermes_cli.local_runtime.presets import preset_for_model, read_preset_decisions
sup = get_supervisor()
if sup is None or not is_managed_endpoint(base_url):
return None
gguf = next((p for p in staged_models() if model_id_from_stem(p.stem) == model_id), None)
if gguf is None:
return None
try:
profile = profile_from_gguf(read_gguf_header(gguf))
except (ValueError, OSError) as exc:
logger.debug("growth skip %s: unreadable gguf (%s)", model_id, exc)
return None
try:
server_idle = sup.is_idle(model_id)
except Exception: # noqa: BLE001
server_idle = False
budget = probe_budget(planning=True)
decision = growth_decision(
# Capacity budget, not live-free: growth executes via a server bounce, so the grown
# instance loads onto a freed card. Live-free is distorted by the very model being grown
# — it reads its own residency as unavailable and vetoes rungs that fit.
profile, budget,
current_window=current_window,
session_tokens=session_tokens,
measured_decode_tok_s=measured_decode_tok_s,
server_idle=server_idle,
# The caller IS the occupancy signal: this runs from the agent's compression gate, which
# fired on its own threshold. Two separately-derived edges must not deadlock into
# compress-before-grow.
occupancy_confirmed=True,
)
if decision.action != "grow" or not decision.next_window:
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
return None
plan = preset_for_model(gguf, budget, set(), requested_window=decision.next_window)
if plan is None or plan.refusal or plan.window < decision.next_window:
logger.debug("growth %s: complete launch footprint does not admit the next rung", model_id)
return None
logger.info("context growth %s: %s", model_id, decision.reason)
save_window_override(model_id, decision.next_window)
if not refresh_local_runtime():
# The override still lands at the next boot; report no growth NOW so the caller
# compresses instead of overflowing a stale window.
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
return None
materialized = read_preset_decisions().get(model_id)
if materialized is None or materialized.window < decision.next_window:
logger.warning("growth %s: refreshed preset did not grant the requested window", model_id)
return None
return materialized.window