0316d3d404
Price weights, context, runtime, projector and batch overhead consistently across catalog admission, initial launch, growth and restored windows. Keep MTP and the larger window when lean batches avoid unnecessary spill. Admit optional external drafts only when their complete footprint fits. Use preset-only model discovery so refused files cannot autoload, and preserve refusal/spill decisions atomically for desktop status read-back. Add regression coverage for complete-footprint boundaries, MTP restarts, growth admission, draft budgets and placement status transitions. Builds on the overhead-accounting contribution in #102993 and the restored-window MTP contribution in #106897. Does not adopt the 40% host-RAM reserve or resolve the remaining requests in #102865/#106895. Co-authored-by: infinitycrew39 <infinitycrew39@gmail.com> Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
135 lines
5.3 KiB
Python
135 lines
5.3 KiB
Python
"""In-session context growth for the managed llama.cpp runtime.
|
|
|
|
Scope guard: only a server THIS process supervises grows. Detected external servers and
|
|
other-process supervisors keep their own policies.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from contextlib import suppress
|
|
import json
|
|
import logging
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def window_overrides_path():
|
|
from hermes_cli.local_runtime.binaries import runtimes_root
|
|
|
|
return runtimes_root() / "window_overrides.json"
|
|
|
|
|
|
def load_window_overrides() -> dict:
|
|
"""model_id -> granted window (int). Empty on any read problem."""
|
|
with suppress(Exception):
|
|
with open(window_overrides_path(), encoding="utf-8") as fh:
|
|
data = json.load(fh)
|
|
return {str(k): int(v) for k, v in data.items()}
|
|
return {}
|
|
|
|
|
|
def _write_overrides(overrides: dict) -> None:
|
|
path = window_overrides_path()
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(json.dumps(overrides, indent=1), encoding="utf-8")
|
|
|
|
|
|
def save_window_override(model_id: str, window: int) -> None:
|
|
overrides = load_window_overrides()
|
|
overrides[model_id] = int(window)
|
|
_write_overrides(overrides)
|
|
|
|
|
|
def clear_window_override(model_id: str) -> None:
|
|
"""Drop a model's growth state (delete/re-download paths)."""
|
|
overrides = load_window_overrides()
|
|
if model_id in overrides:
|
|
del overrides[model_id]
|
|
_write_overrides(overrides)
|
|
|
|
|
|
def is_managed_endpoint(base_url: str) -> bool:
|
|
"""True when base_url is the server this process's state file points at."""
|
|
with suppress(Exception):
|
|
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
|
|
|
state = _state_endpoint()
|
|
return state is not None and (
|
|
(base_url or "").rstrip("/") == str(state.get("base_url", "")).rstrip("/"))
|
|
return False
|
|
|
|
|
|
def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
|
current_window: int,
|
|
measured_decode_tok_s: float | None = None) -> int | None:
|
|
"""One growth evaluation + execution. Returns the NEW window when the ladder granted a bigger
|
|
one, else None (hold / compress / not ours).
|
|
|
|
The caller sits at a request boundary by construction (the pre-API compression gate), so
|
|
re-prefill growth is safe at any call: the next request rebuilds server state in the larger
|
|
window — nothing rewinds.
|
|
"""
|
|
from hermes_cli.local_runtime.bootstrap import (
|
|
get_supervisor, refresh_local_runtime, staged_models)
|
|
from hermes_cli.local_runtime.context_policy import growth_decision
|
|
from hermes_cli.local_runtime.estimator import profile_from_gguf
|
|
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
|
|
from hermes_cli.local_runtime.hardware import probe_budget
|
|
from hermes_cli.local_runtime.presets import preset_for_model, read_preset_decisions
|
|
|
|
sup = get_supervisor()
|
|
if sup is None or not is_managed_endpoint(base_url):
|
|
return None
|
|
|
|
gguf = next((p for p in staged_models() if model_id_from_stem(p.stem) == model_id), None)
|
|
if gguf is None:
|
|
return None
|
|
|
|
try:
|
|
profile = profile_from_gguf(read_gguf_header(gguf))
|
|
except (ValueError, OSError) as exc:
|
|
logger.debug("growth skip %s: unreadable gguf (%s)", model_id, exc)
|
|
return None
|
|
|
|
try:
|
|
server_idle = sup.is_idle(model_id)
|
|
except Exception: # noqa: BLE001
|
|
server_idle = False
|
|
|
|
budget = probe_budget(planning=True)
|
|
decision = growth_decision(
|
|
# Capacity budget, not live-free: growth executes via a server bounce, so the grown
|
|
# instance loads onto a freed card. Live-free is distorted by the very model being grown
|
|
# — it reads its own residency as unavailable and vetoes rungs that fit.
|
|
profile, budget,
|
|
current_window=current_window,
|
|
session_tokens=session_tokens,
|
|
measured_decode_tok_s=measured_decode_tok_s,
|
|
server_idle=server_idle,
|
|
# The caller IS the occupancy signal: this runs from the agent's compression gate, which
|
|
# fired on its own threshold. Two separately-derived edges must not deadlock into
|
|
# compress-before-grow.
|
|
occupancy_confirmed=True,
|
|
)
|
|
if decision.action != "grow" or not decision.next_window:
|
|
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
|
|
return None
|
|
|
|
plan = preset_for_model(gguf, budget, set(), requested_window=decision.next_window)
|
|
if plan is None or plan.refusal or plan.window < decision.next_window:
|
|
logger.debug("growth %s: complete launch footprint does not admit the next rung", model_id)
|
|
return None
|
|
|
|
logger.info("context growth %s: %s", model_id, decision.reason)
|
|
save_window_override(model_id, decision.next_window)
|
|
if not refresh_local_runtime():
|
|
# The override still lands at the next boot; report no growth NOW so the caller
|
|
# compresses instead of overflowing a stale window.
|
|
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
|
|
return None
|
|
materialized = read_preset_decisions().get(model_id)
|
|
if materialized is None or materialized.window < decision.next_window:
|
|
logger.warning("growth %s: refreshed preset did not grant the requested window", model_id)
|
|
return None
|
|
return materialized.window
|