Merge pull request #107959 from NousResearch/fix/local-fit-mtp-accounting

fix(local-runtime): unify memory accounting and effective-window MTP
This commit is contained in:
Jeffrey Quesnelle
2026-09-11 02:05:40 -04:00
committed by GitHub
15 changed files with 523 additions and 111 deletions
+3 -2
View File
@@ -91,8 +91,9 @@ def _presets_stale() -> bool:
with suppress(Exception):
from hermes_cli.local_runtime.presets import read_preset_decisions
known = set(read_preset_decisions())
return any(mid not in known for mid in staged_model_ids())
known = read_preset_decisions()
return any(mid not in known or (not known[mid].refusal and not (known[mid].keys or {}).get("model"))
for mid in staged_model_ids())
return False
+14 -16
View File
@@ -17,8 +17,8 @@ from dataclasses import dataclass, field
from pathlib import PurePosixPath
from hermes_cli.local_runtime.context_policy import (
FLOOR, RUNTIME_OVERHEAD_BYTES, TARGET_WINDOW, ub_logits_bytes)
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
FLOOR, RUNTIME_OVERHEAD_BYTES, TARGET_WINDOW, LaunchPlan, plan_launch)
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, PhysicsRefusal
from hermes_cli.local_runtime.gguf import model_id_from_stem
logger = logging.getLogger(__name__)
@@ -115,6 +115,12 @@ class CatalogEntry:
n_ctx_train=self.n_ctx_train, layers=layers, swa_window=self.swa_window, moe=self.moe,
n_vocab=self.n_vocab, kv_scale=1.2 if self.mtp else 1.0)
def launch_plan(self, variant: QuantVariant, budget: HardwareBudget) -> LaunchPlan:
# Optional external drafts may use spare memory after download, never reduce this grant.
return plan_launch(self.profile(variant), budget, mtp_capable=self.mtp,
fixed_overhead=RUNTIME_OVERHEAD_BYTES
+ (self.mmproj.size_bytes if self.mmproj else 0))
def download_files(self, variant: QuantVariant) -> tuple:
"""Everything a download job fetches for this variant, in order."""
extras = tuple(a for a in (self.mmproj, self.draft) if a is not None)
@@ -141,22 +147,14 @@ def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice
"best-large-window": zero-spill at TARGET_WINDOW; "best-fits": zero-spill at the 64K floor;
"smallest-fits-spilled": weights spill to host RAM, priced honestly; None: physics refuses.
"""
overhead = (RUNTIME_OVERHEAD_BYTES
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
+ ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
native = entry.n_ctx_train or FLOOR
variant = entry.variants[-1]
profile = entry.profile(variant)
need = variant.weights_bytes + overhead
vram = budget.usable_vram_bytes
if need + ctx_bytes(profile, min(TARGET_WINDOW, native)) <= vram:
return VariantChoice(variant, zero_spill=True, reason_key="best-large-window")
floor_kv = ctx_bytes(profile, min(FLOOR, native))
if need + floor_kv <= vram:
return VariantChoice(variant, zero_spill=True, reason_key="best-fits")
if need + floor_kv <= vram + budget.ram_available_bytes:
decision = entry.launch_plan(variant, budget).decision
if isinstance(decision, PhysicsRefusal):
return None
if decision.spilled:
return VariantChoice(variant, zero_spill=False, reason_key="smallest-fits-spilled")
return None
reason = "best-large-window" if decision.window >= min(TARGET_WINDOW, entry.n_ctx_train or FLOOR) else "best-fits"
return VariantChoice(variant, zero_spill=True, reason_key=reason)
# ── recommendation: best quality that fits and isn't miserably slow ──
+71 -7
View File
@@ -7,10 +7,10 @@ behavior measured on real hardware (llama.cpp, discrete NVIDIA on Windows/WDDM,
from __future__ import annotations
from dataclasses import dataclass, field
from dataclasses import dataclass, field, replace
from hermes_cli.local_runtime.estimator import (
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, physics_check)
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, footprint_bytes, physics_check)
FLOOR = 64 * 1024 # = target; one internal constant
_LADDER_GROWTH = 1.5
@@ -64,7 +64,8 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
everywhere, capped at native. ``overhead_bytes`` is runtime cost beyond weights+KV; zero keeps
this pure physics for decision-table tests, production callers pass it.
"""
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention)
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention,
overhead_bytes=overhead_bytes)
if refusal:
return refusal
@@ -74,9 +75,13 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
def kv(rung: int) -> int:
return ctx_bytes(profile, rung, flash_attention=flash_attention)
def need(rung: int) -> int:
return footprint_bytes(profile, rung, flash_attention=flash_attention,
overhead_bytes=overhead_bytes)
best_zero_spill: int | None = None
for rung in rungs:
if profile.weights_bytes + overhead_bytes + kv(rung) > budget.usable_vram_bytes:
if need(rung) > budget.usable_vram_bytes:
break
best_zero_spill = rung
@@ -91,15 +96,74 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
for rung in rungs:
if rung < window:
continue
if kv(rung) > cap:
if (kv(rung) > cap
or need(rung) > budget.usable_vram_bytes + budget.ram_available_bytes):
break
window = rung
reason = f"floor held at {window // 1024}K; weights spill (deliberate price of the guarantee)"
kv_bytes = kv(window)
return WindowDecision(window=window, reasons=[reason],
spill_bytes=max(0, profile.weights_bytes + kv_bytes - budget.usable_vram_bytes),
kv_on_gpu=kv_bytes <= budget.usable_vram_bytes)
spill_bytes=max(0, need(window) - budget.usable_vram_bytes),
kv_on_gpu=kv_bytes + overhead_bytes <= budget.usable_vram_bytes)
@dataclass
class LaunchPlan:
decision: WindowDecision | PhysicsRefusal
mtp_prefill: bool
overhead_bytes: int
def plan_launch(profile: ModelProfile, budget: HardwareBudget, *, mtp_capable: bool = False,
fixed_overhead: int = RUNTIME_OVERHEAD_BYTES,
requested_window: int | None = None) -> LaunchPlan:
"""Window first, then prefill; price both postures at the effective window.
A restored window may fit only under lean MTP. Evaluate it before discarding it because
stacked exceeds memory, and keep deliberate spill when neither posture is resident.
"""
if mtp_capable and profile.kv_scale == 1.0:
profile = replace(profile, kv_scale=1.2)
initial: dict[bool, WindowDecision | PhysicsRefusal] = {}
def candidate(stacked: bool) -> LaunchPlan:
overhead = fixed_overhead + ub_logits_bytes(
profile.n_vocab, mtp_capable=mtp_capable, mtp_prefill=stacked)
decision = initial_window(profile, budget, overhead_bytes=overhead)
initial[stacked] = decision
if isinstance(decision, WindowDecision) and requested_window:
target = min(requested_window, profile.n_ctx_train or requested_window)
if target > decision.window and physics_check(
profile, budget, target, overhead_bytes=overhead) is None:
need = footprint_bytes(profile, target, overhead_bytes=overhead)
decision = WindowDecision(
window=target, spill_bytes=max(0, need - budget.usable_vram_bytes),
kv_on_gpu=ctx_bytes(profile, target) + overhead <= budget.usable_vram_bytes,
reasons=[f"grown window restored ({target // 1024}K)"])
return LaunchPlan(decision, stacked, overhead)
lean = candidate(False)
if not mtp_capable:
return lean
stacked = candidate(True)
if isinstance(stacked.decision, PhysicsRefusal):
return lean
if isinstance(lean.decision, PhysicsRefusal):
return stacked
if stacked.decision.window < lean.decision.window:
return lean
if not stacked.decision.spilled:
return stacked
# A previously granted window keeps its spill policy unless lean can make it resident.
stacked_initial, lean_initial = initial[True], initial[False]
if (requested_window and isinstance(stacked_initial, WindowDecision)
and isinstance(lean_initial, WindowDecision) and not stacked_initial.spilled
and stacked_initial.window >= lean_initial.window
and stacked.decision.window > stacked_initial.window and lean.decision.spilled):
return stacked
return lean
@dataclass
+12 -5
View File
@@ -131,11 +131,18 @@ class PhysicsRefusal:
message: str
def footprint_bytes(profile: ModelProfile, window: int, *, flash_attention: bool = True,
overhead_bytes: int = 0) -> int:
"""Complete estimated footprint; the hardware budget already excludes its reserve."""
return (profile.weights_bytes + ctx_bytes(profile, window, flash_attention=flash_attention)
+ max(0, overhead_bytes))
def physics_check(profile: ModelProfile, budget: HardwareBudget,
floor: int, *, flash_attention: bool = True) -> PhysicsRefusal | None:
needed = (profile.weights_bytes
+ ctx_bytes(profile, min(floor, profile.n_ctx_train or floor),
flash_attention=flash_attention))
floor: int, *, flash_attention: bool = True,
overhead_bytes: int = 0) -> PhysicsRefusal | None:
needed = footprint_bytes(profile, min(floor, profile.n_ctx_train or floor),
flash_attention=flash_attention, overhead_bytes=overhead_bytes)
available = budget.usable_vram_bytes + budget.ram_available_bytes
if needed <= available:
return None
@@ -144,4 +151,4 @@ def physics_check(profile: ModelProfile, budget: HardwareBudget,
needed_bytes=needed, available_bytes=available,
message=(f"{profile.name}: needs ~{needed / gib:.1f} GiB at the "
f"{floor // 1024}K floor but only ~{available / gib:.1f} GiB "
"of VRAM+RAM exist — try a smaller quant (UD-Q3/Q2)"))
"of VRAM+RAM are available — try a smaller model or a supported smaller quant"))
+15 -4
View File
@@ -73,14 +73,15 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
get_supervisor, refresh_local_runtime, staged_models)
from hermes_cli.local_runtime.context_policy import growth_decision
from hermes_cli.local_runtime.estimator import profile_from_gguf
from hermes_cli.local_runtime.gguf import read_gguf_header
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
from hermes_cli.local_runtime.hardware import probe_budget
from hermes_cli.local_runtime.presets import preset_for_model, read_preset_decisions
sup = get_supervisor()
if sup is None or not is_managed_endpoint(base_url):
return None
gguf = next((p for p in staged_models() if p.stem.startswith(model_id) or model_id in p.stem), None)
gguf = next((p for p in staged_models() if model_id_from_stem(p.stem) == model_id), None)
if gguf is None:
return None
@@ -95,11 +96,12 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
except Exception: # noqa: BLE001
server_idle = False
budget = probe_budget(planning=True)
decision = growth_decision(
# Capacity budget, not live-free: growth executes via a server bounce, so the grown
# instance loads onto a freed card. Live-free is distorted by the very model being grown
# — it reads its own residency as unavailable and vetoes rungs that fit.
profile, probe_budget(planning=True),
profile, budget,
current_window=current_window,
session_tokens=session_tokens,
measured_decode_tok_s=measured_decode_tok_s,
@@ -113,6 +115,11 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
return None
plan = preset_for_model(gguf, budget, set(), requested_window=decision.next_window)
if plan is None or plan.refusal or plan.window < decision.next_window:
logger.debug("growth %s: complete launch footprint does not admit the next rung", model_id)
return None
logger.info("context growth %s: %s", model_id, decision.reason)
save_window_override(model_id, decision.next_window)
if not refresh_local_runtime():
@@ -120,4 +127,8 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
# compresses instead of overflowing a stale window.
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
return None
return decision.next_window
materialized = read_preset_decisions().get(model_id)
if materialized is None or materialized.window < decision.next_window:
logger.warning("growth %s: refreshed preset did not grant the requested window", model_id)
return None
return materialized.window
+59 -67
View File
@@ -4,14 +4,15 @@ launch decisions.
from __future__ import annotations
import json
import logging
from dataclasses import dataclass, replace
from dataclasses import dataclass
from pathlib import Path
from hermes_cli.local_runtime.context_policy import (
RUNTIME_OVERHEAD_BYTES, WindowDecision, initial_window, launch_args, ub_logits_bytes)
RUNTIME_OVERHEAD_BYTES, launch_args, plan_launch, ub_logits_bytes)
from hermes_cli.local_runtime.estimator import (
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, profile_from_gguf)
HardwareBudget, PhysicsRefusal, ctx_bytes, footprint_bytes, profile_from_gguf)
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
logger = logging.getLogger(__name__)
@@ -56,53 +57,30 @@ def _asset_path(asset) -> "Path | None":
return path if path.exists() else None
def _choose_mtp_posture(profile: ModelProfile, budget: HardwareBudget,
fixed_overhead: int) -> tuple[bool, int]:
"""(mtp_prefill, logits_bytes) for an MTP model — window first, prefill second.
def _draft_fits(path: Path, profile, budget: HardwareBudget, window: int, overhead: int) -> bool:
"""Optional draft never shrinks the advertised window or displaces its GPU buffers.
Price the launch under both postures and keep whichever grants the larger window: the stacked
posture's bigger compute buffer buys ~3x short-prompt prefill but costs ~2 GiB that would
otherwise be window (measured at 256K the ub512 posture still prefills at 2.7K tok/s), so
never trade context away for prefill. Same window -> stacked.
The catalog does not know the draft's context layout. Admit it only after reading the file;
draft KV defaults to f16, independently of the target's q8 cache.
"""
plain_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True)
stacked_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
stacked = initial_window(profile, budget, overhead_bytes=fixed_overhead + stacked_logits)
plain = initial_window(profile, budget, overhead_bytes=fixed_overhead + plain_logits)
if (not isinstance(stacked, PhysicsRefusal) and not stacked.spilled
and (isinstance(plain, PhysicsRefusal) or stacked.window >= plain.window)):
return True, stacked_logits
return False, plain_logits
def _restore_grown_window(model_id: str, profile: ModelProfile, budget: HardwareBudget,
decision: WindowDecision, overhead: int) -> WindowDecision:
"""Session growth (growth.py): a persisted override lifts the launch window to where the ladder
last grew it — capped at native, and only when physics still clears the bigger window on THIS
boot's budget (a smaller-VRAM day re-fits honestly back down)."""
try:
from hermes_cli.local_runtime.growth import load_window_overrides
override = load_window_overrides().get(model_id)
native = profile.n_ctx_train or decision.window
if override and override > decision.window:
target = min(int(override), native)
kv = ctx_bytes(profile, target)
need = profile.weights_bytes + kv + overhead
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
return WindowDecision(
window=target, spill_bytes=max(0, need - budget.usable_vram_bytes),
kv_on_gpu=kv <= budget.usable_vram_bytes,
reasons=[f"grown window restored ({target // 1024}K)"])
except Exception as exc: # noqa: BLE001 — overrides are advisory
logger.debug("window override skipped for %s: %s", model_id, exc)
return decision
draft = profile_from_gguf(read_gguf_header(path))
draft_need = footprint_bytes(
draft, window, flash_attention=False,
overhead_bytes=RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(draft.n_vocab, mtp_capable=False))
except (ValueError, OSError) as exc:
logger.warning("draft omitted %s: %s", path.name, exc)
return False
return (footprint_bytes(profile, window, overhead_bytes=overhead) + draft_need
<= budget.usable_vram_bytes + budget.ram_available_bytes
and ctx_bytes(profile, window) + overhead + draft_need <= budget.usable_vram_bytes)
def _preset_for(gguf: Path, budget: HardwareBudget,
mtp_capable: set[str]) -> PresetEntry | None:
def preset_for_model(gguf: Path, budget: HardwareBudget,
mtp_capable: set[str], *, requested_window: int | None = None) -> PresetEntry | None:
"""The launch decision for one staged model, or None when its header is unreadable."""
from hermes_cli.local_runtime.catalog import entry_for_model
from hermes_cli.local_runtime.growth import load_window_overrides
model_id = model_id_from_stem(gguf.stem)
try:
@@ -113,30 +91,22 @@ def _preset_for(gguf: Path, budget: HardwareBudget,
return None
entry = entry_for_model(model_id)
is_mtp = entry.mtp if entry is not None else model_id in mtp_capable
if is_mtp and profile.kv_scale == 1.0:
# Header-derived profiles don't know about MTP's draft context; apply the calibrated KV
# multiplier so the launch fit prices what the server will actually allocate.
profile = replace(profile, kv_scale=1.2)
mmproj_path = _asset_path(entry.mmproj) if entry is not None else None
# Overhead beyond weights+KV: runtime buffers, the vision projector when present, and the
# logits buffers of whichever microbatch/MTP posture launch_args will choose — flag and price
# decided together, from the same facts.
fixed_overhead = RUNTIME_OVERHEAD_BYTES + (
entry.mmproj.size_bytes if entry is not None and mmproj_path is not None else 0)
if is_mtp:
mtp_prefill, logits_bytes = _choose_mtp_posture(profile, budget, fixed_overhead)
else:
mtp_prefill, logits_bytes = False, ub_logits_bytes(profile.n_vocab, mtp_capable=False)
overhead = fixed_overhead + logits_bytes
decision = initial_window(profile, budget, overhead_bytes=overhead)
plan = plan_launch(profile, budget, mtp_capable=is_mtp, fixed_overhead=fixed_overhead,
requested_window=(load_window_overrides().get(model_id)
if requested_window is None else requested_window))
decision = plan.decision
if isinstance(decision, PhysicsRefusal):
return PresetEntry(model_id=model_id, window=0, spilled=False, refusal=decision.message)
decision = _restore_grown_window(model_id, profile, budget, decision, overhead)
# The launch flags MUST match the pricing above (same entry/is_mtp/posture).
# Router discovery is preset-only: refused files must never autoload with stock fit.
keys = _args_to_keys(launch_args(
profile, decision, mtp_capable=is_mtp, uma=budget.uma, mtp_prefill=mtp_prefill,
profile, decision, mtp_capable=is_mtp, uma=budget.uma, mtp_prefill=plan.mtp_prefill,
mtp_draft_depth=entry.mtp_draft_depth if entry is not None else 3))
keys["model"] = str(gguf)
if entry is not None and is_mtp:
# Integrated-MTP targets sample on the backend, and so does the draft (pairing validated
# against the vendor's published llama.cpp recipes).
@@ -155,7 +125,7 @@ def _preset_for(gguf: Path, budget: HardwareBudget,
if mmproj_path is not None:
keys["mmproj"] = str(mmproj_path)
draft_path = _asset_path(entry.draft) if decision.spilled else None
if draft_path is not None:
if draft_path is not None and _draft_fits(draft_path, profile, budget, decision.window, plan.overhead_bytes):
keys["model-draft"] = str(draft_path)
keys["spec-type"] = "draft-dspark"
# Unsloth's measured cliff: acceptance 83% at 2-3 drafts, collapses at 4.
@@ -172,18 +142,31 @@ def generate_presets(models_dir: Path, budget: HardwareBudget, preset_path: Path
entries: list[PresetEntry] = []
sections: list[str] = []
for gguf in staged_in(models_dir, require_complete=False):
entry = _preset_for(gguf, budget, mtp_capable or set())
for gguf in staged_in(models_dir):
entry = preset_for_model(gguf, budget, mtp_capable or set())
if entry is None:
continue
entries.append(entry)
# INI comments preserve non-flag facts atomically with the launch policy.
sections.append("# hermes-decision: " + json.dumps({
"model_id": entry.model_id, "window": entry.window,
"spilled": entry.spilled, "refusal": entry.refusal}) + "\n")
if entry.keys is not None:
body = "\n".join(f"{k} = {v}" for k, v in entry.keys.items())
sections.append(f"[{entry.model_id}]\n{body}\n")
preset_path.parent.mkdir(parents=True, exist_ok=True)
preset_path.write_text("\n".join(sections), encoding="utf-8")
logger.info("wrote %d preset sections to %s", len(sections), preset_path)
import os
import tempfile
fd, tmp = tempfile.mkstemp(prefix=preset_path.name, suffix=".tmp", dir=preset_path.parent)
try:
with os.fdopen(fd, "w", encoding="utf-8") as stream:
stream.write("\n".join(sections))
os.replace(tmp, preset_path)
finally:
Path(tmp).unlink(missing_ok=True)
logger.info("wrote %d preset sections to %s", sum(e.keys is not None for e in entries), preset_path)
return entries
@@ -198,12 +181,21 @@ def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEn
preset_path = runtimes_root() / "presets.ini"
out: dict[str, PresetEntry] = {}
try:
parser = configparser.ConfigParser()
parser.read(preset_path, encoding="utf-8")
parser = configparser.ConfigParser(interpolation=None)
text = preset_path.read_text(encoding="utf-8")
parser.read_string(text)
recorded = {}
for line in text.splitlines():
if line.startswith("# hermes-decision: "):
fact = json.loads(line.removeprefix("# hermes-decision: "))
recorded[fact["model_id"]] = fact
if fact.get("refusal"):
out[fact["model_id"]] = PresetEntry(**fact)
for section in parser.sections():
out[section] = PresetEntry(
model_id=section, window=parser.getint(section, "ctx-size", fallback=0),
spilled=parser.has_option(section, "override-tensor"))
spilled=recorded.get(section, {}).get("spilled", parser.has_option(section, "override-tensor")),
keys=dict(parser[section]))
except Exception as exc: # noqa: BLE001
logger.debug("preset read-back failed: %s", exc)
return out
+2 -1
View File
@@ -152,7 +152,6 @@ class LlamaServerSupervisor:
"--host", "127.0.0.1",
"--port", str(self.port),
"--api-key", self.api_key,
"--models-dir", str(self.models_dir),
"--models-max", str(self.models_max),
# Residency contract: a chat request to a staged-but-unloaded model loads it (slow
# first token) instead of a bare 400/404 after an eject.
@@ -167,6 +166,8 @@ class LlamaServerSupervisor:
]
if self.preset_path and self.preset_path.exists():
cmd += ["--models-preset", str(self.preset_path)]
else:
cmd += ["--models-dir", str(self.models_dir)]
cmd += self.extra_args
self.log_path.parent.mkdir(parents=True, exist_ok=True)
if self._log_handle is not None:
+1 -6
View File
@@ -552,12 +552,7 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) ->
return row
variant = choice.variant
# Same overhead the launch decision prices (runtime buffers + vision projector + microbatch/MTP
# logits): the row must advertise the window the model will actually get, not a paper number.
overhead = (context_policy.RUNTIME_OVERHEAD_BYTES
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
+ context_policy.ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
decision = context_policy.initial_window(entry.profile(variant), budget, overhead_bytes=overhead)
decision = entry.launch_plan(variant, budget).decision
download_total = entry.download_bytes(variant)
row.update({
"fits": True, "model_id": variant.model_id, "quant": variant.quant,
+11 -1
View File
@@ -28,7 +28,7 @@ def _stage(home, name):
def _write_presets(home, *model_ids):
pdir = home / "runtimes" / "llamacpp"
pdir.mkdir(parents=True, exist_ok=True)
body = "\n".join(f"[{m}]\nctx-size = 65536\n" for m in model_ids)
body = "\n".join(f"[{m}]\nmodel = {home / 'models' / (m + '.gguf')}\nctx-size = 65536\n" for m in model_ids)
(pdir / "presets.ini").write_text(body, encoding="utf-8")
@@ -49,6 +49,16 @@ def test_presets_current_when_every_staged_model_is_covered(hermes_home):
assert _presets_stale() is False
def test_legacy_presets_without_model_paths_are_regenerated(hermes_home):
from hermes_cli.local_runtime.bootstrap import _presets_stale
_stage(hermes_home, "model-a")
_write_presets(hermes_home, "model-a")
ini = hermes_home / "runtimes/llamacpp/presets.ini"
ini.write_text("[model-a]\nctx-size = 65536\n")
assert _presets_stale()
def test_no_models_is_never_stale(hermes_home):
from hermes_cli.local_runtime.bootstrap import _presets_stale
+39
View File
@@ -151,6 +151,45 @@ def test_find_entry_for_model_resolves_split_ids():
assert variant.quant == "UD-Q4_K_XL"
def test_catalog_and_preset_agree_on_identical_model_facts(tmp_path, monkeypatch):
from types import SimpleNamespace
from hermes_cli.local_runtime import bootstrap, catalog, presets
from hermes_cli.local_runtime.context_policy import RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
from hermes_cli.local_runtime.estimator import ctx_bytes
from hermes_cli.web_routers.local_models import _catalog_row
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
monkeypatch.setattr("hermes_cli.web_routers.local_models._engine_too_old", lambda tag: False)
for entry in catalog.CATALOG:
variant = entry.variants[0]
profile = entry.profile(variant)
path = tmp_path / f"{variant.model_id}.gguf"
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
if entry.mmproj:
asset = bootstrap.assets_dir() / entry.mmproj.local_name
asset.parent.mkdir(parents=True, exist_ok=True)
asset.touch()
for vram in (16, 24, 32, 48):
for uma in (False, True):
machine = HardwareBudget(int(vram * GIB * 0.8), vram * GIB,
0 if uma else 32 * GIB, uma)
row = _catalog_row(entry, machine, None, None, set())
preset = presets.preset_for_model(path, machine, set())
assert row["fits"] == (preset.refusal is None)
if preset.refusal:
continue
assert row["start_window"] == preset.window
assert row["spilled"] == preset.spilled
overhead = (RUNTIME_OVERHEAD_BYTES + (entry.mmproj.size_bytes if entry.mmproj else 0)
+ ub_logits_bytes(profile.n_vocab, mtp_capable=entry.mtp,
mtp_prefill=preset.keys.get("ubatch-size") == "2048" and entry.mtp))
need = profile.weights_bytes + ctx_bytes(profile, preset.window) + overhead
assert preset.spilled == (need > machine.usable_vram_bytes)
assert need <= machine.usable_vram_bytes + machine.ram_available_bytes
def test_hybrid_long_context_stays_cheap():
"""The reason Nemotron/Qwen3.6 headline the catalog: their priced
64K-floor KV must be a small fraction of a dense model's."""
+32
View File
@@ -177,6 +177,38 @@ def test_physics_check_prices_at_floor_not_native():
assert physics_check(p, card(24, ram_gib=8), FLOOR) is None
@pytest.mark.parametrize("uma", [False, True])
def test_initial_window_accounts_for_overhead_in_every_verdict(uma):
from dataclasses import replace
profile = hybrid(weights_gib=8, native=FLOOR)
base = profile.weights_bytes + ctx_bytes(profile, FLOOR)
overhead = 2 * GIB
budget = HardwareBudget(base + GIB, base + GIB, 0 if uma else 4 * GIB, uma)
decision = initial_window(profile, budget, overhead_bytes=overhead)
if uma:
assert isinstance(decision, PhysicsRefusal)
assert decision.needed_bytes == base + overhead
else:
assert isinstance(decision, WindowDecision)
assert decision.spill_bytes == GIB
exact = replace(budget, usable_vram_bytes=base + overhead, ram_available_bytes=0)
assert not initial_window(profile, exact, overhead_bytes=overhead).spilled
short = replace(exact, usable_vram_bytes=exact.usable_vram_bytes - 1)
assert isinstance(initial_window(profile, short, overhead_bytes=overhead), PhysicsRefusal)
# A cheap-KV model may grow on the spill path, but only into memory that exists.
growing = hybrid(weights_gib=20, full_layers=4, recurrent_layers=0,
per_token_f16=1024, native=1024 * KIB)
floor_need = growing.weights_bytes + ctx_bytes(growing, FLOOR) + overhead
limited = HardwareBudget(8 * GIB, 8 * GIB, floor_need - 8 * GIB)
decision = initial_window(growing, limited, overhead_bytes=overhead)
assert isinstance(decision, WindowDecision)
assert decision.window == FLOOR
assert decision.spill_bytes == limited.ram_available_bytes
# ── ladder + initial window ──────────────────────────────────
+115
View File
@@ -230,6 +230,121 @@ def test_preset_restores_grown_window_midladder(hermes_home, tmp_path, monkeypat
assert restored.window >= grown, "override must lift the launch window"
def test_mtp_plan_matches_cost_at_initial_and_restored_windows(hermes_home, tmp_path, monkeypatch):
from dataclasses import replace
from types import SimpleNamespace
from hermes_cli.local_runtime import presets
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
from hermes_cli.local_runtime.growth import save_window_override
gib = 1 << 30
profile = ModelProfile(name="mtp-fit", weights_bytes=16 * gib, embd_table_bytes=0,
n_ctx_train=262144, layers=[(LayerKind.FULL, 4096)] * 32,
moe=True, n_vocab=151936)
priced = replace(profile, kv_scale=1.2)
lean = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
stacked = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, profile.name)
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
def generate(device, ram, override=0):
save_window_override(profile.name, override)
budget = HardwareBudget(device, device, ram)
return presets.generate_presets(mdir, budget, tmp_path / "presets.ini", {profile.name})[0]
floor_need = profile.weights_bytes + ctx_bytes(priced, FLOOR)
initial = generate(floor_need + lean, 8 * gib)
assert initial.window == FLOOR
assert not initial.spilled
assert "ubatch-size" not in initial.keys
assert initial.keys["spec-type"] == "draft-mtp"
# Persisting a floor grant must not turn a lean spilled boot into stacked prefill.
for override in (0, FLOOR, 73728):
spilled_boot = generate(16 * gib, 64 * gib, override)
assert spilled_boot.spilled and "ubatch-size" not in spilled_boot.keys
device = floor_need + stacked
control = generate(device, 8 * gib)
assert control.keys["ubatch-size"] == "2048"
grown_window = 73728
for ram in (8 * gib, 0):
# Also preserve a grown window when stacked exceeds total memory, not just VRAM.
grown = generate(device, ram, grown_window)
assert grown.window == grown_window
assert not grown.spilled
assert "ubatch-size" not in grown.keys
assert "override-tensor" not in grown.keys
assert grown.keys["spec-type"] == "draft-mtp"
assert profile.weights_bytes + ctx_bytes(priced, grown.window) + lean <= device
both_spill = generate(device, 64 * gib, 147456)
assert both_spill.window == 147456 and both_spill.spilled
assert both_spill.keys["ubatch-size"] == "2048"
assert "override-tensor" in both_spill.keys
smaller_boot = generate(floor_need + lean, 0, grown_window)
assert smaller_boot.window == FLOOR and not smaller_boot.spilled
assert profile.weights_bytes + ctx_bytes(priced, control.window) + stacked <= device
def test_growth_requires_an_admissible_materialized_preset(hermes_home, tmp_path, monkeypatch):
from dataclasses import replace
from types import SimpleNamespace
from hermes_cli.local_runtime import bootstrap, catalog, growth, hardware, presets
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
from hermes_cli.local_runtime.estimator import HardwareBudget, ctx_bytes
entry = next(e for e in catalog.CATALOG if e.mtp and e.mmproj)
model_id = entry.variants[-1].model_id
mdir = tmp_path / "models"
_stage_fake_gguf(mdir, model_id)
profile = replace(entry.profile(entry.variants[-1]), kv_scale=1.0)
monkeypatch.setattr(bootstrap, "staged_models", lambda: list(mdir.glob("*.gguf")))
monkeypatch.setattr(bootstrap, "get_supervisor", lambda: SimpleNamespace(is_idle=lambda m: True))
monkeypatch.setattr(growth, "is_managed_endpoint", lambda url: True)
from hermes_cli.local_runtime import gguf, estimator
monkeypatch.setattr(gguf, "read_gguf_header", lambda p: _header_stub())
monkeypatch.setattr(estimator, "profile_from_gguf", lambda h: profile)
monkeypatch.setattr(presets, "read_gguf_header", lambda p: _header_stub())
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
asset = bootstrap.assets_dir() / entry.mmproj.local_name
asset.parent.mkdir(parents=True, exist_ok=True)
asset.touch()
overhead = RUNTIME_OVERHEAD_BYTES + entry.mmproj.size_bytes + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
priced = replace(profile, kv_scale=1.2)
next_window = FLOOR * 3 // 2
floor_need = profile.weights_bytes + ctx_bytes(priced, FLOOR) + overhead
next_need = profile.weights_bytes + ctx_bytes(priced, next_window) + overhead
budget = HardwareBudget(floor_need, floor_need, 0, True)
monkeypatch.setattr(hardware, "probe_budget", lambda **kw: budget)
from hermes_cli.local_runtime.binaries import runtimes_root
preset_path = runtimes_root() / "presets.ini"
calls = []
def refresh():
calls.append(True)
presets.generate_presets(mdir, budget, preset_path)
return True
monkeypatch.setattr(bootstrap, "refresh_local_runtime", refresh)
args = dict(base_url="http://127.0.0.1:1/v1", session_tokens=FLOOR, current_window=FLOOR)
assert growth.maybe_grow_window(model_id, **args) is None
assert not calls and not growth.load_window_overrides()
budget = replace(budget, usable_vram_bytes=next_need, total_device_bytes=next_need)
assert growth.maybe_grow_window(model_id, **args) == next_window
assert presets.read_preset_decisions(preset_path)[model_id].window == next_window
assert growth.load_window_overrides()[model_id] == next_window
# A restart that claims success but does not materialize the grant must not tell the agent it grew.
monkeypatch.setattr(bootstrap, "refresh_local_runtime", lambda: True)
budget = replace(budget, usable_vram_bytes=64 << 30, total_device_bytes=64 << 30)
assert growth.maybe_grow_window(model_id, **{**args, "current_window": next_window}) is None
def test_sampling_ladder_file_beats_catalog_beats_nothing(hermes_home, tmp_path, monkeypatch):
"""The sampling deference ladder: the GGUF's own general.sampling.*
wins per key, catalog fills only what the file left silent, and a
@@ -66,6 +66,61 @@ def test_status_lists_staged_models_with_labels(client, tmp_path):
assert row["size_label"].endswith("GB")
def test_status_tracks_preset_spill_and_restored_window(client, tmp_path, monkeypatch):
from dataclasses import replace
from types import SimpleNamespace
from hermes_cli.local_runtime import bootstrap, presets
from hermes_cli.local_runtime.binaries import runtimes_root
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
from hermes_cli.local_runtime.growth import save_window_override
from hermes_cli.web_routers import local_models
# Dense spill has no override-tensor flag: status must use the recorded decision.
profile = ModelProfile("status-mtp", 16 << 30, 0, 262144,
[(LayerKind.FULL, 4096)] * 32, n_vocab=151936)
model_id = profile.name
_write_fake_gguf(bootstrap.models_dir() / f"{model_id}.gguf")
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
monkeypatch.setattr(local_models, "_state_endpoint", lambda: {"base_url": "http://127.0.0.1:1/v1"})
server_window = FLOOR
def router_response(running, route, **kwargs):
if route == "/models":
return {"data": [{"id": model_id, "status": {"value": "loaded"}}]}
assert route == f"/props?model={model_id}"
return {"default_generation_settings": {"n_ctx": server_window}}
monkeypatch.setattr(local_models, "_router_request", router_response)
floor_need = profile.weights_bytes + ctx_bytes(replace(profile, kv_scale=1.2), FLOOR)
lean = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
stacked = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
ini = runtimes_root() / "presets.ini"
grown = 73728
for device, override, spilled in ((floor_need + lean - 1, FLOOR, True),
(floor_need + stacked, grown, False)):
save_window_override(model_id, override)
preset = presets.generate_presets(bootstrap.models_dir(),
HardwareBudget(device, device, 8 << 30), ini, {model_id})[0]
assert preset.window == override and preset.spilled is spilled
assert preset.keys["spec-type"] == "draft-mtp"
assert "ubatch-size" not in preset.keys and "override-tensor" not in preset.keys
# Deliberately differ from the plan to prove the server remains the grant authority.
server_window = preset.window - 1024
response = client.get("/api/local-models/status")
assert response.status_code == 200
data = response.json()
assert data["loaded_models"][model_id] == "loaded"
placement = data["placement"][model_id]
assert placement["spilled"] is spilled
assert placement["window"] == preset.window
assert placement["granted_window"] == server_window
assert placement["granted_window_label"] == local_models._k_label(server_window)
# ── hardware ─────────────────────────────────────────────────
@@ -0,0 +1,86 @@
"""The router must serve only admitted presets, retaining refusal and spill facts on read-back."""
from pathlib import Path
from types import SimpleNamespace
from hermes_cli.local_runtime import presets, supervisor
from hermes_cli.local_runtime.estimator import HardwareBudget, ModelProfile
def test_preset_roundtrip_keeps_refusals_and_dense_spill(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
mdir = tmp_path / "models"
mdir.mkdir()
for name in ("allowed", "refused"):
(mdir / f"{name}.gguf").touch()
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(path=p, sampling_defaults={}))
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: ModelProfile(
name=h.path.stem, weights_bytes=(4 if h.path.stem == "allowed" else 40) << 30,
embd_table_bytes=0, n_ctx_train=65536, layers=[]))
ini = tmp_path / "presets.ini"
generated = presets.generate_presets(mdir, HardwareBudget(2 << 30, 2 << 30, 8 << 30), ini)
reread = presets.read_preset_decisions(ini)
assert set(reread) == {p.model_id for p in generated}
assert reread["refused"].refusal
assert reread["allowed"].spilled
assert reread["allowed"].keys["model"] == str(mdir / "allowed.gguf")
assert "override-tensor" not in reread["allowed"].keys # Dense spill has no tensor-pattern override.
def test_optional_draft_is_enabled_only_with_room_at_the_selected_window(tmp_path, monkeypatch):
from dataclasses import replace
from hermes_cli.local_runtime import bootstrap, catalog
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
entry = next(e for e in catalog.CATALOG if e.draft)
main = tmp_path / f"{entry.variants[0].model_id}.gguf"
draft = bootstrap.assets_dir() / entry.draft.local_name
draft.parent.mkdir(parents=True, exist_ok=True)
draft.touch()
main_profile = ModelProfile("main", 12 << 30, 0, 65536, [], moe=True)
draft_profile = ModelProfile("draft", 1 << 30, 0, 65536, [])
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(path=p, sampling_defaults={}))
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: draft_profile if h.path == draft else main_profile)
tight = HardwareBudget(8 << 30, 8 << 30, 6 << 30)
result = presets.preset_for_model(main, tight, set())
assert result.window == 65536 and result.spilled
assert "model-draft" not in result.keys
roomy = replace(tight, ram_available_bytes=16 << 30)
with_draft = presets.preset_for_model(main, roomy, set())
assert with_draft.window == result.window
assert with_draft.keys["model-draft"] == str(draft)
assert with_draft.keys["spec-type"] == "draft-dspark"
# Full target-window f16 state and logits count even above the draft's native window.
from hermes_cli.local_runtime.context_policy import RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
from hermes_cli.local_runtime.estimator import LayerKind, ctx_bytes
draft_profile = replace(draft_profile, n_ctx_train=32768,
layers=[(LayerKind.FULL, 4096)] * 4, n_vocab=32768)
draft_cost = (draft_profile.weights_bytes
+ ctx_bytes(draft_profile, result.window, flash_attention=False)
+ RUNTIME_OVERHEAD_BYTES
+ ub_logits_bytes(draft_profile.n_vocab, mtp_capable=False))
device_boundary = RUNTIME_OVERHEAD_BYTES + draft_cost
exact = replace(roomy, usable_vram_bytes=device_boundary, total_device_bytes=device_boundary)
assert "model-draft" in presets.preset_for_model(main, exact, set()).keys
below = replace(exact, usable_vram_bytes=device_boundary - 1)
assert "model-draft" not in presets.preset_for_model(main, below, set()).keys
# A draft too large for GPU memory is optional, not permission to move its buffers to RAM.
draft_profile = replace(draft_profile, weights_bytes=9 << 30)
assert "model-draft" not in presets.preset_for_model(main, roomy, set()).keys
def test_supervisor_with_presets_does_not_scan_unadmitted_files(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
ini = tmp_path / "presets.ini"
ini.write_text("[allowed]\nmodel = allowed.gguf\nctx-size = 65536\n")
calls = []
monkeypatch.setattr(supervisor, "server_binary", lambda p: Path("llama-server"))
monkeypatch.setattr(supervisor.subprocess, "Popen", lambda cmd, **kw: calls.append(cmd) or SimpleNamespace(pid=123))
sup = supervisor.LlamaServerSupervisor(tmp_path, tmp_path, port=1234, preset_path=ini)
try:
sup._spawn()
finally:
sup._log_handle.close()
assert "--models-preset" in calls[0]
assert "--models-dir" not in calls[0]
+8 -2
View File
@@ -64,8 +64,14 @@ end-to-end and exposes no knobs:
overflow in system RAM in the order that hurts least (expert weights
first, never the attention cache), trading some speed to protect the
context guarantee.
- **Conversation compression only kicks in at the model's maximum
window** — growth always comes first.
- **Memory fit includes the launch configuration**, not just the model file:
context state, runtime buffers, the vision projector, and MTP buffers all
count. For multi-token prediction (MTP), Hermes uses smaller batches when
larger batches would spill at the same context window. MTP stays enabled.
The same calculation runs when a grown window is restored after restart.
- **Conversation compression follows a growth check.** If a larger window
cannot fit, generation is too slow, or the native maximum is reached,
Hermes compresses instead of claiming a window the server did not receive.
- Idle models are unloaded after 15 minutes to free GPU memory; they
reload automatically on the next message.