Merge pull request #107959 from NousResearch/fix/local-fit-mtp-accounting
fix(local-runtime): unify memory accounting and effective-window MTP
This commit is contained in:
@@ -91,8 +91,9 @@ def _presets_stale() -> bool:
|
||||
with suppress(Exception):
|
||||
from hermes_cli.local_runtime.presets import read_preset_decisions
|
||||
|
||||
known = set(read_preset_decisions())
|
||||
return any(mid not in known for mid in staged_model_ids())
|
||||
known = read_preset_decisions()
|
||||
return any(mid not in known or (not known[mid].refusal and not (known[mid].keys or {}).get("model"))
|
||||
for mid in staged_model_ids())
|
||||
return False
|
||||
|
||||
|
||||
|
||||
@@ -17,8 +17,8 @@ from dataclasses import dataclass, field
|
||||
from pathlib import PurePosixPath
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
FLOOR, RUNTIME_OVERHEAD_BYTES, TARGET_WINDOW, ub_logits_bytes)
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
|
||||
FLOOR, RUNTIME_OVERHEAD_BYTES, TARGET_WINDOW, LaunchPlan, plan_launch)
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, PhysicsRefusal
|
||||
from hermes_cli.local_runtime.gguf import model_id_from_stem
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -115,6 +115,12 @@ class CatalogEntry:
|
||||
n_ctx_train=self.n_ctx_train, layers=layers, swa_window=self.swa_window, moe=self.moe,
|
||||
n_vocab=self.n_vocab, kv_scale=1.2 if self.mtp else 1.0)
|
||||
|
||||
def launch_plan(self, variant: QuantVariant, budget: HardwareBudget) -> LaunchPlan:
|
||||
# Optional external drafts may use spare memory after download, never reduce this grant.
|
||||
return plan_launch(self.profile(variant), budget, mtp_capable=self.mtp,
|
||||
fixed_overhead=RUNTIME_OVERHEAD_BYTES
|
||||
+ (self.mmproj.size_bytes if self.mmproj else 0))
|
||||
|
||||
def download_files(self, variant: QuantVariant) -> tuple:
|
||||
"""Everything a download job fetches for this variant, in order."""
|
||||
extras = tuple(a for a in (self.mmproj, self.draft) if a is not None)
|
||||
@@ -141,22 +147,14 @@ def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice
|
||||
"best-large-window": zero-spill at TARGET_WINDOW; "best-fits": zero-spill at the 64K floor;
|
||||
"smallest-fits-spilled": weights spill to host RAM, priced honestly; None: physics refuses.
|
||||
"""
|
||||
overhead = (RUNTIME_OVERHEAD_BYTES
|
||||
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
|
||||
native = entry.n_ctx_train or FLOOR
|
||||
variant = entry.variants[-1]
|
||||
profile = entry.profile(variant)
|
||||
need = variant.weights_bytes + overhead
|
||||
vram = budget.usable_vram_bytes
|
||||
if need + ctx_bytes(profile, min(TARGET_WINDOW, native)) <= vram:
|
||||
return VariantChoice(variant, zero_spill=True, reason_key="best-large-window")
|
||||
floor_kv = ctx_bytes(profile, min(FLOOR, native))
|
||||
if need + floor_kv <= vram:
|
||||
return VariantChoice(variant, zero_spill=True, reason_key="best-fits")
|
||||
if need + floor_kv <= vram + budget.ram_available_bytes:
|
||||
decision = entry.launch_plan(variant, budget).decision
|
||||
if isinstance(decision, PhysicsRefusal):
|
||||
return None
|
||||
if decision.spilled:
|
||||
return VariantChoice(variant, zero_spill=False, reason_key="smallest-fits-spilled")
|
||||
return None
|
||||
reason = "best-large-window" if decision.window >= min(TARGET_WINDOW, entry.n_ctx_train or FLOOR) else "best-fits"
|
||||
return VariantChoice(variant, zero_spill=True, reason_key=reason)
|
||||
|
||||
|
||||
# ── recommendation: best quality that fits and isn't miserably slow ──
|
||||
|
||||
@@ -7,10 +7,10 @@ behavior measured on real hardware (llama.cpp, discrete NVIDIA on Windows/WDDM,
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass, field, replace
|
||||
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, physics_check)
|
||||
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, footprint_bytes, physics_check)
|
||||
|
||||
FLOOR = 64 * 1024 # = target; one internal constant
|
||||
_LADDER_GROWTH = 1.5
|
||||
@@ -64,7 +64,8 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
|
||||
everywhere, capped at native. ``overhead_bytes`` is runtime cost beyond weights+KV; zero keeps
|
||||
this pure physics for decision-table tests, production callers pass it.
|
||||
"""
|
||||
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention)
|
||||
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention,
|
||||
overhead_bytes=overhead_bytes)
|
||||
if refusal:
|
||||
return refusal
|
||||
|
||||
@@ -74,9 +75,13 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
|
||||
def kv(rung: int) -> int:
|
||||
return ctx_bytes(profile, rung, flash_attention=flash_attention)
|
||||
|
||||
def need(rung: int) -> int:
|
||||
return footprint_bytes(profile, rung, flash_attention=flash_attention,
|
||||
overhead_bytes=overhead_bytes)
|
||||
|
||||
best_zero_spill: int | None = None
|
||||
for rung in rungs:
|
||||
if profile.weights_bytes + overhead_bytes + kv(rung) > budget.usable_vram_bytes:
|
||||
if need(rung) > budget.usable_vram_bytes:
|
||||
break
|
||||
best_zero_spill = rung
|
||||
|
||||
@@ -91,15 +96,74 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
|
||||
for rung in rungs:
|
||||
if rung < window:
|
||||
continue
|
||||
if kv(rung) > cap:
|
||||
if (kv(rung) > cap
|
||||
or need(rung) > budget.usable_vram_bytes + budget.ram_available_bytes):
|
||||
break
|
||||
window = rung
|
||||
reason = f"floor held at {window // 1024}K; weights spill (deliberate price of the guarantee)"
|
||||
|
||||
kv_bytes = kv(window)
|
||||
return WindowDecision(window=window, reasons=[reason],
|
||||
spill_bytes=max(0, profile.weights_bytes + kv_bytes - budget.usable_vram_bytes),
|
||||
kv_on_gpu=kv_bytes <= budget.usable_vram_bytes)
|
||||
spill_bytes=max(0, need(window) - budget.usable_vram_bytes),
|
||||
kv_on_gpu=kv_bytes + overhead_bytes <= budget.usable_vram_bytes)
|
||||
|
||||
|
||||
@dataclass
|
||||
class LaunchPlan:
|
||||
decision: WindowDecision | PhysicsRefusal
|
||||
mtp_prefill: bool
|
||||
overhead_bytes: int
|
||||
|
||||
|
||||
def plan_launch(profile: ModelProfile, budget: HardwareBudget, *, mtp_capable: bool = False,
|
||||
fixed_overhead: int = RUNTIME_OVERHEAD_BYTES,
|
||||
requested_window: int | None = None) -> LaunchPlan:
|
||||
"""Window first, then prefill; price both postures at the effective window.
|
||||
|
||||
A restored window may fit only under lean MTP. Evaluate it before discarding it because
|
||||
stacked exceeds memory, and keep deliberate spill when neither posture is resident.
|
||||
"""
|
||||
if mtp_capable and profile.kv_scale == 1.0:
|
||||
profile = replace(profile, kv_scale=1.2)
|
||||
|
||||
initial: dict[bool, WindowDecision | PhysicsRefusal] = {}
|
||||
|
||||
def candidate(stacked: bool) -> LaunchPlan:
|
||||
overhead = fixed_overhead + ub_logits_bytes(
|
||||
profile.n_vocab, mtp_capable=mtp_capable, mtp_prefill=stacked)
|
||||
decision = initial_window(profile, budget, overhead_bytes=overhead)
|
||||
initial[stacked] = decision
|
||||
if isinstance(decision, WindowDecision) and requested_window:
|
||||
target = min(requested_window, profile.n_ctx_train or requested_window)
|
||||
if target > decision.window and physics_check(
|
||||
profile, budget, target, overhead_bytes=overhead) is None:
|
||||
need = footprint_bytes(profile, target, overhead_bytes=overhead)
|
||||
decision = WindowDecision(
|
||||
window=target, spill_bytes=max(0, need - budget.usable_vram_bytes),
|
||||
kv_on_gpu=ctx_bytes(profile, target) + overhead <= budget.usable_vram_bytes,
|
||||
reasons=[f"grown window restored ({target // 1024}K)"])
|
||||
return LaunchPlan(decision, stacked, overhead)
|
||||
|
||||
lean = candidate(False)
|
||||
if not mtp_capable:
|
||||
return lean
|
||||
stacked = candidate(True)
|
||||
if isinstance(stacked.decision, PhysicsRefusal):
|
||||
return lean
|
||||
if isinstance(lean.decision, PhysicsRefusal):
|
||||
return stacked
|
||||
if stacked.decision.window < lean.decision.window:
|
||||
return lean
|
||||
if not stacked.decision.spilled:
|
||||
return stacked
|
||||
# A previously granted window keeps its spill policy unless lean can make it resident.
|
||||
stacked_initial, lean_initial = initial[True], initial[False]
|
||||
if (requested_window and isinstance(stacked_initial, WindowDecision)
|
||||
and isinstance(lean_initial, WindowDecision) and not stacked_initial.spilled
|
||||
and stacked_initial.window >= lean_initial.window
|
||||
and stacked.decision.window > stacked_initial.window and lean.decision.spilled):
|
||||
return stacked
|
||||
return lean
|
||||
|
||||
|
||||
@dataclass
|
||||
|
||||
@@ -131,11 +131,18 @@ class PhysicsRefusal:
|
||||
message: str
|
||||
|
||||
|
||||
def footprint_bytes(profile: ModelProfile, window: int, *, flash_attention: bool = True,
|
||||
overhead_bytes: int = 0) -> int:
|
||||
"""Complete estimated footprint; the hardware budget already excludes its reserve."""
|
||||
return (profile.weights_bytes + ctx_bytes(profile, window, flash_attention=flash_attention)
|
||||
+ max(0, overhead_bytes))
|
||||
|
||||
|
||||
def physics_check(profile: ModelProfile, budget: HardwareBudget,
|
||||
floor: int, *, flash_attention: bool = True) -> PhysicsRefusal | None:
|
||||
needed = (profile.weights_bytes
|
||||
+ ctx_bytes(profile, min(floor, profile.n_ctx_train or floor),
|
||||
flash_attention=flash_attention))
|
||||
floor: int, *, flash_attention: bool = True,
|
||||
overhead_bytes: int = 0) -> PhysicsRefusal | None:
|
||||
needed = footprint_bytes(profile, min(floor, profile.n_ctx_train or floor),
|
||||
flash_attention=flash_attention, overhead_bytes=overhead_bytes)
|
||||
available = budget.usable_vram_bytes + budget.ram_available_bytes
|
||||
if needed <= available:
|
||||
return None
|
||||
@@ -144,4 +151,4 @@ def physics_check(profile: ModelProfile, budget: HardwareBudget,
|
||||
needed_bytes=needed, available_bytes=available,
|
||||
message=(f"{profile.name}: needs ~{needed / gib:.1f} GiB at the "
|
||||
f"{floor // 1024}K floor but only ~{available / gib:.1f} GiB "
|
||||
"of VRAM+RAM exist — try a smaller quant (UD-Q3/Q2)"))
|
||||
"of VRAM+RAM are available — try a smaller model or a supported smaller quant"))
|
||||
|
||||
@@ -73,14 +73,15 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
get_supervisor, refresh_local_runtime, staged_models)
|
||||
from hermes_cli.local_runtime.context_policy import growth_decision
|
||||
from hermes_cli.local_runtime.estimator import profile_from_gguf
|
||||
from hermes_cli.local_runtime.gguf import read_gguf_header
|
||||
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
|
||||
from hermes_cli.local_runtime.hardware import probe_budget
|
||||
from hermes_cli.local_runtime.presets import preset_for_model, read_preset_decisions
|
||||
|
||||
sup = get_supervisor()
|
||||
if sup is None or not is_managed_endpoint(base_url):
|
||||
return None
|
||||
|
||||
gguf = next((p for p in staged_models() if p.stem.startswith(model_id) or model_id in p.stem), None)
|
||||
gguf = next((p for p in staged_models() if model_id_from_stem(p.stem) == model_id), None)
|
||||
if gguf is None:
|
||||
return None
|
||||
|
||||
@@ -95,11 +96,12 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
except Exception: # noqa: BLE001
|
||||
server_idle = False
|
||||
|
||||
budget = probe_budget(planning=True)
|
||||
decision = growth_decision(
|
||||
# Capacity budget, not live-free: growth executes via a server bounce, so the grown
|
||||
# instance loads onto a freed card. Live-free is distorted by the very model being grown
|
||||
# — it reads its own residency as unavailable and vetoes rungs that fit.
|
||||
profile, probe_budget(planning=True),
|
||||
profile, budget,
|
||||
current_window=current_window,
|
||||
session_tokens=session_tokens,
|
||||
measured_decode_tok_s=measured_decode_tok_s,
|
||||
@@ -113,6 +115,11 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
|
||||
return None
|
||||
|
||||
plan = preset_for_model(gguf, budget, set(), requested_window=decision.next_window)
|
||||
if plan is None or plan.refusal or plan.window < decision.next_window:
|
||||
logger.debug("growth %s: complete launch footprint does not admit the next rung", model_id)
|
||||
return None
|
||||
|
||||
logger.info("context growth %s: %s", model_id, decision.reason)
|
||||
save_window_override(model_id, decision.next_window)
|
||||
if not refresh_local_runtime():
|
||||
@@ -120,4 +127,8 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
# compresses instead of overflowing a stale window.
|
||||
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
|
||||
return None
|
||||
return decision.next_window
|
||||
materialized = read_preset_decisions().get(model_id)
|
||||
if materialized is None or materialized.window < decision.next_window:
|
||||
logger.warning("growth %s: refreshed preset did not grant the requested window", model_id)
|
||||
return None
|
||||
return materialized.window
|
||||
|
||||
@@ -4,14 +4,15 @@ launch decisions.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from dataclasses import dataclass, replace
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
RUNTIME_OVERHEAD_BYTES, WindowDecision, initial_window, launch_args, ub_logits_bytes)
|
||||
RUNTIME_OVERHEAD_BYTES, launch_args, plan_launch, ub_logits_bytes)
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, profile_from_gguf)
|
||||
HardwareBudget, PhysicsRefusal, ctx_bytes, footprint_bytes, profile_from_gguf)
|
||||
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -56,53 +57,30 @@ def _asset_path(asset) -> "Path | None":
|
||||
return path if path.exists() else None
|
||||
|
||||
|
||||
def _choose_mtp_posture(profile: ModelProfile, budget: HardwareBudget,
|
||||
fixed_overhead: int) -> tuple[bool, int]:
|
||||
"""(mtp_prefill, logits_bytes) for an MTP model — window first, prefill second.
|
||||
def _draft_fits(path: Path, profile, budget: HardwareBudget, window: int, overhead: int) -> bool:
|
||||
"""Optional draft never shrinks the advertised window or displaces its GPU buffers.
|
||||
|
||||
Price the launch under both postures and keep whichever grants the larger window: the stacked
|
||||
posture's bigger compute buffer buys ~3x short-prompt prefill but costs ~2 GiB that would
|
||||
otherwise be window (measured at 256K the ub512 posture still prefills at 2.7K tok/s), so
|
||||
never trade context away for prefill. Same window -> stacked.
|
||||
The catalog does not know the draft's context layout. Admit it only after reading the file;
|
||||
draft KV defaults to f16, independently of the target's q8 cache.
|
||||
"""
|
||||
plain_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
stacked_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
|
||||
stacked = initial_window(profile, budget, overhead_bytes=fixed_overhead + stacked_logits)
|
||||
plain = initial_window(profile, budget, overhead_bytes=fixed_overhead + plain_logits)
|
||||
if (not isinstance(stacked, PhysicsRefusal) and not stacked.spilled
|
||||
and (isinstance(plain, PhysicsRefusal) or stacked.window >= plain.window)):
|
||||
return True, stacked_logits
|
||||
return False, plain_logits
|
||||
|
||||
|
||||
def _restore_grown_window(model_id: str, profile: ModelProfile, budget: HardwareBudget,
|
||||
decision: WindowDecision, overhead: int) -> WindowDecision:
|
||||
"""Session growth (growth.py): a persisted override lifts the launch window to where the ladder
|
||||
last grew it — capped at native, and only when physics still clears the bigger window on THIS
|
||||
boot's budget (a smaller-VRAM day re-fits honestly back down)."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.growth import load_window_overrides
|
||||
|
||||
override = load_window_overrides().get(model_id)
|
||||
native = profile.n_ctx_train or decision.window
|
||||
if override and override > decision.window:
|
||||
target = min(int(override), native)
|
||||
kv = ctx_bytes(profile, target)
|
||||
need = profile.weights_bytes + kv + overhead
|
||||
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
|
||||
return WindowDecision(
|
||||
window=target, spill_bytes=max(0, need - budget.usable_vram_bytes),
|
||||
kv_on_gpu=kv <= budget.usable_vram_bytes,
|
||||
reasons=[f"grown window restored ({target // 1024}K)"])
|
||||
except Exception as exc: # noqa: BLE001 — overrides are advisory
|
||||
logger.debug("window override skipped for %s: %s", model_id, exc)
|
||||
return decision
|
||||
draft = profile_from_gguf(read_gguf_header(path))
|
||||
draft_need = footprint_bytes(
|
||||
draft, window, flash_attention=False,
|
||||
overhead_bytes=RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(draft.n_vocab, mtp_capable=False))
|
||||
except (ValueError, OSError) as exc:
|
||||
logger.warning("draft omitted %s: %s", path.name, exc)
|
||||
return False
|
||||
return (footprint_bytes(profile, window, overhead_bytes=overhead) + draft_need
|
||||
<= budget.usable_vram_bytes + budget.ram_available_bytes
|
||||
and ctx_bytes(profile, window) + overhead + draft_need <= budget.usable_vram_bytes)
|
||||
|
||||
|
||||
def _preset_for(gguf: Path, budget: HardwareBudget,
|
||||
mtp_capable: set[str]) -> PresetEntry | None:
|
||||
def preset_for_model(gguf: Path, budget: HardwareBudget,
|
||||
mtp_capable: set[str], *, requested_window: int | None = None) -> PresetEntry | None:
|
||||
"""The launch decision for one staged model, or None when its header is unreadable."""
|
||||
from hermes_cli.local_runtime.catalog import entry_for_model
|
||||
from hermes_cli.local_runtime.growth import load_window_overrides
|
||||
|
||||
model_id = model_id_from_stem(gguf.stem)
|
||||
try:
|
||||
@@ -113,30 +91,22 @@ def _preset_for(gguf: Path, budget: HardwareBudget,
|
||||
return None
|
||||
entry = entry_for_model(model_id)
|
||||
is_mtp = entry.mtp if entry is not None else model_id in mtp_capable
|
||||
if is_mtp and profile.kv_scale == 1.0:
|
||||
# Header-derived profiles don't know about MTP's draft context; apply the calibrated KV
|
||||
# multiplier so the launch fit prices what the server will actually allocate.
|
||||
profile = replace(profile, kv_scale=1.2)
|
||||
|
||||
mmproj_path = _asset_path(entry.mmproj) if entry is not None else None
|
||||
# Overhead beyond weights+KV: runtime buffers, the vision projector when present, and the
|
||||
# logits buffers of whichever microbatch/MTP posture launch_args will choose — flag and price
|
||||
# decided together, from the same facts.
|
||||
fixed_overhead = RUNTIME_OVERHEAD_BYTES + (
|
||||
entry.mmproj.size_bytes if entry is not None and mmproj_path is not None else 0)
|
||||
if is_mtp:
|
||||
mtp_prefill, logits_bytes = _choose_mtp_posture(profile, budget, fixed_overhead)
|
||||
else:
|
||||
mtp_prefill, logits_bytes = False, ub_logits_bytes(profile.n_vocab, mtp_capable=False)
|
||||
overhead = fixed_overhead + logits_bytes
|
||||
decision = initial_window(profile, budget, overhead_bytes=overhead)
|
||||
plan = plan_launch(profile, budget, mtp_capable=is_mtp, fixed_overhead=fixed_overhead,
|
||||
requested_window=(load_window_overrides().get(model_id)
|
||||
if requested_window is None else requested_window))
|
||||
decision = plan.decision
|
||||
if isinstance(decision, PhysicsRefusal):
|
||||
return PresetEntry(model_id=model_id, window=0, spilled=False, refusal=decision.message)
|
||||
decision = _restore_grown_window(model_id, profile, budget, decision, overhead)
|
||||
|
||||
# The launch flags MUST match the pricing above (same entry/is_mtp/posture).
|
||||
# Router discovery is preset-only: refused files must never autoload with stock fit.
|
||||
keys = _args_to_keys(launch_args(
|
||||
profile, decision, mtp_capable=is_mtp, uma=budget.uma, mtp_prefill=mtp_prefill,
|
||||
profile, decision, mtp_capable=is_mtp, uma=budget.uma, mtp_prefill=plan.mtp_prefill,
|
||||
mtp_draft_depth=entry.mtp_draft_depth if entry is not None else 3))
|
||||
keys["model"] = str(gguf)
|
||||
if entry is not None and is_mtp:
|
||||
# Integrated-MTP targets sample on the backend, and so does the draft (pairing validated
|
||||
# against the vendor's published llama.cpp recipes).
|
||||
@@ -155,7 +125,7 @@ def _preset_for(gguf: Path, budget: HardwareBudget,
|
||||
if mmproj_path is not None:
|
||||
keys["mmproj"] = str(mmproj_path)
|
||||
draft_path = _asset_path(entry.draft) if decision.spilled else None
|
||||
if draft_path is not None:
|
||||
if draft_path is not None and _draft_fits(draft_path, profile, budget, decision.window, plan.overhead_bytes):
|
||||
keys["model-draft"] = str(draft_path)
|
||||
keys["spec-type"] = "draft-dspark"
|
||||
# Unsloth's measured cliff: acceptance 83% at 2-3 drafts, collapses at 4.
|
||||
@@ -172,18 +142,31 @@ def generate_presets(models_dir: Path, budget: HardwareBudget, preset_path: Path
|
||||
|
||||
entries: list[PresetEntry] = []
|
||||
sections: list[str] = []
|
||||
for gguf in staged_in(models_dir, require_complete=False):
|
||||
entry = _preset_for(gguf, budget, mtp_capable or set())
|
||||
for gguf in staged_in(models_dir):
|
||||
entry = preset_for_model(gguf, budget, mtp_capable or set())
|
||||
if entry is None:
|
||||
continue
|
||||
entries.append(entry)
|
||||
# INI comments preserve non-flag facts atomically with the launch policy.
|
||||
sections.append("# hermes-decision: " + json.dumps({
|
||||
"model_id": entry.model_id, "window": entry.window,
|
||||
"spilled": entry.spilled, "refusal": entry.refusal}) + "\n")
|
||||
if entry.keys is not None:
|
||||
body = "\n".join(f"{k} = {v}" for k, v in entry.keys.items())
|
||||
sections.append(f"[{entry.model_id}]\n{body}\n")
|
||||
|
||||
preset_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
preset_path.write_text("\n".join(sections), encoding="utf-8")
|
||||
logger.info("wrote %d preset sections to %s", len(sections), preset_path)
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
fd, tmp = tempfile.mkstemp(prefix=preset_path.name, suffix=".tmp", dir=preset_path.parent)
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as stream:
|
||||
stream.write("\n".join(sections))
|
||||
os.replace(tmp, preset_path)
|
||||
finally:
|
||||
Path(tmp).unlink(missing_ok=True)
|
||||
logger.info("wrote %d preset sections to %s", sum(e.keys is not None for e in entries), preset_path)
|
||||
return entries
|
||||
|
||||
|
||||
@@ -198,12 +181,21 @@ def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEn
|
||||
preset_path = runtimes_root() / "presets.ini"
|
||||
out: dict[str, PresetEntry] = {}
|
||||
try:
|
||||
parser = configparser.ConfigParser()
|
||||
parser.read(preset_path, encoding="utf-8")
|
||||
parser = configparser.ConfigParser(interpolation=None)
|
||||
text = preset_path.read_text(encoding="utf-8")
|
||||
parser.read_string(text)
|
||||
recorded = {}
|
||||
for line in text.splitlines():
|
||||
if line.startswith("# hermes-decision: "):
|
||||
fact = json.loads(line.removeprefix("# hermes-decision: "))
|
||||
recorded[fact["model_id"]] = fact
|
||||
if fact.get("refusal"):
|
||||
out[fact["model_id"]] = PresetEntry(**fact)
|
||||
for section in parser.sections():
|
||||
out[section] = PresetEntry(
|
||||
model_id=section, window=parser.getint(section, "ctx-size", fallback=0),
|
||||
spilled=parser.has_option(section, "override-tensor"))
|
||||
spilled=recorded.get(section, {}).get("spilled", parser.has_option(section, "override-tensor")),
|
||||
keys=dict(parser[section]))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("preset read-back failed: %s", exc)
|
||||
return out
|
||||
|
||||
@@ -152,7 +152,6 @@ class LlamaServerSupervisor:
|
||||
"--host", "127.0.0.1",
|
||||
"--port", str(self.port),
|
||||
"--api-key", self.api_key,
|
||||
"--models-dir", str(self.models_dir),
|
||||
"--models-max", str(self.models_max),
|
||||
# Residency contract: a chat request to a staged-but-unloaded model loads it (slow
|
||||
# first token) instead of a bare 400/404 after an eject.
|
||||
@@ -167,6 +166,8 @@ class LlamaServerSupervisor:
|
||||
]
|
||||
if self.preset_path and self.preset_path.exists():
|
||||
cmd += ["--models-preset", str(self.preset_path)]
|
||||
else:
|
||||
cmd += ["--models-dir", str(self.models_dir)]
|
||||
cmd += self.extra_args
|
||||
self.log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if self._log_handle is not None:
|
||||
|
||||
@@ -552,12 +552,7 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) ->
|
||||
return row
|
||||
|
||||
variant = choice.variant
|
||||
# Same overhead the launch decision prices (runtime buffers + vision projector + microbatch/MTP
|
||||
# logits): the row must advertise the window the model will actually get, not a paper number.
|
||||
overhead = (context_policy.RUNTIME_OVERHEAD_BYTES
|
||||
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ context_policy.ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
|
||||
decision = context_policy.initial_window(entry.profile(variant), budget, overhead_bytes=overhead)
|
||||
decision = entry.launch_plan(variant, budget).decision
|
||||
download_total = entry.download_bytes(variant)
|
||||
row.update({
|
||||
"fits": True, "model_id": variant.model_id, "quant": variant.quant,
|
||||
|
||||
@@ -28,7 +28,7 @@ def _stage(home, name):
|
||||
def _write_presets(home, *model_ids):
|
||||
pdir = home / "runtimes" / "llamacpp"
|
||||
pdir.mkdir(parents=True, exist_ok=True)
|
||||
body = "\n".join(f"[{m}]\nctx-size = 65536\n" for m in model_ids)
|
||||
body = "\n".join(f"[{m}]\nmodel = {home / 'models' / (m + '.gguf')}\nctx-size = 65536\n" for m in model_ids)
|
||||
(pdir / "presets.ini").write_text(body, encoding="utf-8")
|
||||
|
||||
|
||||
@@ -49,6 +49,16 @@ def test_presets_current_when_every_staged_model_is_covered(hermes_home):
|
||||
assert _presets_stale() is False
|
||||
|
||||
|
||||
def test_legacy_presets_without_model_paths_are_regenerated(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
_stage(hermes_home, "model-a")
|
||||
_write_presets(hermes_home, "model-a")
|
||||
ini = hermes_home / "runtimes/llamacpp/presets.ini"
|
||||
ini.write_text("[model-a]\nctx-size = 65536\n")
|
||||
assert _presets_stale()
|
||||
|
||||
|
||||
def test_no_models_is_never_stale(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
|
||||
@@ -151,6 +151,45 @@ def test_find_entry_for_model_resolves_split_ids():
|
||||
assert variant.quant == "UD-Q4_K_XL"
|
||||
|
||||
|
||||
def test_catalog_and_preset_agree_on_identical_model_facts(tmp_path, monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import bootstrap, catalog, presets
|
||||
from hermes_cli.local_runtime.context_policy import RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import ctx_bytes
|
||||
from hermes_cli.web_routers.local_models import _catalog_row
|
||||
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
monkeypatch.setattr("hermes_cli.web_routers.local_models._engine_too_old", lambda tag: False)
|
||||
for entry in catalog.CATALOG:
|
||||
variant = entry.variants[0]
|
||||
profile = entry.profile(variant)
|
||||
path = tmp_path / f"{variant.model_id}.gguf"
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
if entry.mmproj:
|
||||
asset = bootstrap.assets_dir() / entry.mmproj.local_name
|
||||
asset.parent.mkdir(parents=True, exist_ok=True)
|
||||
asset.touch()
|
||||
for vram in (16, 24, 32, 48):
|
||||
for uma in (False, True):
|
||||
machine = HardwareBudget(int(vram * GIB * 0.8), vram * GIB,
|
||||
0 if uma else 32 * GIB, uma)
|
||||
row = _catalog_row(entry, machine, None, None, set())
|
||||
preset = presets.preset_for_model(path, machine, set())
|
||||
assert row["fits"] == (preset.refusal is None)
|
||||
if preset.refusal:
|
||||
continue
|
||||
assert row["start_window"] == preset.window
|
||||
assert row["spilled"] == preset.spilled
|
||||
overhead = (RUNTIME_OVERHEAD_BYTES + (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ ub_logits_bytes(profile.n_vocab, mtp_capable=entry.mtp,
|
||||
mtp_prefill=preset.keys.get("ubatch-size") == "2048" and entry.mtp))
|
||||
need = profile.weights_bytes + ctx_bytes(profile, preset.window) + overhead
|
||||
assert preset.spilled == (need > machine.usable_vram_bytes)
|
||||
assert need <= machine.usable_vram_bytes + machine.ram_available_bytes
|
||||
|
||||
|
||||
def test_hybrid_long_context_stays_cheap():
|
||||
"""The reason Nemotron/Qwen3.6 headline the catalog: their priced
|
||||
64K-floor KV must be a small fraction of a dense model's."""
|
||||
|
||||
@@ -177,6 +177,38 @@ def test_physics_check_prices_at_floor_not_native():
|
||||
assert physics_check(p, card(24, ram_gib=8), FLOOR) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("uma", [False, True])
|
||||
def test_initial_window_accounts_for_overhead_in_every_verdict(uma):
|
||||
from dataclasses import replace
|
||||
|
||||
profile = hybrid(weights_gib=8, native=FLOOR)
|
||||
base = profile.weights_bytes + ctx_bytes(profile, FLOOR)
|
||||
overhead = 2 * GIB
|
||||
budget = HardwareBudget(base + GIB, base + GIB, 0 if uma else 4 * GIB, uma)
|
||||
decision = initial_window(profile, budget, overhead_bytes=overhead)
|
||||
if uma:
|
||||
assert isinstance(decision, PhysicsRefusal)
|
||||
assert decision.needed_bytes == base + overhead
|
||||
else:
|
||||
assert isinstance(decision, WindowDecision)
|
||||
assert decision.spill_bytes == GIB
|
||||
|
||||
exact = replace(budget, usable_vram_bytes=base + overhead, ram_available_bytes=0)
|
||||
assert not initial_window(profile, exact, overhead_bytes=overhead).spilled
|
||||
short = replace(exact, usable_vram_bytes=exact.usable_vram_bytes - 1)
|
||||
assert isinstance(initial_window(profile, short, overhead_bytes=overhead), PhysicsRefusal)
|
||||
|
||||
# A cheap-KV model may grow on the spill path, but only into memory that exists.
|
||||
growing = hybrid(weights_gib=20, full_layers=4, recurrent_layers=0,
|
||||
per_token_f16=1024, native=1024 * KIB)
|
||||
floor_need = growing.weights_bytes + ctx_bytes(growing, FLOOR) + overhead
|
||||
limited = HardwareBudget(8 * GIB, 8 * GIB, floor_need - 8 * GIB)
|
||||
decision = initial_window(growing, limited, overhead_bytes=overhead)
|
||||
assert isinstance(decision, WindowDecision)
|
||||
assert decision.window == FLOOR
|
||||
assert decision.spill_bytes == limited.ram_available_bytes
|
||||
|
||||
|
||||
# ── ladder + initial window ──────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -230,6 +230,121 @@ def test_preset_restores_grown_window_midladder(hermes_home, tmp_path, monkeypat
|
||||
assert restored.window >= grown, "override must lift the launch window"
|
||||
|
||||
|
||||
def test_mtp_plan_matches_cost_at_initial_and_restored_windows(hermes_home, tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import presets
|
||||
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
|
||||
gib = 1 << 30
|
||||
profile = ModelProfile(name="mtp-fit", weights_bytes=16 * gib, embd_table_bytes=0,
|
||||
n_ctx_train=262144, layers=[(LayerKind.FULL, 4096)] * 32,
|
||||
moe=True, n_vocab=151936)
|
||||
priced = replace(profile, kv_scale=1.2)
|
||||
lean = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
stacked = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, profile.name)
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
|
||||
def generate(device, ram, override=0):
|
||||
save_window_override(profile.name, override)
|
||||
budget = HardwareBudget(device, device, ram)
|
||||
return presets.generate_presets(mdir, budget, tmp_path / "presets.ini", {profile.name})[0]
|
||||
|
||||
floor_need = profile.weights_bytes + ctx_bytes(priced, FLOOR)
|
||||
initial = generate(floor_need + lean, 8 * gib)
|
||||
assert initial.window == FLOOR
|
||||
assert not initial.spilled
|
||||
assert "ubatch-size" not in initial.keys
|
||||
assert initial.keys["spec-type"] == "draft-mtp"
|
||||
# Persisting a floor grant must not turn a lean spilled boot into stacked prefill.
|
||||
for override in (0, FLOOR, 73728):
|
||||
spilled_boot = generate(16 * gib, 64 * gib, override)
|
||||
assert spilled_boot.spilled and "ubatch-size" not in spilled_boot.keys
|
||||
|
||||
device = floor_need + stacked
|
||||
control = generate(device, 8 * gib)
|
||||
assert control.keys["ubatch-size"] == "2048"
|
||||
grown_window = 73728
|
||||
for ram in (8 * gib, 0):
|
||||
# Also preserve a grown window when stacked exceeds total memory, not just VRAM.
|
||||
grown = generate(device, ram, grown_window)
|
||||
assert grown.window == grown_window
|
||||
assert not grown.spilled
|
||||
assert "ubatch-size" not in grown.keys
|
||||
assert "override-tensor" not in grown.keys
|
||||
assert grown.keys["spec-type"] == "draft-mtp"
|
||||
assert profile.weights_bytes + ctx_bytes(priced, grown.window) + lean <= device
|
||||
|
||||
both_spill = generate(device, 64 * gib, 147456)
|
||||
assert both_spill.window == 147456 and both_spill.spilled
|
||||
assert both_spill.keys["ubatch-size"] == "2048"
|
||||
assert "override-tensor" in both_spill.keys
|
||||
smaller_boot = generate(floor_need + lean, 0, grown_window)
|
||||
assert smaller_boot.window == FLOOR and not smaller_boot.spilled
|
||||
assert profile.weights_bytes + ctx_bytes(priced, control.window) + stacked <= device
|
||||
|
||||
|
||||
def test_growth_requires_an_admissible_materialized_preset(hermes_home, tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import bootstrap, catalog, growth, hardware, presets
|
||||
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, ctx_bytes
|
||||
|
||||
entry = next(e for e in catalog.CATALOG if e.mtp and e.mmproj)
|
||||
model_id = entry.variants[-1].model_id
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, model_id)
|
||||
profile = replace(entry.profile(entry.variants[-1]), kv_scale=1.0)
|
||||
monkeypatch.setattr(bootstrap, "staged_models", lambda: list(mdir.glob("*.gguf")))
|
||||
monkeypatch.setattr(bootstrap, "get_supervisor", lambda: SimpleNamespace(is_idle=lambda m: True))
|
||||
monkeypatch.setattr(growth, "is_managed_endpoint", lambda url: True)
|
||||
from hermes_cli.local_runtime import gguf, estimator
|
||||
monkeypatch.setattr(gguf, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(estimator, "profile_from_gguf", lambda h: profile)
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
asset = bootstrap.assets_dir() / entry.mmproj.local_name
|
||||
asset.parent.mkdir(parents=True, exist_ok=True)
|
||||
asset.touch()
|
||||
overhead = RUNTIME_OVERHEAD_BYTES + entry.mmproj.size_bytes + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
priced = replace(profile, kv_scale=1.2)
|
||||
next_window = FLOOR * 3 // 2
|
||||
floor_need = profile.weights_bytes + ctx_bytes(priced, FLOOR) + overhead
|
||||
next_need = profile.weights_bytes + ctx_bytes(priced, next_window) + overhead
|
||||
budget = HardwareBudget(floor_need, floor_need, 0, True)
|
||||
monkeypatch.setattr(hardware, "probe_budget", lambda **kw: budget)
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
preset_path = runtimes_root() / "presets.ini"
|
||||
calls = []
|
||||
|
||||
def refresh():
|
||||
calls.append(True)
|
||||
presets.generate_presets(mdir, budget, preset_path)
|
||||
return True
|
||||
|
||||
monkeypatch.setattr(bootstrap, "refresh_local_runtime", refresh)
|
||||
args = dict(base_url="http://127.0.0.1:1/v1", session_tokens=FLOOR, current_window=FLOOR)
|
||||
assert growth.maybe_grow_window(model_id, **args) is None
|
||||
assert not calls and not growth.load_window_overrides()
|
||||
budget = replace(budget, usable_vram_bytes=next_need, total_device_bytes=next_need)
|
||||
assert growth.maybe_grow_window(model_id, **args) == next_window
|
||||
assert presets.read_preset_decisions(preset_path)[model_id].window == next_window
|
||||
assert growth.load_window_overrides()[model_id] == next_window
|
||||
|
||||
# A restart that claims success but does not materialize the grant must not tell the agent it grew.
|
||||
monkeypatch.setattr(bootstrap, "refresh_local_runtime", lambda: True)
|
||||
budget = replace(budget, usable_vram_bytes=64 << 30, total_device_bytes=64 << 30)
|
||||
assert growth.maybe_grow_window(model_id, **{**args, "current_window": next_window}) is None
|
||||
|
||||
|
||||
def test_sampling_ladder_file_beats_catalog_beats_nothing(hermes_home, tmp_path, monkeypatch):
|
||||
"""The sampling deference ladder: the GGUF's own general.sampling.*
|
||||
wins per key, catalog fills only what the file left silent, and a
|
||||
|
||||
@@ -66,6 +66,61 @@ def test_status_lists_staged_models_with_labels(client, tmp_path):
|
||||
assert row["size_label"].endswith("GB")
|
||||
|
||||
|
||||
def test_status_tracks_preset_spill_and_restored_window(client, tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import bootstrap, presets
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
from hermes_cli.web_routers import local_models
|
||||
|
||||
# Dense spill has no override-tensor flag: status must use the recorded decision.
|
||||
profile = ModelProfile("status-mtp", 16 << 30, 0, 262144,
|
||||
[(LayerKind.FULL, 4096)] * 32, n_vocab=151936)
|
||||
model_id = profile.name
|
||||
_write_fake_gguf(bootstrap.models_dir() / f"{model_id}.gguf")
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
monkeypatch.setattr(local_models, "_state_endpoint", lambda: {"base_url": "http://127.0.0.1:1/v1"})
|
||||
|
||||
server_window = FLOOR
|
||||
|
||||
def router_response(running, route, **kwargs):
|
||||
if route == "/models":
|
||||
return {"data": [{"id": model_id, "status": {"value": "loaded"}}]}
|
||||
assert route == f"/props?model={model_id}"
|
||||
return {"default_generation_settings": {"n_ctx": server_window}}
|
||||
|
||||
monkeypatch.setattr(local_models, "_router_request", router_response)
|
||||
floor_need = profile.weights_bytes + ctx_bytes(replace(profile, kv_scale=1.2), FLOOR)
|
||||
lean = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
stacked = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
|
||||
ini = runtimes_root() / "presets.ini"
|
||||
grown = 73728
|
||||
for device, override, spilled in ((floor_need + lean - 1, FLOOR, True),
|
||||
(floor_need + stacked, grown, False)):
|
||||
save_window_override(model_id, override)
|
||||
preset = presets.generate_presets(bootstrap.models_dir(),
|
||||
HardwareBudget(device, device, 8 << 30), ini, {model_id})[0]
|
||||
assert preset.window == override and preset.spilled is spilled
|
||||
assert preset.keys["spec-type"] == "draft-mtp"
|
||||
assert "ubatch-size" not in preset.keys and "override-tensor" not in preset.keys
|
||||
# Deliberately differ from the plan to prove the server remains the grant authority.
|
||||
server_window = preset.window - 1024
|
||||
response = client.get("/api/local-models/status")
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert data["loaded_models"][model_id] == "loaded"
|
||||
placement = data["placement"][model_id]
|
||||
assert placement["spilled"] is spilled
|
||||
assert placement["window"] == preset.window
|
||||
assert placement["granted_window"] == server_window
|
||||
assert placement["granted_window_label"] == local_models._k_label(server_window)
|
||||
|
||||
|
||||
# ── hardware ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
"""The router must serve only admitted presets, retaining refusal and spill facts on read-back."""
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import presets, supervisor
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, ModelProfile
|
||||
|
||||
|
||||
def test_preset_roundtrip_keeps_refusals_and_dense_spill(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
mdir = tmp_path / "models"
|
||||
mdir.mkdir()
|
||||
for name in ("allowed", "refused"):
|
||||
(mdir / f"{name}.gguf").touch()
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(path=p, sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: ModelProfile(
|
||||
name=h.path.stem, weights_bytes=(4 if h.path.stem == "allowed" else 40) << 30,
|
||||
embd_table_bytes=0, n_ctx_train=65536, layers=[]))
|
||||
ini = tmp_path / "presets.ini"
|
||||
generated = presets.generate_presets(mdir, HardwareBudget(2 << 30, 2 << 30, 8 << 30), ini)
|
||||
reread = presets.read_preset_decisions(ini)
|
||||
assert set(reread) == {p.model_id for p in generated}
|
||||
assert reread["refused"].refusal
|
||||
assert reread["allowed"].spilled
|
||||
assert reread["allowed"].keys["model"] == str(mdir / "allowed.gguf")
|
||||
assert "override-tensor" not in reread["allowed"].keys # Dense spill has no tensor-pattern override.
|
||||
|
||||
|
||||
def test_optional_draft_is_enabled_only_with_room_at_the_selected_window(tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from hermes_cli.local_runtime import bootstrap, catalog
|
||||
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
entry = next(e for e in catalog.CATALOG if e.draft)
|
||||
main = tmp_path / f"{entry.variants[0].model_id}.gguf"
|
||||
draft = bootstrap.assets_dir() / entry.draft.local_name
|
||||
draft.parent.mkdir(parents=True, exist_ok=True)
|
||||
draft.touch()
|
||||
main_profile = ModelProfile("main", 12 << 30, 0, 65536, [], moe=True)
|
||||
draft_profile = ModelProfile("draft", 1 << 30, 0, 65536, [])
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(path=p, sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: draft_profile if h.path == draft else main_profile)
|
||||
tight = HardwareBudget(8 << 30, 8 << 30, 6 << 30)
|
||||
result = presets.preset_for_model(main, tight, set())
|
||||
assert result.window == 65536 and result.spilled
|
||||
assert "model-draft" not in result.keys
|
||||
roomy = replace(tight, ram_available_bytes=16 << 30)
|
||||
with_draft = presets.preset_for_model(main, roomy, set())
|
||||
assert with_draft.window == result.window
|
||||
assert with_draft.keys["model-draft"] == str(draft)
|
||||
assert with_draft.keys["spec-type"] == "draft-dspark"
|
||||
# Full target-window f16 state and logits count even above the draft's native window.
|
||||
from hermes_cli.local_runtime.context_policy import RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import LayerKind, ctx_bytes
|
||||
|
||||
draft_profile = replace(draft_profile, n_ctx_train=32768,
|
||||
layers=[(LayerKind.FULL, 4096)] * 4, n_vocab=32768)
|
||||
draft_cost = (draft_profile.weights_bytes
|
||||
+ ctx_bytes(draft_profile, result.window, flash_attention=False)
|
||||
+ RUNTIME_OVERHEAD_BYTES
|
||||
+ ub_logits_bytes(draft_profile.n_vocab, mtp_capable=False))
|
||||
device_boundary = RUNTIME_OVERHEAD_BYTES + draft_cost
|
||||
exact = replace(roomy, usable_vram_bytes=device_boundary, total_device_bytes=device_boundary)
|
||||
assert "model-draft" in presets.preset_for_model(main, exact, set()).keys
|
||||
below = replace(exact, usable_vram_bytes=device_boundary - 1)
|
||||
assert "model-draft" not in presets.preset_for_model(main, below, set()).keys
|
||||
|
||||
# A draft too large for GPU memory is optional, not permission to move its buffers to RAM.
|
||||
draft_profile = replace(draft_profile, weights_bytes=9 << 30)
|
||||
assert "model-draft" not in presets.preset_for_model(main, roomy, set()).keys
|
||||
|
||||
|
||||
def test_supervisor_with_presets_does_not_scan_unadmitted_files(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
ini = tmp_path / "presets.ini"
|
||||
ini.write_text("[allowed]\nmodel = allowed.gguf\nctx-size = 65536\n")
|
||||
calls = []
|
||||
monkeypatch.setattr(supervisor, "server_binary", lambda p: Path("llama-server"))
|
||||
monkeypatch.setattr(supervisor.subprocess, "Popen", lambda cmd, **kw: calls.append(cmd) or SimpleNamespace(pid=123))
|
||||
sup = supervisor.LlamaServerSupervisor(tmp_path, tmp_path, port=1234, preset_path=ini)
|
||||
try:
|
||||
sup._spawn()
|
||||
finally:
|
||||
sup._log_handle.close()
|
||||
assert "--models-preset" in calls[0]
|
||||
assert "--models-dir" not in calls[0]
|
||||
@@ -64,8 +64,14 @@ end-to-end and exposes no knobs:
|
||||
overflow in system RAM in the order that hurts least (expert weights
|
||||
first, never the attention cache), trading some speed to protect the
|
||||
context guarantee.
|
||||
- **Conversation compression only kicks in at the model's maximum
|
||||
window** — growth always comes first.
|
||||
- **Memory fit includes the launch configuration**, not just the model file:
|
||||
context state, runtime buffers, the vision projector, and MTP buffers all
|
||||
count. For multi-token prediction (MTP), Hermes uses smaller batches when
|
||||
larger batches would spill at the same context window. MTP stays enabled.
|
||||
The same calculation runs when a grown window is restored after restart.
|
||||
- **Conversation compression follows a growth check.** If a larger window
|
||||
cannot fit, generation is too slow, or the native maximum is reached,
|
||||
Hermes compresses instead of claiming a window the server did not receive.
|
||||
- Idle models are unloaded after 15 minutes to free GPU memory; they
|
||||
reload automatically on the next message.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user