refactor(local_runtime): GGUFHeader int accessors via one property factory; supervisor best-effort child ops via _quiet; compact __init__ re-exports
This commit is contained in:
@@ -7,36 +7,16 @@ already-running llama-server (external or ours).
|
||||
"""
|
||||
|
||||
from hermes_cli.local_runtime.binaries import ( # noqa: F401
|
||||
BinaryResolutionError,
|
||||
ensure_runtime_installed,
|
||||
resolve_assets,
|
||||
select_backend,
|
||||
)
|
||||
from hermes_cli.local_runtime.bootstrap import ( # noqa: F401
|
||||
ensure_local_runtime,
|
||||
shutdown_local_runtime,
|
||||
)
|
||||
BinaryResolutionError, ensure_runtime_installed, resolve_assets, select_backend)
|
||||
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime, shutdown_local_runtime # noqa: F401
|
||||
from hermes_cli.local_runtime.context_policy import ( # noqa: F401
|
||||
FLOOR,
|
||||
growth_decision,
|
||||
initial_window,
|
||||
ladder,
|
||||
launch_args,
|
||||
)
|
||||
FLOOR, growth_decision, initial_window, ladder, launch_args)
|
||||
from hermes_cli.local_runtime.growth import ( # noqa: F401
|
||||
clear_window_override,
|
||||
load_window_overrides,
|
||||
maybe_grow_window,
|
||||
save_window_override,
|
||||
)
|
||||
clear_window_override, load_window_overrides, maybe_grow_window, save_window_override)
|
||||
from hermes_cli.local_runtime.detect import detect_server # noqa: F401
|
||||
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint # noqa: F401
|
||||
from hermes_cli.local_runtime.estimator import ( # noqa: F401
|
||||
HardwareBudget,
|
||||
ctx_bytes,
|
||||
physics_check,
|
||||
profile_from_gguf,
|
||||
)
|
||||
HardwareBudget, ctx_bytes, physics_check, profile_from_gguf)
|
||||
from hermes_cli.local_runtime.gguf import read_gguf_header # noqa: F401
|
||||
from hermes_cli.local_runtime.hardware import probe_budget # noqa: F401
|
||||
from hermes_cli.local_runtime.presets import generate_presets # noqa: F401
|
||||
|
||||
@@ -67,12 +67,19 @@ class GGUFHeader:
|
||||
def _arch_key(self, suffix: str):
|
||||
return self.metadata.get(f"{self.architecture}.{suffix}")
|
||||
|
||||
def _arch_int(self, suffix: str) -> int:
|
||||
return int(self._arch_key(suffix) or 0)
|
||||
def _arch_int(suffix: str, doc: str = ""): # noqa: N805 — property factory, deleted below
|
||||
return property(lambda self: int(self._arch_key(suffix) or 0), doc=doc)
|
||||
|
||||
@property
|
||||
def n_layer(self) -> int:
|
||||
return self._arch_int("block_count")
|
||||
n_layer = _arch_int("block_count")
|
||||
n_ctx_train = _arch_int("context_length")
|
||||
n_embd = _arch_int("embedding_length")
|
||||
sliding_window = _arch_int("attention.sliding_window")
|
||||
expert_count = _arch_int("expert_count")
|
||||
full_attention_interval = _arch_int(
|
||||
"full_attention_interval",
|
||||
"GDN-hybrid discriminator (qwen35 family): every Nth layer is full attention, the rest "
|
||||
"are linear/recurrent. 0 = not present.")
|
||||
del _arch_int
|
||||
|
||||
@property
|
||||
def n_vocab(self) -> int:
|
||||
@@ -84,10 +91,6 @@ class GGUFHeader:
|
||||
toks = self.metadata.get("tokenizer.ggml.tokens")
|
||||
return len(toks) if isinstance(toks, list) else 0
|
||||
|
||||
@property
|
||||
def n_ctx_train(self) -> int:
|
||||
return self._arch_int("context_length")
|
||||
|
||||
@property
|
||||
def sampling_defaults(self) -> dict:
|
||||
"""Upstream's recommended sampling as preset INI keys, when the file carries it.
|
||||
@@ -106,10 +109,6 @@ class GGUFHeader:
|
||||
out[name] = str(int(num)) if num == int(num) else str(num)
|
||||
return out
|
||||
|
||||
@property
|
||||
def n_embd(self) -> int:
|
||||
return self._arch_int("embedding_length")
|
||||
|
||||
@property
|
||||
def n_head(self) -> int:
|
||||
v = self._arch_key("attention.head_count")
|
||||
@@ -117,12 +116,6 @@ class GGUFHeader:
|
||||
return int(max(v))
|
||||
return int(v or 0)
|
||||
|
||||
@property
|
||||
def full_attention_interval(self) -> int:
|
||||
"""GDN-hybrid discriminator (qwen35 family): every Nth layer is full attention, the rest
|
||||
are linear/recurrent. 0 = not present."""
|
||||
return self._arch_int("full_attention_interval")
|
||||
|
||||
def head_counts_kv(self) -> list[int]:
|
||||
"""Per-layer KV head counts; 0 marks a recurrent/linear layer (n_head_kv == 0).
|
||||
|
||||
@@ -155,14 +148,6 @@ class GGUFHeader:
|
||||
return int(v)
|
||||
return self.head_dim_k
|
||||
|
||||
@property
|
||||
def sliding_window(self) -> int:
|
||||
return self._arch_int("attention.sliding_window")
|
||||
|
||||
@property
|
||||
def expert_count(self) -> int:
|
||||
return self._arch_int("expert_count")
|
||||
|
||||
|
||||
def read_gguf_header(path: str | Path) -> GGUFHeader:
|
||||
path = Path(path)
|
||||
|
||||
@@ -28,15 +28,9 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
# args list -> INI keys. Flags the policy owns; everything else stays out of the preset.
|
||||
_FLAG_TO_KEY = {
|
||||
"-c": "ctx-size",
|
||||
"-b": "batch-size",
|
||||
"-ub": "ubatch-size",
|
||||
"-ctk": "cache-type-k",
|
||||
"-ctv": "cache-type-v",
|
||||
"-fa": "flash-attn",
|
||||
"-ot": "override-tensor",
|
||||
"--spec-type": "spec-type",
|
||||
"--spec-draft-n-max": "spec-draft-n-max",
|
||||
"-c": "ctx-size", "-b": "batch-size", "-ub": "ubatch-size",
|
||||
"-ctk": "cache-type-k", "-ctv": "cache-type-v", "-fa": "flash-attn",
|
||||
"-ot": "override-tensor", "--spec-type": "spec-type", "--spec-draft-n-max": "spec-draft-n-max",
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -43,6 +43,14 @@ def state_path() -> Path:
|
||||
return runtimes_root() / "server.json"
|
||||
|
||||
|
||||
def _quiet(fn) -> None:
|
||||
"""Best-effort call; a child that vanished mid-walk is not an error."""
|
||||
try:
|
||||
fn()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
with socket.socket() as s:
|
||||
s.bind(("127.0.0.1", 0))
|
||||
@@ -272,20 +280,13 @@ class LlamaServerSupervisor:
|
||||
children = []
|
||||
proc.terminate()
|
||||
for child in children:
|
||||
try:
|
||||
child.terminate()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
_quiet(child.terminate)
|
||||
try:
|
||||
proc.wait(timeout=15)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
for child in children:
|
||||
try:
|
||||
if child.is_running():
|
||||
child.kill()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
_quiet(lambda: child.is_running() and child.kill())
|
||||
|
||||
def _reap_orphaned_children(self) -> None:
|
||||
"""Kill model children orphaned by a router crash, before respawn.
|
||||
|
||||
Reference in New Issue
Block a user