From 9aa1f66484f86ce6e0da9c3365198d1a7e5238b7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 20:58:40 -0700 Subject: [PATCH] refactor(local_runtime): GGUFHeader int accessors via one property factory; supervisor best-effort child ops via _quiet; compact __init__ re-exports --- hermes_cli/local_runtime/__init__.py | 30 ++++---------------- hermes_cli/local_runtime/gguf.py | 39 ++++++++------------------ hermes_cli/local_runtime/presets.py | 12 ++------ hermes_cli/local_runtime/supervisor.py | 19 +++++++------ 4 files changed, 30 insertions(+), 70 deletions(-) diff --git a/hermes_cli/local_runtime/__init__.py b/hermes_cli/local_runtime/__init__.py index da81558953..94e612162d 100644 --- a/hermes_cli/local_runtime/__init__.py +++ b/hermes_cli/local_runtime/__init__.py @@ -7,36 +7,16 @@ already-running llama-server (external or ours). """ from hermes_cli.local_runtime.binaries import ( # noqa: F401 - BinaryResolutionError, - ensure_runtime_installed, - resolve_assets, - select_backend, -) -from hermes_cli.local_runtime.bootstrap import ( # noqa: F401 - ensure_local_runtime, - shutdown_local_runtime, -) + BinaryResolutionError, ensure_runtime_installed, resolve_assets, select_backend) +from hermes_cli.local_runtime.bootstrap import ensure_local_runtime, shutdown_local_runtime # noqa: F401 from hermes_cli.local_runtime.context_policy import ( # noqa: F401 - FLOOR, - growth_decision, - initial_window, - ladder, - launch_args, -) + FLOOR, growth_decision, initial_window, ladder, launch_args) from hermes_cli.local_runtime.growth import ( # noqa: F401 - clear_window_override, - load_window_overrides, - maybe_grow_window, - save_window_override, -) + clear_window_override, load_window_overrides, maybe_grow_window, save_window_override) from hermes_cli.local_runtime.detect import detect_server # noqa: F401 from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint # noqa: F401 from hermes_cli.local_runtime.estimator import ( # noqa: F401 - HardwareBudget, - ctx_bytes, - physics_check, - profile_from_gguf, -) + HardwareBudget, ctx_bytes, physics_check, profile_from_gguf) from hermes_cli.local_runtime.gguf import read_gguf_header # noqa: F401 from hermes_cli.local_runtime.hardware import probe_budget # noqa: F401 from hermes_cli.local_runtime.presets import generate_presets # noqa: F401 diff --git a/hermes_cli/local_runtime/gguf.py b/hermes_cli/local_runtime/gguf.py index e22f1b5d6c..ad753ceee0 100644 --- a/hermes_cli/local_runtime/gguf.py +++ b/hermes_cli/local_runtime/gguf.py @@ -67,12 +67,19 @@ class GGUFHeader: def _arch_key(self, suffix: str): return self.metadata.get(f"{self.architecture}.{suffix}") - def _arch_int(self, suffix: str) -> int: - return int(self._arch_key(suffix) or 0) + def _arch_int(suffix: str, doc: str = ""): # noqa: N805 — property factory, deleted below + return property(lambda self: int(self._arch_key(suffix) or 0), doc=doc) - @property - def n_layer(self) -> int: - return self._arch_int("block_count") + n_layer = _arch_int("block_count") + n_ctx_train = _arch_int("context_length") + n_embd = _arch_int("embedding_length") + sliding_window = _arch_int("attention.sliding_window") + expert_count = _arch_int("expert_count") + full_attention_interval = _arch_int( + "full_attention_interval", + "GDN-hybrid discriminator (qwen35 family): every Nth layer is full attention, the rest " + "are linear/recurrent. 0 = not present.") + del _arch_int @property def n_vocab(self) -> int: @@ -84,10 +91,6 @@ class GGUFHeader: toks = self.metadata.get("tokenizer.ggml.tokens") return len(toks) if isinstance(toks, list) else 0 - @property - def n_ctx_train(self) -> int: - return self._arch_int("context_length") - @property def sampling_defaults(self) -> dict: """Upstream's recommended sampling as preset INI keys, when the file carries it. @@ -106,10 +109,6 @@ class GGUFHeader: out[name] = str(int(num)) if num == int(num) else str(num) return out - @property - def n_embd(self) -> int: - return self._arch_int("embedding_length") - @property def n_head(self) -> int: v = self._arch_key("attention.head_count") @@ -117,12 +116,6 @@ class GGUFHeader: return int(max(v)) return int(v or 0) - @property - def full_attention_interval(self) -> int: - """GDN-hybrid discriminator (qwen35 family): every Nth layer is full attention, the rest - are linear/recurrent. 0 = not present.""" - return self._arch_int("full_attention_interval") - def head_counts_kv(self) -> list[int]: """Per-layer KV head counts; 0 marks a recurrent/linear layer (n_head_kv == 0). @@ -155,14 +148,6 @@ class GGUFHeader: return int(v) return self.head_dim_k - @property - def sliding_window(self) -> int: - return self._arch_int("attention.sliding_window") - - @property - def expert_count(self) -> int: - return self._arch_int("expert_count") - def read_gguf_header(path: str | Path) -> GGUFHeader: path = Path(path) diff --git a/hermes_cli/local_runtime/presets.py b/hermes_cli/local_runtime/presets.py index 6f2f36c81c..876de8d844 100644 --- a/hermes_cli/local_runtime/presets.py +++ b/hermes_cli/local_runtime/presets.py @@ -28,15 +28,9 @@ logger = logging.getLogger(__name__) # args list -> INI keys. Flags the policy owns; everything else stays out of the preset. _FLAG_TO_KEY = { - "-c": "ctx-size", - "-b": "batch-size", - "-ub": "ubatch-size", - "-ctk": "cache-type-k", - "-ctv": "cache-type-v", - "-fa": "flash-attn", - "-ot": "override-tensor", - "--spec-type": "spec-type", - "--spec-draft-n-max": "spec-draft-n-max", + "-c": "ctx-size", "-b": "batch-size", "-ub": "ubatch-size", + "-ctk": "cache-type-k", "-ctv": "cache-type-v", "-fa": "flash-attn", + "-ot": "override-tensor", "--spec-type": "spec-type", "--spec-draft-n-max": "spec-draft-n-max", } diff --git a/hermes_cli/local_runtime/supervisor.py b/hermes_cli/local_runtime/supervisor.py index 2fd5a2e80f..d70d8281c7 100644 --- a/hermes_cli/local_runtime/supervisor.py +++ b/hermes_cli/local_runtime/supervisor.py @@ -43,6 +43,14 @@ def state_path() -> Path: return runtimes_root() / "server.json" +def _quiet(fn) -> None: + """Best-effort call; a child that vanished mid-walk is not an error.""" + try: + fn() + except Exception: # noqa: BLE001 + pass + + def _free_port() -> int: with socket.socket() as s: s.bind(("127.0.0.1", 0)) @@ -272,20 +280,13 @@ class LlamaServerSupervisor: children = [] proc.terminate() for child in children: - try: - child.terminate() - except Exception: # noqa: BLE001 - pass + _quiet(child.terminate) try: proc.wait(timeout=15) except subprocess.TimeoutExpired: proc.kill() for child in children: - try: - if child.is_running(): - child.kill() - except Exception: # noqa: BLE001 - pass + _quiet(lambda: child.is_running() and child.kill()) def _reap_orphaned_children(self) -> None: """Kill model children orphaned by a router crash, before respawn.