From ebcbab4eb974f59c03124c0d5d861df4a67075f3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:28:14 -0700 Subject: [PATCH] refactor(hermes_cli/web_routers): messaging fallback table as tuple rows; local_models dict.update/compact literals (966/1073 LOC) --- hermes_cli/web_routers/local_models.py | 77 ++++++++----------------- hermes_cli/web_routers/messaging.py | 79 ++++++++++++-------------- 2 files changed, 59 insertions(+), 97 deletions(-) diff --git a/hermes_cli/web_routers/local_models.py b/hermes_cli/web_routers/local_models.py index e025260a98..0726bb621c 100644 --- a/hermes_cli/web_routers/local_models.py +++ b/hermes_cli/web_routers/local_models.py @@ -71,8 +71,7 @@ def _job(kind: str, target: str, model_id: str | None = None) -> Dict[str, Any]: "model_id": model_id, # catalog id for downloads; None otherwise "status": "running", # running | done | error "phase": "starting", # human-readable step name - "detail": "", "total_bytes": None, "done_bytes": 0, - "started_at": time.time(), "error": None, + "detail": "", "total_bytes": None, "done_bytes": 0, "started_at": time.time(), "error": None, } with _JOBS_LOCK: _JOBS[job["job_id"]] = job @@ -133,9 +132,8 @@ def _router_request(endpoint: Dict[str, Any], path: str, *, timeout: float, if payload is not None: headers["Content-Type"] = "application/json" data = json.dumps(payload).encode() - req = urllib.request.Request( - endpoint["base_url"].rsplit("/v1", 1)[0] + path, data=data, headers=headers, - method="POST" if payload is not None else None) + req = urllib.request.Request(endpoint["base_url"].rsplit("/v1", 1)[0] + path, data=data, headers=headers, + method="POST" if payload is not None else None) with urllib.request.urlopen(req, timeout=timeout) as r: return None if payload is not None else json.loads(r.read()) @@ -278,8 +276,7 @@ def _hf_url(repo: str, path: str) -> str: def _download_plan(entry, variant) -> list: """Everything a variant needs: split parts + mmproj/draft assets, as (url, dest, bytes) tuples.""" - plan = [(_hf_url(entry.repo, a.path), _models_dir() / a.local_name, a.size_bytes) - for a in variant.files] + plan = [(_hf_url(entry.repo, a.path), _models_dir() / a.local_name, a.size_bytes) for a in variant.files] plan += [(_hf_url(entry.repo, a.path), bootstrap.assets_dir() / a.local_name, a.size_bytes) for a in (entry.mmproj, entry.draft) if a is not None] return plan @@ -393,27 +390,21 @@ def _loaded_models(running: Dict[str, Any]) -> "tuple[Dict[str, str], Dict[str, # Everything resident or becoming resident: 'loading' renders as its own # state in the pane (a 20-GB load in flight is the most important thing # the pane can show). - loaded = { - m["id"]: m.get("status", {}).get("value", "unknown") - for m in data.get("data", []) - if m.get("status", {}).get("value") in ("loaded", "ready", "loading") - } + loaded = {m["id"]: m.get("status", {}).get("value", "unknown") for m in data.get("data", []) + if m.get("status", {}).get("value") in ("loaded", "ready", "loading")} placement: Dict[str, Any] = {} decisions = presets.read_preset_decisions() for model_id, state in loaded.items(): facts: Dict[str, Any] = {} plan = decisions.get(model_id) if plan is not None: - facts["window"] = plan.window - facts["window_label"] = _k_label(plan.window) - facts["spilled"] = plan.spilled + facts.update(window=plan.window, window_label=_k_label(plan.window), spilled=plan.spilled) if state in ("loaded", "ready"): try: props = _router_request(running, f"/props?model={model_id}", timeout=3) n_ctx = props.get("default_generation_settings", {}).get("n_ctx") if n_ctx: - facts["granted_window"] = int(n_ctx) - facts["granted_window_label"] = _k_label(int(n_ctx)) + facts.update(granted_window=int(n_ctx), granted_window_label=_k_label(int(n_ctx))) except Exception: # noqa: BLE001 pass if facts: @@ -488,25 +479,17 @@ def local_models_status(): # Never silent: an empty dict here renders as 'Not in memory' # on a machine whose VRAM is visibly full. logger.warning("loaded-models read failed: %r", exc) - loaded = {} return { - "enabled": bool(section.get("enabled")), - "tag": tag, - "configured_tag": configured_tag, - "update_available": update_available, - "runtime_installed": runtime_backend is not None, - "runtime_backend": runtime_backend, - "server_running": running is not None, - "server_base_url": (running or {}).get("base_url"), - "active_model_id": _active_llamacpp_model_id(), + "enabled": bool(section.get("enabled")), "tag": tag, "configured_tag": configured_tag, + "update_available": update_available, "runtime_installed": runtime_backend is not None, + "runtime_backend": runtime_backend, "server_running": running is not None, + "server_base_url": (running or {}).get("base_url"), "active_model_id": _active_llamacpp_model_id(), "loaded_models": loaded, # Live load progress per model (SSE-fed): {model_id: {stage, value, # percent}}. The chat's loading bar and the picker rows poll this. "loading": _loading_progress(), - "placement": placement, - "models": staged, - "models_dir": str(mdir), + "placement": placement, "models": staged, "models_dir": str(mdir), } @@ -527,9 +510,8 @@ def local_models_hardware(): budget = hardware.probe_budget() ram_total, ram_avail = hardware._ram_bytes() out = { - "uma": budget.uma, "vram_total_bytes": budget.total_device_bytes, - "vram_usable_bytes": budget.usable_vram_bytes, "ram_total_bytes": ram_total, - "ram_available_bytes": ram_avail, "vram_label": _human_gb(budget.total_device_bytes), + "uma": budget.uma, "vram_total_bytes": budget.total_device_bytes, "vram_usable_bytes": budget.usable_vram_bytes, + "ram_total_bytes": ram_total, "ram_available_bytes": ram_avail, "vram_label": _human_gb(budget.total_device_bytes), "gpu_name": None, "gpu_util_percent": None, "vram_used_bytes": None, } # GPU identity + live utilization (NVIDIA; other vendors degrade to None @@ -541,9 +523,7 @@ def local_models_hardware(): capture_output=True, text=True, timeout=5) if smi_exe else None if smi and smi.returncode == 0 and smi.stdout.strip(): name, util, used_mib = (x.strip() for x in smi.stdout.strip().splitlines()[0].split(",")) - out["gpu_name"] = name - out["gpu_util_percent"] = int(util) - out["vram_used_bytes"] = int(used_mib) << 20 + out.update(gpu_name=name, gpu_util_percent=int(util), vram_used_bytes=int(used_mib) << 20) except Exception: # noqa: BLE001 pass return out @@ -566,14 +546,11 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) -> dl = next((v for v in entry.variants if v.model_id in staged_ids), None) row: Dict[str, Any] = { "id": entry.id, "display_name": entry.display_name, "description": entry.description, - "native_context": entry.n_ctx_train, - "native_context_label": _k_label(entry.n_ctx_train), + "native_context": entry.n_ctx_train, "native_context_label": _k_label(entry.n_ctx_train), "recommended": entry.id == recommended, "recommended_reason": recommended_reason if entry.id == recommended else None, - "downloaded": dl is not None, - "downloaded_model_id": dl.model_id if dl else None, - "downloaded_quant": dl.quant if dl else None, - "mtp": entry.mtp, "vision": entry.mmproj is not None, + "downloaded": dl is not None, "downloaded_model_id": dl.model_id if dl else None, + "downloaded_quant": dl.quant if dl else None, "mtp": entry.mtp, "vision": entry.mmproj is not None, # Day-0 architectures need the llama.cpp release where their support # landed: True gates download/activate until the engine updates, but # the row still renders (visible + explained beats hidden). @@ -609,9 +586,7 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) -> if isinstance(decision, estimator.PhysicsRefusal): row["fit_summary"] = row["quant_reason"] return row - row["start_window"] = decision.window - row["start_window_label"] = _k_label(decision.window) - row["spilled"] = decision.spilled + row.update(start_window=decision.window, start_window_label=_k_label(decision.window), spilled=decision.spilled) if decision.window >= entry.n_ctx_train: shape = f"runs at its full {row['native_context_label']} context" else: @@ -676,10 +651,8 @@ def _runtime_progress_hook(job: Dict[str, Any]): state["asset_total"] = total or done plan_done = state["banked"] + done plan_total = state["banked"] + (total or 0) - job["phase"] = "downloading-runtime" - job["detail"] = f"Downloading the local engine{suffix} — {_human_gb(plan_done)}" - if total: - job["detail"] += f" of {_human_gb(plan_total)}" + _step(job, "downloading-runtime", f"Downloading the local engine{suffix} — {_human_gb(plan_done)}" + + (f" of {_human_gb(plan_total)}" if total else "")) job["done_bytes"] = plan_done job["total_bytes"] = plan_total or None elif stage == "extract": @@ -829,8 +802,7 @@ def _quickstart_target(body: QuickstartBody, budget): else: picked = catalog.recommended_entry(budget, _eligible_entries()) best = picked[0] if picked is not None else None - candidates = ([best] if best is not None else []) + [ - e for e in catalog.CATALOG if best is None or e.id != best.id] + candidates = ([best] if best is not None else []) + [e for e in catalog.CATALOG if best is None or e.id != best.id] for candidate in candidates: choice = catalog.select_variant(candidate, budget) if choice is not None and not _engine_too_old(candidate.min_engine): @@ -888,8 +860,7 @@ async def local_models_quickstart(body: QuickstartBody): _spawn_job(job, "lr-quickstart", _run, fail_msg="quickstart failed: %s", on_exit=_QUICKSTART_LOCK.release) return {"job_id": job["job_id"], "model_id": entry.id, "display_name": entry.display_name, - "needs_runtime": need_runtime, "needs_download": need_download, - "download_bytes": download_bytes} + "needs_runtime": need_runtime, "needs_download": need_download, "download_bytes": download_bytes} # ── server lifecycle: turn the engine on/off ───────────────── diff --git a/hermes_cli/web_routers/messaging.py b/hermes_cli/web_routers/messaging.py index e6f0ceeb48..29ec37e257 100644 --- a/hermes_cli/web_routers/messaging.py +++ b/hermes_cli/web_routers/messaging.py @@ -64,51 +64,42 @@ _telegram_onboarding_pairings = LateState("_telegram_onboarding_pairings") # Display labels for env vars not in OPTIONAL_ENV_VARS (HOME_CHANNEL_*, bridge # toggles, Twilio, HASS, Email, etc.) so the UI can still render a friendly label. +# (key, description, prompt, extra flags: url / password / advanced) _MESSAGING_ENV_FALLBACKS: dict[str, dict[str, Any]] = { - "SIGNAL_HTTP_URL": { - "description": "signal-cli REST API base URL, e.g. http://127.0.0.1:8080", "prompt": "Signal bridge URL", - "url": "https://github.com/bbernhard/signal-cli-rest-api", - }, - "SIGNAL_ACCOUNT": {"description": "Signal account phone number registered with the bridge", "prompt": "Signal account"}, - "SIGNAL_ALLOWED_USERS": {"description": "Comma-separated Signal users allowed to use the bot", "prompt": "Allowed Signal users"}, - "WHATSAPP_ENABLED": {"description": "Enable the WhatsApp gateway adapter", "prompt": "Enable WhatsApp", "advanced": True}, - "WHATSAPP_MODE": {"description": "WhatsApp bridge mode", "prompt": "WhatsApp mode", "advanced": True}, - "WHATSAPP_DM_POLICY": {"description": "How WhatsApp direct messages are authorized", "prompt": "WhatsApp DM policy", "advanced": True}, - "WHATSAPP_ALLOWED_USERS": {"description": "Comma-separated WhatsApp users allowed to use the bot", "prompt": "Allowed WhatsApp users"}, - "HASS_URL": {"description": "Home Assistant base URL, e.g. https://homeassistant.local:8123", "prompt": "Home Assistant URL"}, - "HASS_TOKEN": { - "description": "Long-lived access token from Home Assistant (Profile → Security)", - "prompt": "Home Assistant access token", "password": True, - }, - "EMAIL_ADDRESS": {"description": "Email address to send and receive from", "prompt": "Email address"}, - "EMAIL_PASSWORD": {"description": "Email account password or app password", "prompt": "Email password", "password": True}, - "EMAIL_IMAP_HOST": {"description": "IMAP server host (e.g. imap.gmail.com)", "prompt": "IMAP host"}, - "EMAIL_SMTP_HOST": {"description": "SMTP server host (e.g. smtp.gmail.com)", "prompt": "SMTP host"}, - "TWILIO_ACCOUNT_SID": {"description": "Twilio Account SID", "prompt": "Twilio Account SID", "url": "https://www.twilio.com/console"}, - "TWILIO_AUTH_TOKEN": {"description": "Twilio Auth Token", "prompt": "Twilio Auth Token", "password": True}, - "WECOM_BOT_ID": {"description": "WeCom group bot ID", "prompt": "WeCom Bot ID"}, - "WECOM_SECRET": {"description": "WeCom group bot secret", "prompt": "WeCom Secret", "password": True}, - "WECOM_CALLBACK_CORP_ID": {"description": "WeCom corp ID", "prompt": "WeCom Corp ID"}, - "WECOM_CALLBACK_CORP_SECRET": {"description": "WeCom app corp secret", "prompt": "WeCom Corp Secret", "password": True}, - "WECOM_CALLBACK_AGENT_ID": {"description": "WeCom app agent ID", "prompt": "WeCom Agent ID"}, - "WECOM_CALLBACK_TOKEN": {"description": "WeCom callback verification token", "prompt": "WeCom Token"}, - "WECOM_CALLBACK_ENCODING_AES_KEY": {"description": "WeCom callback AES encoding key", "prompt": "WeCom AES Key", "password": True}, - "WEIXIN_ACCOUNT_ID": { - "description": "iLink Bot account ID obtained through QR login in hermes gateway setup", "prompt": "iLink Bot account ID", - }, - "WEIXIN_TOKEN": { - "description": "iLink Bot token obtained through QR login in hermes gateway setup", "prompt": "iLink Bot token", - "password": True, - }, - "WEIXIN_BASE_URL": { - "description": "iLink API base URL saved by QR login (default: https://ilinkai.weixin.qq.com)", "prompt": "iLink API base URL", - }, - "FEISHU_APP_ID": {"description": "Feishu / Lark app ID", "prompt": "App ID"}, - "FEISHU_APP_SECRET": {"description": "Feishu / Lark app secret", "prompt": "App secret", "password": True}, - "FEISHU_ENCRYPT_KEY": {"description": "Feishu / Lark encrypt key", "prompt": "Encrypt key", "password": True}, - "FEISHU_VERIFICATION_TOKEN": {"description": "Feishu / Lark verification token", "prompt": "Verification token", "password": True}, - "DINGTALK_CLIENT_ID": {"description": "DingTalk client ID (App key)", "prompt": "Client ID"}, - "DINGTALK_CLIENT_SECRET": {"description": "DingTalk client secret (App secret)", "prompt": "Client secret", "password": True}, + key: {"description": description, "prompt": prompt, **extra} + for key, description, prompt, extra in ( + ("SIGNAL_HTTP_URL", "signal-cli REST API base URL, e.g. http://127.0.0.1:8080", "Signal bridge URL", {"url": "https://github.com/bbernhard/signal-cli-rest-api"}), + ("SIGNAL_ACCOUNT", "Signal account phone number registered with the bridge", "Signal account", {}), + ("SIGNAL_ALLOWED_USERS", "Comma-separated Signal users allowed to use the bot", "Allowed Signal users", {}), + ("WHATSAPP_ENABLED", "Enable the WhatsApp gateway adapter", "Enable WhatsApp", {"advanced": True}), + ("WHATSAPP_MODE", "WhatsApp bridge mode", "WhatsApp mode", {"advanced": True}), + ("WHATSAPP_DM_POLICY", "How WhatsApp direct messages are authorized", "WhatsApp DM policy", {"advanced": True}), + ("WHATSAPP_ALLOWED_USERS", "Comma-separated WhatsApp users allowed to use the bot", "Allowed WhatsApp users", {}), + ("HASS_URL", "Home Assistant base URL, e.g. https://homeassistant.local:8123", "Home Assistant URL", {}), + ("HASS_TOKEN", "Long-lived access token from Home Assistant (Profile → Security)", "Home Assistant access token", {"password": True}), + ("EMAIL_ADDRESS", "Email address to send and receive from", "Email address", {}), + ("EMAIL_PASSWORD", "Email account password or app password", "Email password", {"password": True}), + ("EMAIL_IMAP_HOST", "IMAP server host (e.g. imap.gmail.com)", "IMAP host", {}), + ("EMAIL_SMTP_HOST", "SMTP server host (e.g. smtp.gmail.com)", "SMTP host", {}), + ("TWILIO_ACCOUNT_SID", "Twilio Account SID", "Twilio Account SID", {"url": "https://www.twilio.com/console"}), + ("TWILIO_AUTH_TOKEN", "Twilio Auth Token", "Twilio Auth Token", {"password": True}), + ("WECOM_BOT_ID", "WeCom group bot ID", "WeCom Bot ID", {}), + ("WECOM_SECRET", "WeCom group bot secret", "WeCom Secret", {"password": True}), + ("WECOM_CALLBACK_CORP_ID", "WeCom corp ID", "WeCom Corp ID", {}), + ("WECOM_CALLBACK_CORP_SECRET", "WeCom app corp secret", "WeCom Corp Secret", {"password": True}), + ("WECOM_CALLBACK_AGENT_ID", "WeCom app agent ID", "WeCom Agent ID", {}), + ("WECOM_CALLBACK_TOKEN", "WeCom callback verification token", "WeCom Token", {}), + ("WECOM_CALLBACK_ENCODING_AES_KEY", "WeCom callback AES encoding key", "WeCom AES Key", {"password": True}), + ("WEIXIN_ACCOUNT_ID", "iLink Bot account ID obtained through QR login in hermes gateway setup", "iLink Bot account ID", {}), + ("WEIXIN_TOKEN", "iLink Bot token obtained through QR login in hermes gateway setup", "iLink Bot token", {"password": True}), + ("WEIXIN_BASE_URL", "iLink API base URL saved by QR login (default: https://ilinkai.weixin.qq.com)", "iLink API base URL", {}), + ("FEISHU_APP_ID", "Feishu / Lark app ID", "App ID", {}), + ("FEISHU_APP_SECRET", "Feishu / Lark app secret", "App secret", {"password": True}), + ("FEISHU_ENCRYPT_KEY", "Feishu / Lark encrypt key", "Encrypt key", {"password": True}), + ("FEISHU_VERIFICATION_TOKEN", "Feishu / Lark verification token", "Verification token", {"password": True}), + ("DINGTALK_CLIENT_ID", "DingTalk client ID (App key)", "Client ID", {}), + ("DINGTALK_CLIENT_SECRET", "DingTalk client secret (App secret)", "Client secret", {"password": True}), + ) }