refactor(update): split zip/stash/config/deps/git/maint clusters out of update_cmd.py

This commit is contained in:
Teknium
2026-09-02 16:00:26 -07:00
parent 096826bf7d
commit 4674f8b254
8 changed files with 5314 additions and 4890 deletions
+153 -4889
View File
File diff suppressed because it is too large Load Diff
+398
View File
@@ -0,0 +1,398 @@
"""Config-schema migration after ``hermes update``: run the fresh-process config check / migrate for the active profile and every sibling profile.
Split out of ``hermes_cli/update_cmd.py``; every moved name is re-imported there, so
``hermes_cli.update_cmd.<name>`` keeps resolving (and monkeypatching) as before.
Origin-internal helpers are imported lazily inside each function (no import cycle;
test patches on ``hermes_cli.update_cmd.<name>`` stay effective).
"""
import logging
import sys
from pathlib import Path
# Log-record parity with the origin module.
logger = logging.getLogger("hermes_cli.update_cmd")
def _reload_config_modules() -> None:
"""Force-reload modules from disk after git pull.
``hermes update`` runs in the PRE-pull process, so cached modules hold OLD
code: ``DEFAULT_CONFIG["_config_version"]`` is stale and
``check_config_version()`` reports "up to date" even when the pulled code
has a newer version with a migration to run. Reloads
``config_defaults`` / ``config`` / ``config_migrations`` from disk.
Also reloads ``_subprocess_compat`` and ``dashboard_procs`` so the later
dashboard cleanup (``_finish_dashboard_update_cleanup`` →
``_scan_dashboard_processes``) sees symbols the update added (e.g.
``bounded_probe_run``) instead of dying with ImportError in this process.
"""
import importlib
importlib.invalidate_caches()
for mod_name in (
"hermes_cli.config_defaults",
"hermes_cli.config",
"hermes_cli.config_migrations",
"hermes_cli._subprocess_compat",
"hermes_cli.dashboard_procs",
):
mod = sys.modules.get(mod_name)
if mod is not None:
try:
importlib.reload(mod)
except Exception as exc:
logger.debug("Could not reload %s for fresh post-update code: %s", mod_name, exc)
def _run_config_check_fresh() -> tuple:
"""Check config version using freshly-reloaded modules.
See ``_reload_config_modules`` for why this is necessary.
Returns ``(current_ver, latest_ver)``.
"""
from hermes_cli.update_cmd import _reload_config_modules
_reload_config_modules()
from hermes_cli.config import check_config_version
return check_config_version()
def _run_migrate_config_fresh(*, interactive: bool = False, quiet: bool = False) -> dict:
"""Run config migration using freshly-reloaded modules.
See ``_reload_config_modules`` for why this is necessary.
Returns the migration results dict.
"""
from hermes_cli.update_cmd import _reload_config_modules
_reload_config_modules()
from hermes_cli.config import migrate_config
return migrate_config(interactive=interactive, quiet=quiet)
def _migrate_sibling_profile_configs() -> list[tuple[str, int, int]]:
"""Migrate every SIBLING profile's config.yaml to the current version.
#91277 Phase 2 (fleet-wide config migration; #20438/#54926/#79048): the
shared checkout serves every profile, but ``hermes update`` historically
migrated only the active profile's config — siblings drifted versions
until their gateway hit a config the new code couldn't read.
Per profile home (skipping the active one, already migrated by the
caller): scope config reads/writes via the context-local HERMES_HOME
override (thread-safe — never ``os.environ``), check the version, and
run the NON-INTERACTIVE, quiet migration. Prompt-requiring settings are
left for the profile's own next interactive session, identical to the
gateway-mode contract for the active profile.
Returns ``[(profile_name, from_version, to_version), ...]`` for profiles
actually migrated. Never raises; a failing profile is skipped (its own
startup migration remains the fallback).
"""
from hermes_cli.update_cmd import _run_config_check_fresh, _run_migrate_config_fresh
migrated: list[tuple[str, int, int]] = []
try:
from hermes_constants import (
get_process_hermes_home,
reset_hermes_home_override,
set_hermes_home_override,
)
from hermes_cli.profiles import _get_profiles_root, _PROFILE_ID_RE
active_home = get_process_hermes_home()
root = _get_profiles_root()
if not root.is_dir():
return migrated
for entry in sorted(root.iterdir()):
if not entry.is_dir() or not _PROFILE_ID_RE.match(entry.name):
continue
try:
if entry.resolve() == Path(active_home).resolve():
continue
except OSError:
continue
if not (entry / "config.yaml").is_file():
continue # profile never configured — nothing to migrate
token = set_hermes_home_override(entry)
try:
current_ver, latest_ver = _run_config_check_fresh()
if current_ver >= latest_ver:
continue
_run_migrate_config_fresh(interactive=False, quiet=True)
after_ver, _ = _run_config_check_fresh()
if after_ver > current_ver:
migrated.append((entry.name, current_ver, after_ver))
except Exception as exc:
logger.debug(
"Config migration for profile %s failed: %s", entry.name, exc
)
finally:
reset_hermes_home_override(token)
except Exception as exc:
logger.debug("Sibling profile enumeration failed: %s", exc)
return migrated
def _check_and_apply_config_migration(
*,
assume_yes: bool = False,
gateway_mode: bool = False,
pre_update_snapshot_id: str | None = None,
) -> None:
"""Check and apply configuration migrations on an update completion path (#91360).
Must use freshly-reloaded modules (see ``_reload_config_modules``), and
must run on EVERY completion path — normal post-pull, venv-repair retry,
and the Node-deps repair on the ``commit_count == 0`` branch — so an
interrupted update that already pulled new code doesn't strand the user
on an older config version.
"""
from hermes_cli.update_cmd import (
_gateway_prompt,
_migrate_sibling_profile_configs,
_reload_config_modules,
_run_config_check_fresh,
_run_migrate_config_fresh,
)
print()
print("→ Checking configuration for new options...")
# Reload config modules BEFORE any config reads so get_missing_*,
# check_config_version, and migrate_config all use the updated code.
_reload_config_modules()
from hermes_cli.config import (
get_missing_env_vars,
get_missing_config_fields,
)
# Defensive (#91360): this helper runs on repair/retry completion paths
# too — a config-check failure must not break an otherwise-successful
# update. Log, point at the manual command, and return.
try:
missing_env = get_missing_env_vars(required_only=True)
missing_config = get_missing_config_fields()
current_ver, latest_ver = _run_config_check_fresh()
except Exception as exc:
logger.debug("Config check during update failed: %s", exc)
print(" ⚠️ Could not check config version.")
print(" Run 'hermes config migrate' to check manually.")
return
has_new_options = bool(missing_env or missing_config)
version_bump_only = (
not has_new_options and current_ver < latest_ver
)
needs_migration = has_new_options or current_ver < latest_ver
if version_bump_only:
# Only the format version changed (new defaults merge transparently).
# Prompting "configure new options now?" would look like a no-op on
# yes (ScottFive / Tt2021) — apply silently and say what happened.
print()
print(
f" ℹ Updating config format (v{current_ver} → v{latest_ver})…"
)
try:
_mig_results = _run_migrate_config_fresh(
interactive=False, quiet=True
)
print(" ✓ Config format updated (no new settings to configure)")
# quiet=True also mutes steps that RESET/REMOVE a setting (e.g. the
# v33→v34 personality reset, #81946). Re-surface them so an
# unattended update never silently changes config (#86656). Here
# missing_config is empty, so config_added holds only mutations.
for _note in _mig_results.get("config_added") or []:
print(f" ℹ {_note}")
for _warn in _mig_results.get("warnings") or []:
print(f" ⚠️ {_warn}")
except Exception as _mig_err:
print(f" ⚠️ Config format update failed: {_mig_err}")
print(" Run 'hermes config migrate' to retry.")
elif needs_migration:
print()
# Show WHAT changed, not just a count, so the user can make an
# informed yes/no decision (previously the prompt named nothing).
if missing_env:
print(
f" ⚠️ {len(missing_env)} new required setting(s) need configuration"
)
_print_items(missing_env, "New settings", "name")
if missing_config:
print(f" ℹ️ {len(missing_config)} new config option(s) available")
_print_items(missing_config, "New options", "key")
print()
if assume_yes:
print(
" ℹ --yes: auto-applying config migration (skipping API-key prompts)."
)
response = "y"
elif gateway_mode:
response = (
_gateway_prompt(
"Would you like to configure new options now? [Y/n]", "n"
)
.strip()
.lower()
)
elif not (sys.stdin.isatty() and sys.stdout.isatty()):
print(" ℹ Non-interactive session — applying safe config migrations.")
response = "auto"
else:
try:
response = (
input("Would you like to configure them now? [Y/n]: ")
.strip()
.lower()
)
except EOFError:
response = "n"
except UnicodeDecodeError:
# Non-UTF-8 locales / embedded terminals can make input()
# raise this; uncaught, it crashes the update at this prompt.
print(
" ⚠ Could not read input (encoding issue). Skipping. "
"Run 'hermes config migrate' manually to configure."
)
response = "n"
if response in {"", "y", "yes", "auto"}:
print()
# Gateway mode, --yes and non-interactive contexts can't prompt
# for API keys; still run the non-interactive pass so new defaults
# and version bumps land before the restarted gateway validates.
interactive_migration = not (
gateway_mode or assume_yes or response == "auto"
)
results = _run_migrate_config_fresh(interactive=interactive_migration, quiet=False)
if results["env_added"] or results["config_added"]:
print()
print("✓ Configuration updated!")
if (gateway_mode or assume_yes or response == "auto") and missing_env:
print(" ℹ API keys require manual entry: hermes config migrate")
else:
print()
print("Skipped. Run 'hermes config migrate' later to configure.")
else:
print(" ✓ Configuration is up to date")
# Fleet-wide config migration (#91277 Phase 2; #20438/#54926/#79048):
# the migration above touched only the active profile; siblings drifted
# (field repro: gateway on new code but config v33 vs v37). Run the same
# NON-INTERACTIVE migration per sibling home via the context-local
# HERMES_HOME override (never os.environ — other threads must not see it).
try:
_migrated_siblings = _migrate_sibling_profile_configs()
for _name, _from_ver, _to_ver in _migrated_siblings:
print(
f" ✓ Profile '{_name}': config format updated "
f"(v{_from_ver} → v{_to_ver})"
)
except Exception as exc:
logger.debug("Sibling config migration failed: %s", exc)
# Safety net: migrations have left cron/jobs.json valid-but-empty
# (#34600) and the desktop scheduler has overwritten it with a partial
# set (#52144). Restore from the pre-update snapshot if jobs went missing.
try:
from hermes_cli.backup import restore_cron_jobs_if_emptied
cron_restore = restore_cron_jobs_if_emptied(pre_update_snapshot_id)
if cron_restore:
print()
print(
" ⚠️ cron/jobs.json lost jobs during this update — "
f"restored {cron_restore['job_count']} job(s) from "
f"pre-update snapshot {cron_restore['snapshot_id']}."
)
except Exception as exc:
# Never let the cron safety net break an otherwise-good update.
logger.debug("Cron jobs auto-restore check failed: %s", exc)
# #64160: Desktop update/repair cycles have rewritten model.provider /
# model.default and dropped moa: (settings the gateway and cron consume).
# Restore only those protected keys from the same pre-update snapshot.
try:
from hermes_cli.backup import restore_config_model_settings_if_rewritten
cfg_restore = restore_config_model_settings_if_rewritten(
pre_update_snapshot_id
)
if cfg_restore:
print()
print(
" ⚠️ config.yaml user model settings were rewritten during "
f"this update — restored {', '.join(cfg_restore['keys'])} "
f"from pre-update snapshot {cfg_restore['snapshot_id']}."
)
except Exception as exc:
# Never let the config safety net break an otherwise-good update.
logger.debug("Config model-settings auto-restore check failed: %s", exc)
# #66140: run the same cron-jobs safety net for every sibling
# profile against ITS OWN pre-update snapshot (same-generation by
# construction — both taken by this run).
try:
from hermes_cli.backup import restore_cron_jobs_all_profiles
for _restored in restore_cron_jobs_all_profiles(
_LAST_SIBLING_SNAPSHOTS
):
print()
print(
f" ⚠️ Profile '{_restored['profile']}': cron/jobs.json "
f"lost jobs during this update — restored "
f"{_restored['job_count']} job(s) from pre-update "
f"snapshot {_restored['snapshot_id']}."
)
except Exception as exc:
logger.debug("Sibling cron auto-restore check failed: %s", exc)
# #64160: same config model-settings safety net for sibling profiles.
try:
from hermes_cli.backup import restore_config_model_settings_all_profiles
for _cfg_restored in restore_config_model_settings_all_profiles(
_LAST_SIBLING_SNAPSHOTS
):
print()
print(
f" ⚠️ Profile '{_cfg_restored['profile']}': config.yaml "
f"user model settings were rewritten during this update — "
f"restored {', '.join(_cfg_restored['keys'])} from "
f"pre-update snapshot {_cfg_restored['snapshot_id']}."
)
except Exception as exc:
logger.debug("Sibling config auto-restore check failed: %s", exc)
# {profile: snapshot_id} from this run's pre-update backup, consumed by the
# post-update per-profile cron-jobs safety net (#66140). Module-level because
# snapshot and restore run far apart in _cmd_update_impl.
_LAST_SIBLING_SNAPSHOTS: dict = {}
def _print_items(items, label, key, fallback_key=None):
if not items:
return
print(f" {label}:")
shown = items[:8]
for it in shown:
if isinstance(it, dict):
name = it.get(key) or (fallback_key and it.get(fallback_key)) or "?"
desc = (it.get("description") or "").strip()
else:
# Defensive: some callers/mocks pass bare name strings.
name = str(it)
desc = ""
if desc:
print(f" • {name} — {desc}")
else:
print(f" • {name}")
extra = len(items) - len(shown)
if extra > 0:
print(f" … and {extra} more")
File diff suppressed because it is too large Load Diff
-1
View File
@@ -1287,7 +1287,6 @@ def _restart_gateway_fleet_after_update(_pre_update_plan, gateway_mode: bool):
except Exception:
pass
# --- Post-restart survivor sweep (#17648) ---------------------
# Gateways that ignore SIGTERM (stuck drain, blocked I/O, zombie)
# never exit, so the 120s profile watcher never respawns and the
+801
View File
@@ -0,0 +1,801 @@
"""Git plumbing for ``hermes update``: fork/upstream sync, trampoline-git detection, lockfile/EOL churn cleanup, orphan rescue refs, parked-branch assessment, fetch-failure classification.
Split out of ``hermes_cli/update_cmd.py``; every moved name is re-imported there, so
``hermes_cli.update_cmd.<name>`` keeps resolving (and monkeypatching) as before.
Origin-internal helpers are imported lazily inside each function (no import cycle;
test patches on ``hermes_cli.update_cmd.<name>`` stay effective).
"""
import logging
import subprocess
import sys
from datetime import datetime
from pathlib import Path
from typing import Optional
# Log-record parity with the origin module.
logger = logging.getLogger("hermes_cli.update_cmd")
_ORPHAN_RESCUE_REFS_TO_KEEP = 10
_ORPHAN_RESCUE_REF_MAX_AGE_DAYS = 30
def _prune_orphan_rescue_refs(
git_cmd,
cwd,
branch,
keep=_ORPHAN_RESCUE_REFS_TO_KEEP,
max_age_days=_ORPHAN_RESCUE_REF_MAX_AGE_DAYS,
) -> None:
"""Expire old orphan rescue refs so backups stay bounded.
Each orphan-history divergence (#87694) parks the pre-reset HEAD under
``refs/hermes-update-backups/orphan-<branch>-<ts>-<sha>``. A rescue ref
pins its objects against ``git gc`` — in the incident shape a full
working-tree snapshot, potentially multi-GB — so a repeatedly corrupted
install would grow ``.git`` without bound.
Two limits, both enforced on every orphan incident: keep only the
``keep`` most-recent refs, and drop any older than ``max_age_days`` per
the ``YYYYMMDD-HHMMSS`` stamp in the ref name (unparseable names are left
alone). Names sort chronologically, so ``for-each-ref`` order is creation
order. Disk is reclaimed on the next ``git gc``. Best-effort: never
blocks the update.
"""
from hermes_cli.update_cmd import _git_run
try:
list_result = _git_run(
git_cmd,
["for-each-ref", "--format=%(refname)", "--sort=refname",
f"refs/hermes-update-backups/orphan-{branch}-*"],
cwd,
)
if list_result.returncode != 0:
return
refs = [line.strip() for line in list_result.stdout.splitlines() if line.strip()]
stale = set(refs[:-keep] if keep > 0 else refs)
# Age expiry: ref names embed a UTC YYYYMMDD-HHMMSS timestamp right
# after the branch segment; anything older than max_age_days goes.
if max_age_days > 0:
from datetime import timedelta, timezone
cutoff = datetime.now(timezone.utc) - timedelta(days=max_age_days)
prefix = f"refs/hermes-update-backups/orphan-{branch}-"
for ref in refs:
stamp = ref[len(prefix):][:15] # "YYYYMMDD-HHMMSS"
try:
ref_time = datetime.strptime(stamp, "%Y%m%d-%H%M%S").replace(
tzinfo=timezone.utc
)
except ValueError:
continue
if ref_time < cutoff:
stale.add(ref)
for ref in sorted(stale):
_git_run(git_cmd, ["update-ref", "-d", ref], cwd)
except OSError:
pass
def _branch_head_label(git_cmd=None, cwd=None) -> str | None:
"""``"<branch> @ <short-sha>"`` for the checkout, or None when unknown.
Appended to update summary lines so branch drift is visible (2026-08-17
incident: a checkout parked on a stale feature branch got "✓ Update
complete!" with nothing saying WHERE it sat). Never raises.
"""
from hermes_cli.update_cmd import _m
try:
cmd = list(git_cmd) if git_cmd else ["git"]
root = cwd if cwd is not None else _m().PROJECT_ROOT
branch = subprocess.run(
cmd + ["rev-parse", "--abbrev-ref", "HEAD"],
cwd=root, capture_output=True,
text=True, encoding="utf-8", errors="replace",
)
sha = subprocess.run(
cmd + ["rev-parse", "--short", "HEAD"],
cwd=root, capture_output=True,
text=True, encoding="utf-8", errors="replace",
)
branch_name = branch.stdout.strip()
sha_text = sha.stdout.strip()
if branch.returncode != 0 or sha.returncode != 0 or not sha_text:
return None
if not branch_name:
return None
label = "detached" if branch_name == "HEAD" else branch_name
return f"{label} @ {sha_text}"
except Exception:
return None
def _branch_head_suffix(git_cmd=None, cwd=None) -> str:
"""`` [<branch> @ <sha>]`` suffix for summary lines ("" when unknown)."""
label = _branch_head_label(git_cmd, cwd)
return f" [{label}]" if label else ""
def _assess_parked_branch_switch(
git_cmd: list[str], cwd: Path, current_branch: str, target_branch: str
) -> tuple[bool, str]:
"""Decide whether it is safe to auto-switch a parked feature branch back
to the update target.
Live incident (2026-08-17): the checkout sat on a stale feature branch;
``hermes update`` autostashed, ran post-update steps and printed
"✓ Code updated!" while the running code stayed days behind main.
- (True, "") — tree + index clean AND every parked commit is already in
``origin/<target_branch>`` (``git cherry`` reports no ``+`` lines).
- (True, "unmerged:<count>") — tree clean but the branch has commits not
in the target. Switching is safe (``git checkout`` never discards
committed work) but the caller must print a LOUD notice naming the
branch and count. Non-interactive callers (desktop button, gateway
/update, cron) rely on this: they can't resolve a skip, so a clean
checkout must always reach the target.
- (False, <reason>) — dirty tree, git errors, or the
``updates.auto_switch_parked_branch: false`` opt-out; caller must NOT
touch the branch. A dirty tree is the genuinely unsafe case: uncommitted
work riding an autostash across branches is how the incident started.
Block reasons: "disabled", "dirty", "unverifiable".
"""
from hermes_cli.update_cmd import _git_run
try:
from hermes_cli.config import load_config
_update_cfg = (load_config() or {}).get("updates", {})
if isinstance(_update_cfg, dict) and not bool(
_update_cfg.get("auto_switch_parked_branch", True)
):
return False, "disabled"
except Exception as exc:
# A config read failure must not disable the guard's safety checks —
# fall through to them with the default (auto-switch allowed).
logger.debug("Could not read updates.auto_switch_parked_branch: %s", exc)
status = _git_run(git_cmd, ["status", "--porcelain"], cwd)
if status.returncode != 0:
return False, "unverifiable"
if status.stdout.strip():
return False, "dirty"
cherry = _git_run(git_cmd, ["cherry", f"origin/{target_branch}"], cwd)
if cherry.returncode != 0:
return False, "unverifiable"
unmerged = [
line for line in cherry.stdout.splitlines() if line.startswith("+")
]
if unmerged:
# Clean tree: switching is safe (checkout keeps the commits on the
# branch). The reason string tells the caller to print the loud
# "branch kept with N unmerged commit(s)" notice.
return True, f"unmerged:{len(unmerged)}"
return True, ""
def _print_parked_branch_skip_warning(
git_cmd: list[str],
cwd: Path,
current_branch: str,
target_branch: str,
reason: str,
) -> None:
"""LOUD block explaining why the code update was skipped on a parked
branch, with the behind-count and the exact commands to resolve."""
from hermes_cli.update_cmd import _git_run
behind = None
try:
behind_result = _git_run(git_cmd, ["rev-list", f"HEAD..origin/{target_branch}", "--count"], cwd)
if behind_result.returncode == 0 and behind_result.stdout.strip():
behind = int(behind_result.stdout.strip())
except Exception:
behind = None
if reason == "dirty":
why = "the working tree has uncommitted changes"
elif reason == "disabled":
why = "updates.auto_switch_parked_branch is set to false in config.yaml"
else:
why = (
f"the branch state could not be verified against "
f"origin/{target_branch}"
)
bar = "=" * 68
print()
print(bar)
print(f"⚠ CODE UPDATE SKIPPED — checkout is parked on '{current_branch}'")
print(f" Not auto-switching to {target_branch}: {why}.")
if behind is not None and behind > 0:
print(
f" This checkout is {behind} commit(s) BEHIND "
f"origin/{target_branch} — the code you are running is stale."
)
print()
print(" To resolve, inspect the branch and switch back yourself:")
print(f" git -C {cwd} status")
print(f" git -C {cwd} checkout {target_branch} && hermes update")
print(
" (commit or stash your work on the branch first if you want to "
"keep it)"
)
print(bar)
def _print_parked_branch_kept_notice(
current_branch: str, target_branch: str, unmerged_count: str
) -> None:
"""LOUD notice printed when a clean parked branch with unmerged commits
is auto-switched back to the update target.
Non-interactive callers can't resolve a skip, so a clean checkout always
proceeds — but the unmerged work must be impossible to miss. The commits
stay on the branch (``git checkout`` never discards committed work).
"""
bar = "=" * 68
print()
print(bar)
print(
f"⚠ Checkout was parked on '{current_branch}' with "
f"{unmerged_count} commit(s) not merged into origin/{target_branch}."
)
print(
f" Switching to {target_branch} so the update can proceed — your "
f"commit(s) are safe on '{current_branch}'."
)
print()
print(" To pick the work back up later:")
print(f" git checkout {current_branch}")
print(bar)
OFFICIAL_REPO_URLS = {
"https://github.com/NousResearch/hermes-agent.git",
"git@github.com:NousResearch/hermes-agent.git",
"https://github.com/NousResearch/hermes-agent",
"git@github.com:NousResearch/hermes-agent",
}
OFFICIAL_REPO_URL = "https://github.com/NousResearch/hermes-agent.git"
SKIP_UPSTREAM_PROMPT_FILE = ".skip_upstream_prompt"
def _get_origin_url(git_cmd: list[str], cwd: Path) -> Optional[str]:
"""Get the URL of the origin remote, or None if not set."""
from hermes_cli.update_cmd import _git_run
try:
result = _git_run(git_cmd, ["remote", "get-url", "origin"], cwd)
if result.returncode == 0:
return result.stdout.strip()
except Exception:
pass
return None
def _is_fork(origin_url: Optional[str]) -> bool:
"""Check if the origin remote points to a fork (not the official repo)."""
if not origin_url:
return False
# Normalize URL for comparison (strip trailing .git if present)
normalized = origin_url.rstrip("/")
if normalized.endswith(".git"):
normalized = normalized[:-4]
for official in OFFICIAL_REPO_URLS:
official_normalized = official.rstrip("/")
if official_normalized.endswith(".git"):
official_normalized = official_normalized[:-4]
if normalized == official_normalized:
return False
return True
def _has_upstream_remote(git_cmd: list[str], cwd: Path) -> bool:
"""Check if an 'upstream' remote already exists."""
from hermes_cli.update_cmd import _git_run
try:
result = _git_run(git_cmd, ["remote", "get-url", "upstream"], cwd)
return result.returncode == 0
except Exception:
return False
def _add_upstream_remote(git_cmd: list[str], cwd: Path) -> bool:
"""Add the official repo as the 'upstream' remote. Returns True on success."""
from hermes_cli.update_cmd import _git_run
try:
result = _git_run(git_cmd, ["remote", "add", "upstream", OFFICIAL_REPO_URL], cwd)
return result.returncode == 0
except Exception:
return False
def _count_commits_between(git_cmd: list[str], cwd: Path, base: str, head: str) -> int:
"""Count commits on `head` that are not on `base`. Returns -1 on error."""
from hermes_cli.update_cmd import _git_run
try:
result = _git_run(git_cmd, ["rev-list", "--count", f"{base}..{head}"], cwd)
if result.returncode == 0:
return int(result.stdout.strip())
except Exception:
pass
return -1
def _should_skip_upstream_prompt() -> bool:
"""Check if user previously declined to add upstream."""
from hermes_constants import get_hermes_home
return (get_hermes_home() / SKIP_UPSTREAM_PROMPT_FILE).exists()
def _mark_skip_upstream_prompt():
"""Create marker file to skip future upstream prompts."""
try:
from hermes_constants import get_hermes_home
(get_hermes_home() / SKIP_UPSTREAM_PROMPT_FILE).touch()
except Exception:
pass
def _sync_fork_with_upstream(git_cmd: list[str], cwd: Path) -> bool:
"""Attempt to push updated main to origin (sync fork).
Returns True if push succeeded, False otherwise.
"""
from hermes_cli.update_cmd import _git_run
try:
result = _git_run(git_cmd, ["push", "origin", "main", "--force-with-lease"], cwd, network=True)
return result.returncode == 0
except Exception:
return False
def _sync_with_upstream_if_needed(
git_cmd: list[str],
cwd: Path,
*,
assume_yes: bool = False,
input_fn=None,
) -> bool:
"""Check if fork is behind upstream and fast-forward if safe.
Offers to add the ``upstream`` remote, compares origin/main with
upstream/main, pulls when strictly behind, then tries to push origin.
Returns True only when origin/main was actually verified against
upstream/main; False when the check never happened (prompt declined,
remote add/fetch/compare failed) so the caller never reports "up to date"
on an origin-only comparison (#97052).
"""
from hermes_cli.update_cmd import (
_add_upstream_remote,
_count_commits_between,
_has_upstream_remote,
_mark_skip_upstream_prompt,
_no_prompt_git_kwargs,
_should_skip_upstream_prompt,
)
has_upstream = _has_upstream_remote(git_cmd, cwd)
if not has_upstream:
if _should_skip_upstream_prompt():
return False
print()
print("ℹ Your fork is not tracking the official Hermes repository.")
print(" This means you may miss updates from NousResearch/hermes-agent.")
print()
if assume_yes or (
input_fn is None and not (sys.stdin.isatty() and sys.stdout.isatty())
):
# --yes means "don't block", not "mutate my git remotes". Skip
# without persisting the decline so interactive runs still get asked.
print(" Skipping upstream setup (non-interactive run).")
print(
" Add it later with: git remote add upstream https://github.com/NousResearch/hermes-agent.git"
)
return False
if input_fn is not None:
response = (
input_fn("Add official repo as 'upstream' remote? [y/N]", "n")
.strip()
.lower()
)
else:
try:
response = (
input("Add official repo as 'upstream' remote? [Y/n]: ")
.strip()
.lower()
)
except (EOFError, KeyboardInterrupt, UnicodeDecodeError):
print()
response = "n"
if response in {"", "y", "yes"}:
print("→ Adding upstream remote...")
if _add_upstream_remote(git_cmd, cwd):
print(
" ✓ Added upstream: https://github.com/NousResearch/hermes-agent.git"
)
has_upstream = True
else:
print(" ✗ Failed to add upstream remote. Skipping upstream sync.")
return False
else:
print(
" Skipped. Run 'git remote add upstream https://github.com/NousResearch/hermes-agent.git' to add later."
)
_mark_skip_upstream_prompt()
return False
# Fetch only upstream/main: a bare fetch drags in thousands of
# auto-generated branches.
print()
print("→ Fetching upstream...")
try:
subprocess.run(
git_cmd + ["fetch", "upstream", "main", "--quiet"],
cwd=cwd,
capture_output=True,
check=True,
**_no_prompt_git_kwargs(),
)
except subprocess.CalledProcessError:
print(" ✗ Failed to fetch upstream. Skipping upstream sync.")
return False
# Compare origin/main with upstream/main
origin_ahead = _count_commits_between(git_cmd, cwd, "upstream/main", "origin/main")
upstream_ahead = _count_commits_between(
git_cmd, cwd, "origin/main", "upstream/main"
)
if origin_ahead < 0 or upstream_ahead < 0:
print(" ✗ Could not compare branches. Skipping upstream sync.")
return False
# If origin/main has commits not on upstream, don't trample
if origin_ahead > 0:
print()
print(f"ℹ Your fork has {origin_ahead} commit(s) not on upstream.")
print(" Skipping upstream sync to preserve your changes.")
print(" If you want to merge upstream changes, run:")
print(" git pull upstream main")
return True
if upstream_ahead == 0:
print(" ✓ Fork is up to date with upstream")
return True
# origin/main is strictly behind upstream/main (can fast-forward)
print()
print(f"→ Fork is {upstream_ahead} commit(s) behind upstream")
print("→ Pulling from upstream...")
try:
subprocess.run(
git_cmd + ["pull", "--ff-only", "upstream", "main"],
cwd=cwd,
check=True,
**_no_prompt_git_kwargs(),
)
except subprocess.CalledProcessError:
print(
" ✗ Failed to pull from upstream. You may need to resolve conflicts manually."
)
return False
print(" ✓ Updated from upstream")
print("→ Syncing fork...")
if _sync_fork_with_upstream(git_cmd, cwd):
print(" ✓ Fork synced with upstream")
else:
print(
" ℹ Got updates from upstream but couldn't push to fork (no write access?)"
)
print(" Your local repo is updated, but your fork on GitHub may be behind.")
return True
def _classify_fetch_failure(stderr: str) -> str:
"""Map git-fetch stderr to a one-line, user-facing diagnosis.
Order matters: curl reports HTTP failures as ``unable to access '<url>':
The requested URL returned error: 429``, so the rate-limit/outage checks
must run BEFORE the generic "unable to access" network check. The caller
always prints the first raw stderr line too — this adds guidance, it
never replaces the wire error.
"""
def _has_http_code(*codes: str) -> bool:
return any(
f"HTTP {code}" in stderr or f"returned error: {code}" in stderr
for code in codes
)
if _has_http_code("429") or "rate limit" in stderr.lower():
return (
"✗ GitHub is rate limiting requests or having an outage (HTTP 429)"
" — try again in 5 minutes."
)
if _has_http_code("500", "502", "503", "504"):
return (
"✗ GitHub appears to be having an outage — try again in a few"
" minutes (https://www.githubstatus.com)."
)
if "Could not resolve host" in stderr or "unable to access" in stderr:
return "✗ Network error — cannot reach the remote repository."
if "could not read Username" in stderr or "terminal prompts disabled" in stderr:
# Anonymous fetch of a public repo got HTTP 401. GitHub does this
# during outages (and for renamed/private repos) — it is not a
# credentials problem on the user's side.
return (
"✗ GitHub rejected the anonymous fetch (asked for a login) — this"
" usually means a GitHub outage; try again in a few minutes"
" (https://www.githubstatus.com). If it persists, check"
" `git remote -v` points at a public repo."
)
if "Authentication failed" in stderr:
return "✗ Authentication failed — check your git credentials or SSH key."
return "✗ Failed to fetch updates from origin."
def _print_fetch_failure(stderr: str) -> None:
"""Print the classified diagnosis plus the first raw stderr line."""
stderr = (stderr or "").strip()
print(_classify_fetch_failure(stderr))
if stderr:
print(f" {stderr.splitlines()[0]}")
def _git_is_trampoline(git_cmd: list) -> bool:
"""Whether *git_cmd* resolves to a Git-for-Windows trampoline launcher.
Git for Windows ships ~46KB shims (``bin\\git.exe``, ``cmd\\git.exe``) that
re-exec ``mingw64\\libexec\\git-core\\git.exe``. When the shim cannot find
git-core, every git call dies with the launcher's guard message — a broken
PATH entry, not a network/filesystem problem (#87876). Never raises;
unknown states report False so a probe failure can't block an update.
"""
try:
result = subprocess.run(
git_cmd + ["--version"],
capture_output=True,
text=True, encoding="utf-8", errors="replace",
timeout=15,
)
except Exception:
return False
output = ((result.stdout or "") + (result.stderr or "")).lower()
return "fork bomb" in output
def _portable_git_candidates() -> list:
"""PortableGit candidate paths: shared root first, then profile home.
The Hermes-managed PortableGit tree lives under the SHARED root
(``<root>/git/...``), not the profile-scoped HERMES_HOME
(``<root>/profiles/<name>``), so a profile-scoped ``hermes update`` must
look there (monerostar review, #87876). The profile-home candidate is
kept as a fallback for custom layouts that place it there.
"""
from hermes_cli.update_cmd import get_default_hermes_root, get_hermes_home
candidates = []
try:
for root in (get_default_hermes_root(), Path(get_hermes_home())):
candidates.append(
root / "git" / "mingw64" / "libexec" / "git-core" / "git.exe"
)
except Exception:
pass
return candidates
def _locate_real_git() -> Optional[Path]:
"""Find a real Git-for-Windows binary that is not a broken trampoline.
The ~46KB ``bin\\git.exe`` / ``cmd\\git.exe`` shims fail to re-exec
git-core while ``mingw64\\libexec\\git-core\\git.exe`` (≈4.4MB) works
directly (#87876). Check standard Git for Windows locations plus the
Hermes-managed PortableGit; accept the first candidate that runs without
the trampoline guard. None when nothing suits — callers keep the broken
command and let the fetch-failure ZIP fallback handle it.
"""
candidates = [
Path(r"C:\Program Files\Git\mingw64\libexec\git-core\git.exe"),
Path(r"C:\Program Files (x86)\Git\mingw64\libexec\git-core\git.exe"),
] + _portable_git_candidates()
for candidate in candidates:
if not candidate.exists():
continue
try:
result = subprocess.run(
[str(candidate), "--version"],
capture_output=True,
text=True, encoding="utf-8", errors="replace",
timeout=15,
)
except Exception:
continue
output = ((result.stdout or "") + (result.stderr or "")).lower()
if "fork bomb" in output:
continue
return candidate
return None
def _ensure_non_trampoline_git(git_cmd: list) -> list:
"""Swap a broken Git-for-Windows trampoline for a real git binary.
Runs right after the git command is built. If ``git`` is a broken
trampoline, rebuild the command around the real binary so fetch/pull/
checkout keep working instead of degrading to the ZIP fallback; if none is
found, leave the command untouched (the fetch-failure handler falls back to
ZIP on Windows). No-op off Windows and when git is healthy.
"""
from hermes_cli.update_cmd import _locate_real_git
if sys.platform != "win32":
return git_cmd
if not _git_is_trampoline(git_cmd):
return git_cmd
real_git = _locate_real_git()
if real_git is None:
print(
"⚠ Detected a broken git trampoline and could not locate a real "
"git binary — the update will fall back to the ZIP path."
)
return git_cmd
print(
f"⚠ Detected a broken git trampoline; switching to real git at "
f"{real_git}"
)
return [str(real_git)] + list(git_cmd[1:])
def _discard_lockfile_churn(git_cmd, repo_root):
"""Restore tracked ``package-lock.json`` files that npm dirtied locally.
npm rewrites lockfiles non-deterministically at install/build time. On a
managed install those diffs are never intentional, so we discard them so
``hermes update`` sees a clean tree instead of autostashing every run.
Best-effort; only ever touches files named ``package-lock.json``.
"""
from hermes_cli.update_cmd import _git_run
try:
diff = _git_run(git_cmd, ["diff", "--name-only"], repo_root)
if diff.returncode != 0:
return
dirty_package_dirs = {
Path(line.strip()).parent
for line in diff.stdout.splitlines()
if line.strip().endswith("package.json")
}
dirty = [
line.strip()
for line in diff.stdout.splitlines()
if line.strip().endswith("package-lock.json")
and Path(line.strip()).parent not in dirty_package_dirs
]
if not dirty:
return
_git_run(git_cmd, ["checkout", "--", *dirty], repo_root)
print(f"→ Discarded npm lockfile churn ({len(dirty)} file(s))")
except Exception:
# Never let lockfile cleanup block an update.
pass
def _normalize_managed_eol(git_cmd, repo_root):
"""Take a managed checkout off ``core.autocrlf=true`` without leaving it dirty.
Git for Windows ships ``core.autocrlf=true`` system-wide, which turns this
repo's LF files CRLF in the working tree and breaks ``git checkout`` on
update ("Your local changes would be overwritten"); ``install.ps1`` pins
``core.autocrlf=false`` on the managed clone (#67730). Older checkouts never
got the pin and the bootstrap installer reuses its build-pinned
``install.ps1`` forever, so ``hermes update`` is the only path that can fix them.
The pin and the cleanup are one operation: under ``autocrlf=true`` a CRLF
tree reads clean, so pinning alone would expose every text file as
modified and hand the update a whole-tree autostash. The pin is written
only after the tree is verified clean under it; a checkout we cannot fully
normalize is left as it was. Best-effort: never blocks an update.
"""
from hermes_cli.update_cmd import _git_run
# -c, not config: evaluate the tree as it WOULD look pinned, without
# persisting anything we might not be able to follow through on.
probe = git_cmd + ["-c", "core.autocrlf=false"]
def _dirty(*extra):
out = subprocess.run(
probe + ["diff", "-z", "--name-only", *extra],
cwd=repo_root,
capture_output=True,
text=True, encoding="utf-8", errors="replace",
)
if out.returncode != 0:
return None
return {p for p in out.stdout.split("\0") if p}
def _real_dirty():
# Files with a *content* change once CRLF differences are ignored.
# ``diff --name-only --ignore-cr-at-eol`` still LISTS CR-only files
# (names come from blob/stat differences before the CR filter), so use
# ``--numstat``, which honors the filter: a CR-only file produces no
# record. Parse the paths out of numstat.
out = subprocess.run(
probe + ["-c", "core.quotepath=false",
"diff", "--numstat", "--ignore-cr-at-eol"],
cwd=repo_root,
capture_output=True,
text=True, encoding="utf-8", errors="replace",
)
if out.returncode != 0:
return None
paths = set()
for line in out.stdout.splitlines():
if not line.strip():
continue
# Format: "<added>\t<deleted>\t<path>". Rename detection is off in
# plain diff, so there is exactly one path field per record.
parts = line.split("\t", 2)
if len(parts) == 3 and parts[2]:
paths.add(parts[2])
return paths
def _eol_only():
all_dirty, real_dirty = _dirty(), _real_dirty()
if all_dirty is None or real_dirty is None:
return None
return all_dirty - real_dirty
try:
effective = _git_run(git_cmd, ["config", "--get", "core.autocrlf"], repo_root)
# Only "true" rewrites LF to CRLF on checkout. Unset, false, and input
# all leave the working tree alone, so there is nothing to repair.
if effective.stdout.strip().lower() != "true":
return
eol_only = _eol_only()
if eol_only is None:
return
if eol_only:
# Pathspec over stdin, not argv: a fully renormalized checkout is
# thousands of paths, well past the Windows command-line limit.
subprocess.run(
probe
+ ["checkout", "--pathspec-from-file=-", "--pathspec-file-nul", "--"],
cwd=repo_root,
input="\0".join(sorted(eol_only)),
capture_output=True,
text=True, encoding="utf-8", errors="replace",
check=False,
)
if _eol_only():
# Still dirty — persisting the pin here would only surface churn
# we failed to clear. Leave the checkout as we found it.
return
print(f"→ Normalized line-ending churn ({len(eol_only)} file(s))")
subprocess.run(
git_cmd + ["config", "core.autocrlf", "false"],
cwd=repo_root,
capture_output=True,
check=False,
)
except Exception:
# Never let line-ending cleanup block an update.
pass
File diff suppressed because it is too large Load Diff
+561
View File
@@ -0,0 +1,561 @@
"""Autostash handling for ``hermes update``: stash local changes before the pull, restore/park/discard them afterwards, warn about orphaned autostashes.
Split out of ``hermes_cli/update_cmd.py``; every moved name is re-imported there, so
``hermes_cli.update_cmd.<name>`` keeps resolving (and monkeypatching) as before.
Origin-internal helpers are imported lazily inside each function (no import cycle;
test patches on ``hermes_cli.update_cmd.<name>`` stay effective).
"""
import logging
import subprocess
from datetime import datetime
from pathlib import Path
from typing import Optional
# Log-record parity with the origin module.
logger = logging.getLogger("hermes_cli.update_cmd")
def _stash_local_changes_if_needed(git_cmd: list[str], cwd: Path) -> Optional[str]:
from hermes_cli.update_cmd import _git_run
status = _git_run(git_cmd, ["status", "--porcelain"], cwd, check=True)
if not status.stdout.strip():
return None
# If the index has unmerged entries (e.g. from an interrupted merge/rebase),
# git stash will fail with "needs merge / could not write index". Clear the
# conflict state with `git reset` so the stash can proceed. Working-tree
# changes are preserved; only the index conflict markers are dropped.
unmerged = _git_run(git_cmd, ["ls-files", "--unmerged"], cwd)
if unmerged.stdout.strip():
print("→ Clearing unmerged index entries from a previous conflict...")
subprocess.run(git_cmd + ["reset"], cwd=cwd, capture_output=True)
from datetime import datetime, timezone
stash_name = datetime.now(timezone.utc).strftime(
f"{_AUTOSTASH_NAME_PREFIX}%Y%m%d-%H%M%S"
)
print("→ Local changes detected — stashing before update...")
prev_stash = _git_run(git_cmd, ["rev-parse", "--verify", "refs/stash"], cwd).stdout.strip()
push = _git_run(git_cmd, ["stash", "push", "--include-untracked", "-m", stash_name], cwd)
if push.stdout.strip():
print(push.stdout.strip())
stash_probe = _git_run(git_cmd, ["rev-parse", "--verify", "refs/stash"], cwd)
stash_ref = stash_probe.stdout.strip()
stash_created = (
stash_probe.returncode == 0 and bool(stash_ref) and stash_ref != prev_stash
)
if push.returncode != 0:
if stash_created:
# stash push exits non-zero when it saved everything but couldn't
# delete some swept untracked files (e.g. a root-owned dir:
# "failed to remove ...: Permission denied"). The entry is
# complete, so not a failure — leave the files and continue.
if push.stderr.strip():
print(push.stderr.strip())
print(
" ⚠ Some untracked files could not be removed from the "
"working tree (permission denied)."
)
print(
" They were still saved to the stash and were left in "
"place — the update will continue."
)
# A partially-failed stash push also aborts its working-tree
# cleanup for TRACKED modifications — they are saved in the stash
# but still dirty the tree, which would break the checkout/pull
# that follows. Safe to reset: everything is in the stash entry.
subprocess.run(
git_cmd + ["reset", "--hard", "HEAD"],
cwd=cwd,
capture_output=True,
)
else:
# No stash entry was created: the changes were NOT saved. This
# is a real failure — bail out before the update touches HEAD.
print("✗ Could not stash local changes — update aborted.")
if push.stderr.strip():
print(f" {push.stderr.strip().splitlines()[0]}")
print(
" Commit, stash, or clean up your local changes manually, "
"then re-run `hermes update`."
)
raise subprocess.CalledProcessError(
push.returncode, push.args, output=push.stdout, stderr=push.stderr
)
return stash_ref
def _resolve_stash_selector(
git_cmd: list[str], cwd: Path, stash_ref: str
) -> Optional[str]:
from hermes_cli.update_cmd import _git_run
stash_list = _git_run(git_cmd, ["stash", "list", "--format=%gd %H"], cwd, check=True)
for line in stash_list.stdout.splitlines():
selector, _, commit = line.partition(" ")
if commit.strip() == stash_ref:
return selector.strip()
return None
#: Producer/consumer contract for update autostash names: the stash subject is
#: this prefix + a UTC YYYYMMDD-HHMMSS stamp (see _stash_local_changes_if_needed
#: and _warn_orphaned_update_autostashes).
_AUTOSTASH_NAME_PREFIX = "hermes-update-autostash-"
#: Age past which a leftover ``hermes-update-autostash-*`` entry is called out
#: at update time. Entries younger than this are normal (a parked stash from
#: the desktop updater's --keep-stash run minutes ago); older ones are almost
#: always forgotten (#63717 problem 6: an orphan persisted 9+ days unnoticed).
_AUTOSTASH_WARN_AGE_DAYS = 7
def _warn_orphaned_update_autostashes(git_cmd: list[str], cwd: Path) -> int:
"""Surface leftover update autostashes older than the warn threshold.
Autostashes legitimately outlive a run (``--keep-stash`` parks them; a
failed restore preserves them), but nothing re-surfaces them — they sit
invisibly for weeks (#63717 problem 6). Prints a notice with recovery/
cleanup guidance. Deliberately NOT a GC: a stash entry can be the only
copy of the user's uncommitted work, so Hermes never drops one.
Best-effort — any git failure returns 0. Returns the stale-entry count.
"""
from hermes_cli.update_cmd import _git_run
from datetime import timedelta, timezone
try:
stash_list = _git_run(git_cmd, ["stash", "list", "--format=%gd %s"], cwd)
if stash_list.returncode != 0:
return 0
cutoff = datetime.now(timezone.utc) - timedelta(
days=_AUTOSTASH_WARN_AGE_DAYS
)
marker = _AUTOSTASH_NAME_PREFIX
stale: list[tuple[str, str]] = []
for line in stash_list.stdout.splitlines():
selector, _, subject = line.strip().partition(" ")
pos = subject.find(marker)
if pos < 0:
continue
stamp = subject[pos + len(marker):][:15] # "YYYYMMDD-HHMMSS"
try:
stash_time = datetime.strptime(stamp, "%Y%m%d-%H%M%S").replace(
tzinfo=timezone.utc
)
except ValueError:
# Unparseable name — age unknown; leave it alone rather than
# guess (same posture as _prune_orphan_rescue_refs).
continue
if stash_time < cutoff:
stale.append((selector, stamp))
if not stale:
return 0
print()
print(
f"⚠ {len(stale)} leftover update autostash entr"
f"{'y is' if len(stale) == 1 else 'ies are'} more than "
f"{_AUTOSTASH_WARN_AGE_DAYS} days old:"
)
for selector, stamp in stale:
print(f" {selector} ({_AUTOSTASH_NAME_PREFIX}{stamp})")
print(" These hold local changes stashed by earlier updates and never")
print(" restored. Review with: git stash show -p <entry>")
print(" Restore with: git stash apply <entry> Discard with: git stash drop <entry>")
return len(stale)
except Exception as exc:
logger.debug("Autostash age check failed: %s", exc)
return 0
def _print_stash_cleanup_guidance(
stash_ref: str, stash_selector: Optional[str] = None
) -> None:
print(
" Check `git status` first so you don't accidentally reapply the same change twice."
)
print(" Find the saved entry with: git stash list --format='%gd %H %s'")
if stash_selector:
print(f" Remove it with: git stash drop {stash_selector}")
else:
print(
f" Look for commit {stash_ref}, then drop its selector with: git stash drop stash@{{N}}"
)
def _stash_apply_failed_only_on_existing_untracked(stderr: str) -> bool:
"""True when a ``git stash apply`` failure is ONLY about untracked files
that already exist in the working tree.
This is the tail end of the permission-denied autostash class: ``git stash
push --include-untracked`` swept undeletable files (e.g. a root-owned
``packaging/`` directory) into the stash but could not remove them from
disk. On restore, git applies all tracked changes, then refuses to
overwrite those still-present files (``already exists, no checkout`` /
``could not restore untracked files from stash``) and exits non-zero even
though nothing was lost. Any other error line (e.g. ``would be
overwritten by merge`` / ``Aborting``) means the tracked apply itself
failed and this returns False.
"""
lines = [ln.strip() for ln in (stderr or "").splitlines() if ln.strip()]
if not lines:
return False
saw_untracked_error = False
for ln in lines:
if "already exists, no checkout" in ln:
saw_untracked_error = True
elif "could not restore untracked files from stash" in ln:
saw_untracked_error = True
elif ln.startswith(("warning:", "hint:")):
continue
else:
return False
return saw_untracked_error
def _park_stashed_changes(stash_ref: str) -> None:
"""Leave a pre-update autostash parked instead of re-applying it.
Used by ``hermes update --keep-stash`` (the desktop updater's mode): the
stash made the update possible on a dirty tree, but local source edits
must never be silently re-applied onto the updated code. Nothing is
lost — the entry stays in ``git stash`` with printed recovery guidance.
"""
print()
print("ℹ️ Local changes were stashed before updating and were NOT re-applied (--keep-stash).")
print(f" Stash ref: {stash_ref}")
print(f" Restore manually with: git stash apply {stash_ref}")
def _git_untracked_paths(git_cmd: list[str], cwd: Path) -> set[str] | None:
"""Return untracked paths, or ``None`` when Git cannot enumerate them."""
try:
result = subprocess.run(
git_cmd + ["ls-files", "--others", "--exclude-standard", "-z"],
cwd=cwd,
capture_output=True,
text=True,
encoding="utf-8",
errors="surrogateescape",
)
except (OSError, subprocess.SubprocessError):
result = None
if result is None or result.returncode != 0:
print(
" ⚠ Could not enumerate untracked files while validating the "
"restored stash."
)
return None
return {path for path in result.stdout.split("\0") if path}
def _restored_python_paths(
git_cmd: list[str], cwd: Path
) -> tuple[str, ...] | None:
"""Return restored ``.py`` paths changed from ``HEAD``.
This deliberately validates Python source only; non-Python entry scripts
remain outside the executable import-health check.
"""
from hermes_cli.update_cmd import _git_untracked_paths
try:
changed = subprocess.run(
git_cmd + ["diff", "--name-only", "-z", "HEAD", "--", "*.py"],
cwd=cwd,
capture_output=True,
text=True,
encoding="utf-8",
errors="surrogateescape",
)
except (OSError, subprocess.SubprocessError):
changed = None
if changed is None or changed.returncode != 0:
print(" ⚠ Could not enumerate tracked Python files restored from the stash.")
return None
paths = set(changed.stdout.split("\0"))
untracked = _git_untracked_paths(git_cmd, cwd)
if untracked is None:
return None
paths.update(path for path in untracked if path.endswith(".py"))
paths.discard("")
return tuple(sorted(paths))
def _reject_unsafe_stash_restore(
git_cmd: list[str],
cwd: Path,
stash_ref: str,
preexisting_untracked: set[str],
failing_target: str,
detail: str | None,
) -> None:
"""Restore the clean updated tree, preserve the stash, and abort the update."""
from hermes_cli.update_cmd import _git_untracked_paths
print()
print("✗ Restored local changes made the Hermes agent unexecutable.")
print(f" Health check failed: {failing_target}")
if detail:
for line in str(detail).splitlines()[:6]:
print(f" {line}")
current_untracked = _git_untracked_paths(git_cmd, cwd)
restored_untracked = (
current_untracked - preexisting_untracked
if current_untracked is not None
else set()
)
try:
reset = subprocess.run(
git_cmd + ["reset", "--hard", "HEAD"], cwd=cwd, capture_output=True
)
except (OSError, subprocess.SubprocessError):
reset = None
clean = None
if restored_untracked:
try:
clean = subprocess.run(
git_cmd + ["clean", "-fd", "--", *sorted(restored_untracked)],
cwd=cwd,
capture_output=True,
)
except (OSError, subprocess.SubprocessError):
clean = None
cleanup_ok = (
current_untracked is not None
and reset is not None
and reset.returncode == 0
and (not restored_untracked or (clean is not None and clean.returncode == 0))
)
if cleanup_ok:
try:
verify = subprocess.run(
git_cmd + ["diff", "--quiet", "HEAD", "--"],
cwd=cwd,
capture_output=True,
)
cleanup_ok = verify.returncode == 0
except (OSError, subprocess.SubprocessError):
cleanup_ok = False
if cleanup_ok:
print(" The clean updated tree has been restored; the gateway was not restarted.")
else:
print(" ⚠ The clean updated tree could not be fully restored automatically.")
print(" Inspect `git status` and run `git reset --hard HEAD` before retrying.")
print(" Platform connectivity alone does not mean the agent can execute turns.")
print(f" Your local changes remain preserved in stash: {stash_ref}")
print(f" Inspect them with: git stash show --stat {stash_ref}")
print(f" Restore manually after fixing them: git stash apply {stash_ref}")
raise SystemExit(1)
def _restore_stashed_changes(
git_cmd: list[str],
cwd: Path,
stash_ref: str,
prompt_user: bool = False,
input_fn=None,
) -> bool:
from hermes_cli.update_cmd import (
_critical_module_import_failures,
_git_run,
_git_untracked_paths,
_restored_python_paths,
_validate_python_files_syntax,
)
if prompt_user:
remote_prompt = input_fn is not None
prompt_suffix = "[y/N]" if remote_prompt else "[Y/n]"
print()
print("⚠ Local changes were stashed before updating.")
print(
" Restoring them may reapply local customizations onto the updated codebase."
)
print(" Review the result afterward if Hermes behaves unexpectedly.")
print(f"Restore local changes now? {prompt_suffix}")
if input_fn is not None:
response = input_fn(f"Restore local changes now? {prompt_suffix}", "n")
else:
try:
response = input().strip().lower()
except (EOFError, UnicodeDecodeError):
# A closed stdin or terminal-encoding error must not crash the
# update mid-restore; fall through to the skip-restore path.
response = "n"
accepted = response in {"y", "yes"} or (not remote_prompt and response == "")
if not accepted:
print("Skipped restoring local changes.")
print("Your changes are still preserved in git stash.")
print(f"Restore manually with: git stash apply {stash_ref}")
return False
preexisting_untracked = _git_untracked_paths(git_cmd, cwd)
if preexisting_untracked is None:
print(" The stash was not restored because its cleanup baseline is unknown.")
print(f" Restore manually with: git stash apply {stash_ref}")
return False
clean_import_failures = _critical_module_import_failures(
cwd, report_runtime_errors=True
)
print("→ Restoring local changes...")
restore = _git_run(git_cmd, ["stash", "apply", stash_ref], cwd)
# Check for unmerged (conflicted) files — can happen even when returncode is 0
unmerged = _git_run(git_cmd, ["diff", "--name-only", "--diff-filter=U"], cwd)
has_conflicts = bool(unmerged.stdout.strip())
if restore.returncode != 0 and not has_conflicts and (
_stash_apply_failed_only_on_existing_untracked(restore.stderr)
):
# Tracked changes applied cleanly; the only "failure" is untracked files
# git couldn't delete at stash time and now refuses to overwrite. Their
# content is untouched — treat as restored.
print(
" ⚠ Some stashed untracked files already exist in the working "
"tree and were kept as-is."
)
elif restore.returncode != 0 or has_conflicts:
print("✗ Update pulled new code, but restoring local changes hit conflicts.")
if restore.stdout.strip():
print(restore.stdout.strip())
if restore.stderr.strip():
print(restore.stderr.strip())
conflicted_files = unmerged.stdout.strip()
if conflicted_files:
print("\nConflicted files:")
for f in conflicted_files.splitlines():
print(f" • {f}")
print("\nYour stashed changes are preserved — nothing is lost.")
print(f" Stash ref: {stash_ref}")
# Always reset: conflict markers in source make hermes unrunnable
# (SyntaxError on import). The user's changes remain in the stash.
subprocess.run(
git_cmd + ["reset", "--hard", "HEAD"],
cwd=cwd,
capture_output=True,
)
print("Working tree reset to clean state.")
print(f"Restore your changes later with: git stash apply {stash_ref}")
# Don't exit: the code update succeeded; let cmd_update continue with
# pip install, skill sync, and gateway restart.
return False
restored_python = _restored_python_paths(git_cmd, cwd)
if restored_python is None:
_reject_unsafe_stash_restore(
git_cmd,
cwd,
stash_ref,
preexisting_untracked,
"restored Python source discovery",
"could not determine which restored Python files require validation",
)
syntax_ok, failing_path, syntax_error = _validate_python_files_syntax(
cwd, restored_python
)
if not syntax_ok:
_reject_unsafe_stash_restore(
git_cmd,
cwd,
stash_ref,
preexisting_untracked,
failing_path or "restored Python source",
syntax_error,
)
restored_import_failures = _critical_module_import_failures(
cwd, report_runtime_errors=True
)
changed_import_failure = next(
(
(module, error)
for module, error in restored_import_failures.items()
if clean_import_failures.get(module) != error
),
None,
)
if changed_import_failure is not None:
failing_module, import_error = changed_import_failure
_reject_unsafe_stash_restore(
git_cmd,
cwd,
stash_ref,
preexisting_untracked,
f"agent import {failing_module or 'unknown'}",
import_error[1],
)
stash_selector = _resolve_stash_selector(git_cmd, cwd, stash_ref)
if stash_selector is None:
print(
"⚠ Local changes were restored, but Hermes couldn't find the stash entry to drop."
)
print(
" The stash was left in place. You can remove it manually after checking the result."
)
_print_stash_cleanup_guidance(stash_ref)
else:
drop = _git_run(git_cmd, ["stash", "drop", stash_selector], cwd)
if drop.returncode != 0:
print(
"⚠ Local changes were restored, but Hermes couldn't drop the saved stash entry."
)
if drop.stdout.strip():
print(drop.stdout.strip())
if drop.stderr.strip():
print(drop.stderr.strip())
print(
" The stash was left in place. You can remove it manually after checking the result."
)
_print_stash_cleanup_guidance(stash_ref, stash_selector)
print("⚠ Local changes were restored on top of the updated codebase.")
print(" Review `git diff` / `git status` if Hermes behaves unexpectedly.")
return True
def _discard_stashed_changes(
git_cmd: list[str],
cwd: Path,
stash_ref: str,
) -> bool:
"""Drop a pre-update stash without applying it.
Only for NON-interactive updates with
``updates.non_interactive_local_changes: discard``. Unlike ``git reset
--hard`` + ``git clean -fd``, this touches only what was stashed — ignored
paths (node_modules, venv, build outputs) are never affected.
Returns True if dropped, False on git failure (stash left in place).
"""
from hermes_cli.update_cmd import _git_run
stash_selector = _resolve_stash_selector(git_cmd, cwd, stash_ref)
if stash_selector is None:
print(
"⚠ Configured to discard local changes on non-interactive update, "
"but Hermes couldn't find the stash entry to drop."
)
_print_stash_cleanup_guidance(stash_ref)
return False
drop = _git_run(git_cmd, ["stash", "drop", stash_selector], cwd)
if drop.returncode != 0:
print(
"⚠ Configured to discard local changes, but Hermes couldn't drop "
"the saved stash entry."
)
if drop.stderr.strip():
print(f" {drop.stderr.strip().splitlines()[0]}")
_print_stash_cleanup_guidance(stash_ref, stash_selector)
return False
print("→ Discarded local source changes (updates.non_interactive_local_changes=discard).")
return True
+566
View File
@@ -0,0 +1,566 @@
"""ZIP-download fallback update path for ``hermes update`` (Windows with broken git): two-phase stage/commit directory swap, dirty-tree guard.
Split out of ``hermes_cli/update_cmd.py``; every moved name is re-imported there, so
``hermes_cli.update_cmd.<name>`` keeps resolving (and monkeypatching) as before.
Origin-internal helpers are imported lazily inside each function (no import cycle;
test patches on ``hermes_cli.update_cmd.<name>`` stay effective).
"""
import logging
import os
import shutil
import subprocess
import sys
from pathlib import Path
from typing import Optional
# Log-record parity with the origin module.
logger = logging.getLogger("hermes_cli.update_cmd")
def _atomic_replace_dir(src: str, dst: str) -> None:
"""Replace directory *dst* with *src* without leaving *dst* half-deleted.
Naive ``rmtree(dst); copytree(src, dst)`` has a destructive window: a
copy that fails partway (common on the Windows ZIP path, which only runs
because file I/O is already flaky) leaves the old tree gone and nothing
in its place (#49145: ``ui-tui/`` vanished and broke the TUI).
Now a thin alias over the two-phase helpers below (#76104); retained as
part of the ``hermes_cli.main`` re-export surface and the #49145 guard.
"""
_commit_staged_replacements([(_stage_replacement(src, dst), dst)])
def _stage_replacement(src: str, dst: str) -> str:
"""Copy *src* to a sibling staging path for *dst*; return the staging path.
Phase 1 of the two-phase replace. Handles both directories and plain
files. Touches nothing live, so a failure here leaves the whole install
untouched.
"""
staging = f"{dst}.hermes-update-staging"
backup = f"{dst}.hermes-update-old"
# A previous run may have died between "move dst aside" and "move staging
# in", leaving the backup as the ONLY copy. Restore it BEFORE clearing
# leftovers: deleting it and then failing to stage (disk exhaustion is
# likely here) would leave a hole with nothing to roll back to.
if not os.path.exists(dst) and os.path.exists(backup):
os.rename(backup, dst)
for leftover in (staging, backup):
if os.path.isdir(leftover):
shutil.rmtree(leftover, ignore_errors=True)
elif os.path.exists(leftover):
os.remove(leftover)
if os.path.isdir(src):
shutil.copytree(src, staging)
else:
shutil.copy2(src, staging)
return staging
def _discard_staged(staged) -> None:
"""Remove staging paths for entries that were never committed.
Otherwise a phase-1 failure (typically disk exhaustion) orphans one
staging copy per processed entry — up to a full second tree — and the
advised "re-run `hermes update`" retry fails harder with less free space.
"""
for staging, _dst in staged:
try:
if os.path.isdir(staging):
shutil.rmtree(staging, ignore_errors=True)
elif os.path.exists(staging):
os.remove(staging)
except OSError as exc: # best-effort cleanup, never fatal
logger.warning("could not remove staging path %s: %s", staging, exc)
def _commit_staged_replacements(staged) -> None:
"""Phase 2: swap every staged entry into place, rolling back all on failure.
``_atomic_replace_dir`` made each *individual* swap safe, but the ZIP
update loops over ~90 top-level entries and nothing made the loop atomic
*as a whole*: a partway failure left a mixed-version tree — every file
valid, the combination unbootable (#76104; also #76091, #63717).
Covers plain files too: the repo root holds 20 first-party modules, so a
files-only failure reproduces the same bug class. Every swap is an
``os.rename`` onto a just-moved-aside path — atomic on POSIX and NTFS —
so a file swap can't leave a half-written module the way ``copy2`` onto
a live path can.
Stage-all-then-swap-all shrinks the failure window from "a full tree
copy" to "N renames" and makes it recoverable: a failed swap restores
every entry already swapped, so the tree lands wholly new or wholly old.
"""
swapped: list[tuple[str, str]] = [] # (dst, backup) in swap order; "" = absent
try:
for staging, dst in staged:
backup = f"{dst}.hermes-update-old"
if os.path.exists(dst):
os.rename(dst, backup)
swapped.append((dst, backup))
else:
swapped.append((dst, ""))
os.rename(staging, dst)
except OSError:
# Undo every swap already made so the install stays self-consistent.
for dst, backup in reversed(swapped):
try:
if os.path.isdir(dst):
shutil.rmtree(dst, ignore_errors=True)
elif os.path.exists(dst):
os.remove(dst)
if backup and os.path.exists(backup):
os.rename(backup, dst)
except OSError as exc:
# Keep restoring the rest — a silent failure here is the one
# thing that turns a recoverable rollback into a mixed tree,
# so say so rather than swallowing it.
logger.warning("rollback failed for %s: %s", dst, exc)
raise
# All swaps succeeded — drop the backups (best-effort, never fatal).
for _dst, backup in swapped:
if backup and os.path.isdir(backup):
shutil.rmtree(backup, ignore_errors=True)
elif backup and os.path.exists(backup):
try:
os.remove(backup)
except OSError:
pass
def _zip_overlay_block_reason(
root: Path, *, ignore_staging_artifacts: bool = False
) -> Optional[str]:
"""Why overlaying a ZIP onto ``root`` would destroy work, or None if safe.
The ZIP path swaps every top-level entry (minus a tiny preserve set) and
deletes the backups, so uncommitted edits and untracked files are gone.
Fails closed when git status cannot run (#87304).
``ignore_staging_artifacts`` is for the pre-swap re-check: phase 1 leaves
``*.hermes-update-staging`` siblings that git reports as untracked; they
are our own artifacts, and without the filter the re-check always refuses.
"""
if not (root / ".git").exists():
return None
git_cmd = ["git"]
if sys.platform == "win32":
git_cmd = ["git", "-c", "windows.appendAtomically=false"]
result = subprocess.run(
# -uall: a user-level ``status.showUntrackedFiles = no`` must not
# blind this guard. --ignored=matching: gitignored files are still
# USER DATA the overlay would delete (#87392); ``matching`` reports an
# ignored dir as one ``dir/`` line (cheaper, same verdict below).
# NOTE: ``--ignored=all`` is NOT a valid git mode — exits 128 and
# would fail-close every ZIP update.
git_cmd + ["status", "--porcelain", "--untracked-files=all", "--ignored=matching"],
cwd=root,
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
)
if result.returncode != 0:
detail = (result.stderr or result.stdout or "").strip().splitlines()
suffix = f" ({detail[0]})" if detail else ""
return f"could not check the working tree{suffix}"
lines = [line for line in (result.stdout or "").splitlines() if line.strip()]
# --ignored=all reports the ZIP path's own preserved entries (venv,
# node_modules are gitignored on every normal install). The swap never
# touches those top-level entries, so they must not turn into a false
# dirty-tree refusal. Everything else — including ignored files — blocks.
lines = [line for line in lines if not _is_zip_preserved_entry_status_line(line)]
if ignore_staging_artifacts:
lines = [
line for line in lines if not _is_zip_staging_artifact_status_line(line)
]
if lines:
return "the working tree has uncommitted changes or untracked files"
return None
_ZIP_STAGING_ARTIFACT_SUFFIXES = (".hermes-update-staging", ".hermes-update-old")
# Single source of truth for the top-level entries the ZIP swap preserves —
# consumed by both the dirty-tree filter below and _update_via_zip's swap loop.
_ZIP_PRESERVED_TOP_LEVEL = {"venv", "node_modules", ".git", ".env"}
def _is_zip_preserved_entry_status_line(line: str) -> bool:
"""True when every path on a porcelain status line sits under a top-level
entry the ZIP swap preserves.
The ``" -> "`` split applies ONLY to rename/copy codes (R/C): porcelain
v1 doesn't quote plain filenames with spaces, so an ignored file named
``venv -> node_modules`` on a ``!!``/``??`` line is ONE path — splitting
would fail-open into the destructive swap. Requiring EVERY path preserved
keeps renames out of a preserved dir (``R venv/x -> src/x``) blocking.
"""
status, payload = (line[:2], line[3:]) if len(line) >= 3 else ("", line)
is_rename = any(code in "RC" for code in status)
paths = payload.split(" -> ") if is_rename else [payload]
for path in paths:
top_level = (
path.strip().strip('"').replace("\\", "/").rstrip("/").split("/", 1)[0]
)
if top_level not in _ZIP_PRESERVED_TOP_LEVEL:
return False
return True
def _is_zip_staging_artifact_status_line(line: str) -> bool:
"""True when a porcelain status line is our own two-phase-swap artifact."""
payload = line[3:] if len(line) >= 3 else line
top_level = (
payload.strip().strip('"').replace("\\", "/").rstrip("/").split("/", 1)[0]
)
return top_level.endswith(_ZIP_STAGING_ARTIFACT_SUFFIXES)
def _abort_zip_update_if_dirty_tree() -> None:
"""Refuse to overlay a ZIP onto a dirty git checkout (#87304)."""
from hermes_cli.update_cmd import _m
reason = _zip_overlay_block_reason(_m().PROJECT_ROOT)
if reason is None:
return
print(f"✗ ZIP fallback refused: {reason}.")
print(
" Overlaying the ZIP would overwrite uncommitted edits and permanently "
"delete untracked files."
)
print(" Stash or commit your changes, then rerun `hermes update`.")
print(" To inspect: git status --porcelain")
_m().sys.exit(1)
def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> bool:
"""Update Hermes Agent by downloading a ZIP archive.
Used on Windows when git file I/O is broken (antivirus, NTFS filter
drivers causing 'Invalid argument' errors on file creation).
Returns ``False`` when a Desktop rebuild ran and failed; ``True`` otherwise.
"""
from hermes_cli.update_cmd import (
_ensure_uv_for_termux,
_ensure_venv_pip,
_finish_dashboard_update_cleanup,
_m,
_print_bundled_skills_sync_report,
_print_curator_first_run_notice,
_print_curator_recent_run_notice,
_print_update_summary,
_read_project_version,
_rebuild_desktop_after_update,
_refuse_update_for_contended_shims,
_shim_quarantine_error_type,
_sweep_bytecode_after_update,
_update_node_dependencies,
_validate_critical_modules_import,
_verify_and_restore_state_dbs_post_update,
)
active_tool_dependencies = _m()._capture_active_tool_dependencies()
import tempfile
import zipfile
from urllib.request import urlretrieve
# Snapshot the pre-update version before files are replaced so the
# completion line can report the transition (prime-agent#630 port).
pre_update_version = _read_project_version()
# The static GitHub archive is fine for "main" but would silently ignore
# --branch — the exact silent-divergence bug --branch was added to
# prevent. Refuse rather than lie.
branch = _m()._resolve_update_branch(args)
if branch != "main":
print(
f"✗ --branch={branch} is not supported on the Windows ZIP-fallback "
"update path."
)
print(
" This path runs when git file I/O is broken on the system. "
"Either resolve the git-side breakage (typically an antivirus "
"or NTFS filter holding files open) and rerun `hermes update "
f"--branch {branch}`, or update against main with `hermes update`."
)
_m().sys.exit(1)
_abort_zip_update_if_dirty_tree()
zip_url = (
f"https://github.com/NousResearch/hermes-agent/archive/refs/heads/{branch}.zip"
)
print("→ Downloading latest version...")
tmp_dir = tempfile.mkdtemp(prefix="hermes-update-")
try:
zip_path = os.path.join(tmp_dir, f"hermes-agent-{branch}.zip")
urlretrieve(zip_url, zip_path)
print("→ Extracting...")
import stat as _stat
with zipfile.ZipFile(zip_path, "r") as zf:
# Reject zip-slip (path traversal) AND symlink members: a
# hermes-agent source ZIP never legitimately contains symlinks,
# and a compromised mirror could use them to plant files anywhere.
tmp_dir_real = os.path.realpath(tmp_dir)
for member in zf.infolist():
member_path = os.path.realpath(os.path.join(tmp_dir, member.filename))
if (
not member_path.startswith(tmp_dir_real + os.sep)
and member_path != tmp_dir_real
):
raise ValueError(
f"Zip-slip detected: {member.filename} escapes extraction directory"
)
# Unix mode lives in the upper 16 bits of external_attr;
# mask to the file-type bits.
mode = (member.external_attr >> 16) & 0o170000
if _stat.S_ISLNK(mode):
raise ValueError(
f"ZIP contains unsupported symlink member: {member.filename}"
)
zf.extractall(tmp_dir)
# GitHub ZIPs extract to hermes-agent-<branch>/
extracted = os.path.join(tmp_dir, f"hermes-agent-{branch}")
if not os.path.isdir(extracted):
for d in os.listdir(tmp_dir):
candidate = os.path.join(tmp_dir, d)
if os.path.isdir(candidate) and d != "__MACOSX":
extracted = candidate
break
preserve = _ZIP_PRESERVED_TOP_LEVEL
entries = [i for i in os.listdir(extracted) if i not in preserve]
# Two-phase replace (#76104): phase 1 stages every entry (dirs AND
# top-level files — the repo root holds 20 first-party modules) beside
# its target; phase 2 swaps all in with same-filesystem renames and
# rolls back on any failure. One-at-a-time replacement left `agent/`
# new and `tools/` stale on interruption: all files valid, tree
# unbootable. Staging costs one extra tree copy — check space up front.
need = sum(
os.path.getsize(os.path.join(dirpath, f))
for entry in entries
for dirpath, _dirs, files in os.walk(os.path.join(extracted, entry))
for f in files
) + sum(
os.path.getsize(os.path.join(extracted, e))
for e in entries
if os.path.isfile(os.path.join(extracted, e))
)
# Swaps are renames, so only the staging copy is new: require it plus
# 20% headroom, not 2x — which would block updates on exactly the
# space-constrained machines most likely to hit this path.
required = int(need * 1.2)
free = shutil.disk_usage(str(_m().PROJECT_ROOT)).free
if free < required:
raise RuntimeError(
f"not enough free disk space to stage the update safely "
f"(need ~{required // (1024 * 1024)} MB, have "
f"{free // (1024 * 1024)} MB)"
)
staged: list[tuple[str, str]] = []
try:
for item in entries:
src = os.path.join(extracted, item)
dst = os.path.join(str(_m().PROJECT_ROOT), item)
staged.append((_stage_replacement(src, dst), dst))
# #70337/#87331: the source ZIP lacks apps/desktop/release/
# (the BUILT desktop app); swapping `apps` without it deletes
# the build and breaks the shortcut. Graft the live release
# dir into the staged copy BEFORE the swap.
if item == "apps":
live_release = os.path.join(dst, "desktop", "release")
staged_release = os.path.join(
staged[-1][0], "desktop", "release"
)
if os.path.isdir(live_release) and not os.path.exists(
staged_release
):
os.makedirs(os.path.dirname(staged_release), exist_ok=True)
shutil.copytree(live_release, staged_release)
except Exception:
# Nothing is live yet; drop the partial staging copies so a retry
# starts from the same free space this attempt did.
_discard_staged(staged)
raise
try:
# Re-check right before the swap (#87304 TOCTOU): download +
# extract + staging can take minutes, and work created meanwhile
# would be destroyed. Our own staging siblings are filtered out.
recheck_reason = _zip_overlay_block_reason(
_m().PROJECT_ROOT, ignore_staging_artifacts=True
)
if recheck_reason is not None:
_discard_staged(staged)
print(f"✗ ZIP fallback aborted before the swap: {recheck_reason}.")
print(
" Files appeared in the checkout while the update was "
"downloading; committing the swap would delete them."
)
print(" Stash or commit your changes, then rerun `hermes update`.")
_m().sys.exit(1)
_commit_staged_replacements(staged)
except Exception:
# Rollback restored the swapped entries, but staging copies for
# the rest (possibly most of a tree) remain. Drop them, or the
# retry's up-front free-space check (which runs BEFORE per-entry
# leftover cleanup) fails on our litter. Safe post-rollback:
# _discard_staged skips paths that no longer exist.
_discard_staged(staged)
raise
update_count = len(staged)
print(f"✓ Updated {update_count} items from ZIP")
except Exception as e:
print(f"✗ ZIP update failed: {e}")
# The two-phase replace either commits every entry or rolls them all
# back, so a failure here does not leave a mixed-version tree — don't
# scare the user toward a reinstall they don't need.
print(" Your existing install was left in place.")
print(
" Re-run `hermes update` to retry; if the agent won't start, "
"reinstall from https://hermes-agent.nousresearch.com"
)
_m().sys.exit(1)
finally:
shutil.rmtree(tmp_dir, ignore_errors=True)
_sweep_bytecode_after_update(branch)
# Reinstall Python deps: prefer .[all]; if one extra breaks, keep base
# deps and retry the remaining extras individually so working
# capabilities aren't silently stripped. Self-lock deferral (#86735): the
# code swap is committed; defer only the dependency sync when this
# process holds a native extension the sync must rewrite.
_m()._abort_dependency_sync_if_self_locked()
print("→ Updating Python dependencies...")
from hermes_cli.managed_uv import ensure_uv, update_managed_uv
# Keep managed uv current — runs `uv self update` if we already have one.
update_managed_uv()
uv_bin = ensure_uv()
pip_cmd = [_m().sys.executable, "-m", "pip"]
if not uv_bin:
uv_bin = _ensure_uv_for_termux(pip_cmd)
if uv_bin:
# Same third-party UV-env isolation as the main update path (#83914):
# a user-level UV_PYTHON_INSTALL_DIR / UV_PYTHON from unrelated
# software must not steer which interpreter uv resolves here.
from hermes_cli.managed_uv import managed_python_env
uv_env = managed_python_env()
uv_env["VIRTUAL_ENV"] = str(_m().PROJECT_ROOT / "venv")
if _m()._is_termux_env(uv_env):
uv_env.pop("PYTHONPATH", None)
uv_env.pop("PYTHONHOME", None)
try:
_m()._install_python_dependencies_with_optional_fallback([uv_bin, "pip"], env=uv_env)
except _shim_quarantine_error_type() as _sqe:
# #87331: this runs inside the ZIP-fallback error handler, so the
# boundary except clause in cmd_update cannot catch it — refuse
# here with the same defer-via-marker contract.
_refuse_update_for_contended_shims(_sqe)
else:
# sys.executable -m pip avoids PEP 668 'externally-managed-environment' errors.
_ensure_venv_pip(pip_cmd, _m().sys.executable)
_m()._install_python_dependencies_with_optional_fallback(pip_cmd)
install_prefix = [uv_bin, "pip"] if uv_bin else pip_cmd
install_env = uv_env if uv_bin else None
_m()._restore_active_tool_dependencies(
active_tool_dependencies,
install_prefix,
env=install_env,
)
# ZIP path parity: heal the active memory provider's bridge packages
# after the dependency reinstall, same as the git-pull path (#53272,
# #70636).
_m()._refresh_active_memory_provider_dependencies()
# Verify the tree actually imports (catches the parse-OK-but-skewed tree
# an interrupted copy leaves). Placed *after* the dependency reinstall so
# a genuinely-new third-party requirement isn't misreported as a partial
# copy. No SHA to roll back to here — surface a concrete recovery step
# instead of reporting success over a bricked install.
import_ok, failing_module, import_error = _validate_critical_modules_import(
_m().PROJECT_ROOT
)
if not import_ok:
print()
print("✗ Update left the install in an unimportable state:")
print(f" {failing_module}: {import_error}")
print()
print(" This usually means the copy was interrupted partway through.")
print(" Re-run `hermes update` to complete it.")
_m().sys.exit(1)
node_failures = _update_node_dependencies()
_m()._build_web_ui(_m().PROJECT_ROOT / "web")
desktop_build_ok = _rebuild_desktop_after_update(
_m().PROJECT_ROOT / "apps" / "desktop",
had_desktop_app_before_update=had_desktop_app_before_update,
)
try:
print("→ Syncing bundled skills...")
_print_bundled_skills_sync_report()
except Exception:
pass
# Seed the model-catalog disk cache from the freshly-unpacked checkout
# (same rationale as the git-pull path in _cmd_update_impl). Non-fatal.
try:
from hermes_cli.model_catalog import seed_cache_from_checkout
if seed_cache_from_checkout(_m().PROJECT_ROOT):
print(" ✓ Model catalog cache refreshed from checkout")
except Exception as e:
logger.debug("Model catalog seed during zip update failed: %s", e)
# Post-update state.db integrity guard (#68474, #97994): root home AND
# every sibling profile, each auto-restored from its own snapshot.
try:
_verify_and_restore_state_dbs_post_update()
except Exception as exc:
logger.debug(
"Post-update state.db integrity check (zip path) failed: %s", exc
)
update_complete = _print_update_summary(
node_failures=node_failures,
desktop_build_ok=desktop_build_ok,
pre_update_version=pre_update_version,
)
try:
_print_curator_first_run_notice()
except Exception as e:
logger.debug("Curator first-run notice failed: %s", e)
try:
_print_curator_recent_run_notice()
except Exception as e:
logger.debug("Curator recent-run notice failed: %s", e)
# Don't stop a working dashboard when the Node refresh failed — see the
# git-update path for rationale (#30271).
_finish_dashboard_update_cleanup(node_failures)
try:
from hermes_cli.update_receipt import finalize_update_receipt
finalize_update_receipt(
"success" if update_complete and not node_failures else "partial"
)
except Exception as _receipt_exc:
logger.debug("Update receipt finalize (zip path) failed: %s", _receipt_exc)
return update_complete