diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 42e2bced56..c3167cfb05 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -1564,8 +1564,9 @@ export const ar = defineLocale({ cron: { close: 'إغلاق', modelImpact: { - title: 'تحتاج المهام المجدولة إلى المراجعة', - message: count => `سيتم تخطي ${count} من المهام المجدولة حتى تراجع إعدادات النموذج الخاصة بها.`, + title: 'تبقى المهام المجدولة على نموذجها الأصلي', + message: count => + `${count} من المهام المجدولة غير المثبتة ستواصل العمل على النموذج الذي أُنشئت به. ثبّتها أو اضبط cron.model لنقلها.`, detailMore: (names, remaining) => `${names} و${remaining} أخرى`, review: 'مراجعة المهام المجدولة', saveFailed: 'لم يحفظ Hermes تغيير النموذج هذا.', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 0053466ece..fe3ef6fb4d 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -2174,9 +2174,9 @@ export const en: Translations = { title: 'Scheduled jobs', count: count => `${count} ${count === 1 ? 'job' : 'jobs'}`, modelImpact: { - title: 'Scheduled jobs need review', + title: 'Scheduled jobs stay on their original model', message: count => - `${count} scheduled ${count === 1 ? 'job' : 'jobs'} will be skipped until you review their model settings.`, + `${count} unpinned scheduled ${count === 1 ? 'job keeps' : 'jobs keep'} running on the model ${count === 1 ? 'it was' : 'they were'} created under. Pin ${count === 1 ? 'it' : 'them'} or set cron.model to move ${count === 1 ? 'it' : 'them'}.`, detailMore: (names, remaining) => `${names} and ${remaining} more`, review: 'Review scheduled jobs', saveFailed: 'Hermes did not save that model change.', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 8acb78e25e..783824bd53 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -1846,8 +1846,9 @@ export const ja = defineLocale({ title: 'スケジュール済みジョブ', count: count => `${count} 件のジョブ`, modelImpact: { - title: 'スケジュール済みジョブの確認が必要です', - message: count => `モデル設定を確認するまで、${count} 件のスケジュール済みジョブがスキップされます。`, + title: 'スケジュール済みジョブは元のモデルで実行されます', + message: count => + `ピン留めされていない ${count} 件のスケジュール済みジョブは、作成時のモデルで引き続き実行されます。移行するにはピン留めするか cron.model を設定してください。`, detailMore: (names, remaining) => `${names}、ほか ${remaining} 件`, review: 'スケジュール済みジョブを確認', saveFailed: 'Hermes はモデルの変更を保存しませんでした。', diff --git a/apps/desktop/src/i18n/ru.ts b/apps/desktop/src/i18n/ru.ts index 7a6ee0c3cd..54876c3e15 100644 --- a/apps/desktop/src/i18n/ru.ts +++ b/apps/desktop/src/i18n/ru.ts @@ -2212,9 +2212,9 @@ export const ru = defineLocale({ title: 'Запланированные задачи', count: count => `${count} ${RU_PLURAL(count, 'задача', 'задачи', 'задач')}`, modelImpact: { - title: 'Запланированные задачи требуют проверки', + title: 'Запланированные задачи остаются на исходной модели', message: count => - `Будет пропущено ${count} ${RU_NOUN(count, 'задача', 'задачи', 'задач')} до тех пор, пока вы не проверите их настройки модели.`, + `${count} незакреплённых запланированных задач продолжат работать на модели, с которой были созданы. Закрепите их или задайте cron.model, чтобы перевести.`, detailMore: (names, remaining) => `${names} и ещё ${remaining}`, review: 'Проверить запланированные задачи', saveFailed: 'Hermes не сохранил это изменение модели.' diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 04017abe19..a66f37159c 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -1773,8 +1773,9 @@ export const zhHant = defineLocale({ title: '排程工作', count: count => `${count} 個工作`, modelImpact: { - title: '排程工作需要檢查', - message: count => `在您檢查模型設定之前,${count} 個排程工作將被略過。`, + title: '排程工作將繼續使用原模型', + message: count => + `${count} 個未固定的排程工作將繼續使用建立時的模型執行。固定它們或設定 cron.model 以遷移。`, detailMore: (names, remaining) => `${names},以及另外 ${remaining} 個`, review: '檢查排程工作', saveFailed: 'Hermes 未儲存該模型變更。', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index a20d7a7ba6..7fb988775c 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -2340,8 +2340,9 @@ export const zh: Translations = { title: '定时任务', count: count => `${count} 个任务`, modelImpact: { - title: '定时任务需要检查', - message: count => `在您检查模型设置之前,${count} 个定时任务将被跳过。`, + title: '定时任务将继续使用原模型', + message: count => + `${count} 个未固定的定时任务将继续使用创建时的模型运行。固定它们或设置 cron.model 以迁移。`, detailMore: (names, remaining) => `${names},以及另外 ${remaining} 个`, review: '检查定时任务', saveFailed: 'Hermes 未保存该模型更改。', diff --git a/apps/desktop/src/store/cron-model-impact.ts b/apps/desktop/src/store/cron-model-impact.ts index 767583bfde..8dc38435b0 100644 --- a/apps/desktop/src/store/cron-model-impact.ts +++ b/apps/desktop/src/store/cron-model-impact.ts @@ -74,7 +74,6 @@ export function parseCronModelImpact(value: unknown): CronModelImpact | null { if ( typeof impact.available !== 'boolean' || - typeof impact.guard_enabled !== 'boolean' || !Number.isSafeInteger(impact.affected_count) || (impact.affected_count ?? -1) < 0 || typeof impact.truncated !== 'boolean' || @@ -121,15 +120,17 @@ function publishImpact(impact: CronModelImpact, profile: string, connection: str return } - if (!impact.guard_enabled || impact.affected_count === 0) { + if (impact.affected_count === 0) { dismissNotification(CRON_MODEL_IMPACT_NOTIFICATION_ID) return } + // Informational: these jobs keep running on the model they were created under; nothing is + // skipped. The action is a read-only review so the user can pin or move them deliberately. notify({ id: CRON_MODEL_IMPACT_NOTIFICATION_ID, - kind: 'warning', + kind: 'info', title: translateNow('cron.modelImpact.title'), message: translateNow('cron.modelImpact.message', impact.affected_count), detail: detailFor(impact), diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index 1714cbfd13..8d5cd0570d 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -1494,7 +1494,6 @@ export interface CronModelImpactJob { export interface CronModelImpact { available: boolean - guard_enabled: boolean affected_count: number truncated: boolean jobs: CronModelImpactJob[] diff --git a/cli.py b/cli.py index feadecd07d..df7f7464bd 100644 --- a/cli.py +++ b/cli.py @@ -2408,7 +2408,7 @@ def save_config_value(key_path: str, value: any) -> bool: os.chmod(config_path, 0o600) except (OSError, NotImplementedError): pass - # Same fail-closed cron drift warning as `hermes config set` for every model switch. + # Same unpinned-cron notice as `hermes config set` for every model switch. from hermes_cli.config import warn_unpinned_cron_jobs_after_model_config_change warn_unpinned_cron_jobs_after_model_config_change(key_path, value) diff --git a/cron/jobs.py b/cron/jobs.py index 903021c38c..ebd51bca41 100644 --- a/cron/jobs.py +++ b/cron/jobs.py @@ -1494,7 +1494,7 @@ def _normalize_workdir(workdir: Optional[str]) -> Optional[str]: def _resolve_default_model_snapshot() -> Optional[str]: """Default model resolved as the ticker's ``run_job`` does, so unpinned jobs can snapshot it and - detect a later swap. ``None`` on missing config or failure (fail-open: "no snapshot").""" + keep running on it after a later swap. ``None`` on missing config or failure ("no snapshot").""" try: from hermes_cli.config import _expand_env_vars, read_user_config_raw @@ -1599,8 +1599,9 @@ _UPDATE_FIELD_NORMALIZERS: Dict[str, Callable[[Any], Any]] = { def _compute_provider_model_snapshots( *, provider: Any, model: Any, base_url: Any, no_agent: Any, ) -> Tuple[Optional[str], Optional[str]]: - """Snapshot unpinned provider/model resolution so a later global switch fails closed at fire - time instead of silently changing spend. Pinned axes and no-agent jobs carry no snapshot.""" + """Snapshot unpinned provider/model resolution: the scheduler runs the job on this snapshot after + a later global switch instead of silently changing spend. Pinned axes and no-agent jobs carry no + snapshot.""" normalized_provider = _normalize_job_optional_text(provider) normalized_model = _normalize_job_optional_text(model) normalized_base_url = _normalize_base_url(base_url) @@ -2108,13 +2109,11 @@ def remove_job(job_id: str) -> bool: def _set_alert_flag(job_id: str, field: str, value: bool) -> bool: """Set/clear a persisted alert-dedup marker (alert exactly once until the condition heals; - survives restarts) and return the PRIOR value. Fields: ``preflight_alerted``, - ``drift_alerted``. + survives restarts) and return the PRIOR value. Field: ``preflight_alerted`` (blocked config). The marker records that the operator was already alerted about this job's condition, so the scheduler alerts exactly once and stays silent on subsequent ticks until the condition heals (same alert-once - shape as the dead-pin auto-pause in #73506). Fields: ``preflight_alerted`` (blocked config, T1-26) and - ``drift_alerted`` (#44585 drift-guard skip). + shape as the dead-pin auto-pause in #73506). """ def apply(jobs, _i, job): prior = bool(job.get(field)) @@ -2139,11 +2138,6 @@ def clear_preflight_alerted(job_id: str) -> None: _set_alert_flag(job_id, "preflight_alerted", False) -def mark_drift_alerted(job_id: str) -> bool: - """Mark the job as drift-alerted; return True if it already was.""" - return _set_alert_flag(job_id, "drift_alerted", True) - - def note_fire_forward_failure(job_id: str, detail: str) -> bool: """Durably record (as ``last_fire_error``) that a scheduled fire could not be handed to the runner — written by the dashboard fire webhook when the loopback forward fails. Without it @@ -2175,7 +2169,6 @@ def _record_run_outcome( # Healthy run: drop the alert-once dedup markers so a FUTURE break re-alerts, and clear # the forward-failure stamp so it only describes CURRENT auto-fire health. job.pop("preflight_alerted", None) - job.pop("drift_alerted", None) job.pop("last_fire_error", None) job["failure_streak"] = 0 else: diff --git a/cron/scheduler.py b/cron/scheduler.py index 2126f72887..6d758c42dd 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -38,8 +38,7 @@ sys.path.insert(0, str(Path(__file__).parent.parent)) from hermes_constants import get_hermes_home from hermes_cli._subprocess_compat import windows_hide_flags from hermes_cli.config import ( - _expand_env_vars, cron_model_drift_axes, cron_model_drift_guard_enabled, load_config, - resolve_cron_model_drift_defaults) + _expand_env_vars, load_config, resolve_cron_model_drift_defaults) from hermes_cli.fallback_config import get_fallback_chain from hermes_time import now as _hermes_now from agent.interrupt_compat import request_hard_interrupt @@ -220,24 +219,6 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str: text = (error or "unknown error").strip() lower = text.lower() - if "skipped to prevent unintended spend: global inference config drifted" in lower: - if "finite one-shot job is consumed" in lower: - remediation = ( - "This finite one-shot is consumed; create a new one-shot job at " - "a future time with an explicit provider and model." - ) - else: - job_id = job.get("id") or "" - remediation = ( - "On the host running Hermes, pin it explicitly: " - f"`hermes cron edit {job_id} --provider " - "--model `." - ) - return ( - f"⚠️ Cron '{job_name}' skipped before inference to prevent " - f"unintended spend. {remediation}" - ) - # no_agent jobs never reach a model, so provider errors are structurally impossible for them. # Gate on job MODE before substring matching, or a script's own wording ("timed out", "429") # would blame the wrong subsystem; the generic cleaner below reports what actually happened. @@ -1354,10 +1335,26 @@ class _CronJobConfig: cron_default_provider: str +def _snapshot_pin(job: dict, axis: str, current: str, job_id: str) -> str: + """The creation snapshot is an unpinned axis's effective pin: return it, logging once when it + differs from *current* (the live global default); ``""`` for legacy jobs without one, which keep + following the global default. A global model/provider change must never stop a cron job; a job + keeps running on what it was created under until the operator pins it or sets a cron.* fleet + default (#44585).""" + snapshot = str(job.get(f"{axis}_snapshot") or "").strip() + if snapshot and current and snapshot.lower() != current.lower(): + logger.info( + "Job '%s': running on creation-snapshot %s %r (global default is now %r); " + "`hermes cron edit %s --%s ` or cron.%s in config.yaml moves it.", + job_id, axis, snapshot, current, job_id, axis, + "model" if axis == "model" else "model_provider") + return snapshot + + def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConfig: """Load config.yaml and resolve the run's model: per-job override > cron.model (fleet default) > - HERMES_MODEL > config ``model:``. Re-read every tick (no cache) so ``hermes cron edit --model`` - applies next tick. An axis resolved from cron.model/model_provider is explicit (no drift guard).""" + creation snapshot > HERMES_MODEL > config ``model:``. Re-read every tick (no cache) so + ``hermes cron edit --model`` applies next tick.""" model = job.get("model") or os.getenv("HERMES_MODEL") or "" _cron_default_provider = "" _cfg: dict = {} @@ -1383,10 +1380,8 @@ def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConf if _cron_default_model: model = _cron_default_model else: - # Shared with Desktop's impact summary so both compare against the same model. _, _global_model = resolve_cron_model_drift_defaults(_cfg) - if _global_model: - model = _global_model + model = _snapshot_pin(job, "model", _global_model, job_id) or _global_model or model except Exception as e: logger.warning("Job '%s': failed to load config.yaml, using defaults: %s", job_id, e) @@ -1489,42 +1484,34 @@ def _preflight_or_block(job: dict, job_id: str, job_name: str, cfg: dict) -> Opt return False, blocked_doc, "", f"{marker} {_pf_reason}" -def _resolve_job_runtime( - job: dict, job_id: str, jc: _CronJobConfig, -) -> tuple[dict, str, Optional[str]]: +def _resolve_job_runtime(job: dict, job_id: str, jc: _CronJobConfig) -> tuple[dict, str]: """Resolve the runtime, walking the fallback chain on auth/transient-network errors. Returns - ``(runtime, model, primary_provider_for_drift)``; provider+model swap atomically (never swap - only the provider while keeping a paid primary model).""" + ``(runtime, model)``; provider+model swap atomically (never swap only the provider while keeping + a paid primary model). Provider precedence: per-job pin > cron.model_provider > creation + snapshot > persisted global config.""" from hermes_cli.runtime_provider import ( resolve_runtime_provider, format_runtime_provider_error) from hermes_cli.auth import AuthError model = jc.model - configured_provider_for_drift = ( - str(jc.model_cfg.get("provider") or "").strip().lower() - if isinstance(jc.model_cfg, dict) - else "" - ) - primary_provider_for_drift = ( - str(job.get("provider") or "").strip().lower() - or configured_provider_for_drift - or None - ) + requested = job.get("provider") or jc.cron_default_provider or None + if not requested: + global_provider = ( + str(jc.model_cfg.get("provider") or "").strip() if isinstance(jc.model_cfg, dict) else "") + # None (not the config provider) keeps the legacy no-snapshot path resolving from persisted + # config exactly as before. + requested = _snapshot_pin(job, "provider", global_provider, job_id) or None try: # Do NOT pass HERMES_INFERENCE_PROVIDER as `requested`: it would override persisted config # and resurrect stale providers for unpinned jobs. runtime_kwargs = { - "requested": job.get("provider") or jc.cron_default_provider or None, + "requested": requested, # api_mode must derive from the model actually run, not the stale persisted default. "target_model": model, } if job.get("base_url"): runtime_kwargs["explicit_base_url"] = job.get("base_url") - runtime = resolve_runtime_provider(**runtime_kwargs) - primary_provider_for_drift = ( - str(runtime.get("provider") or "").strip().lower() or primary_provider_for_drift - ) - return runtime, model, primary_provider_for_drift + return resolve_runtime_provider(**runtime_kwargs), model except Exception as resolve_exc: # Walk the fallback chain on AuthError AND transient network/DNS failures (e.g. during # OAuth refresh); anything else re-raises. @@ -1533,10 +1520,6 @@ def _resolve_job_runtime( if not (is_auth or is_transient_net): raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc - primary_provider_for_drift = ( - str(getattr(resolve_exc, "provider", "") or "").strip().lower() - or primary_provider_for_drift - ) logger.warning( "Job '%s': primary provider resolve failed (%s: %s), trying fallback", job_id, "auth" if is_auth else "transient network", resolve_exc) @@ -1560,82 +1543,12 @@ def _resolve_job_runtime( logger.info( "Job '%s': fallback resolved to %s model %s", job_id, runtime.get("provider"), fb_model) - return runtime, fb_model, primary_provider_for_drift + return runtime, fb_model except Exception as fb_exc: logger.debug("Job '%s': fallback %s failed: %s", job_id, fb_provider, fb_exc) raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc -def _check_model_drift( - job: dict, job_id: str, cfg: dict, runtime: dict, - primary_provider_for_drift: Optional[str], primary_model_for_drift: str, -) -> None: - """Fail-closed provider/model drift guard; raises RuntimeError (with drift marker) on drift. - An unpinned job follows the global default, which may have switched to a paid provider/model: - each unpinned axis whose creation snapshot (job["_snapshot"]) now resolves differently - skips the run and alerts to pin. No snapshot, pinned axes, or the cron.model fleet default - never count as drift.""" - if not cron_model_drift_guard_enabled(cfg): - return - _current_provider = str( - primary_provider_for_drift or runtime.get("provider") or "" - ).strip().lower() - _current_model = str(primary_model_for_drift or "").strip().lower() - _drift: list[str] = [] - for _axis in cron_model_drift_axes( - job, current_provider=_current_provider, current_model=_current_model, config=cfg): - _snapshot = str(job.get(f"{_axis}_snapshot") or "").strip().lower() - _current = _current_provider if _axis == "provider" else _current_model - _drift.append(f"{_axis} '{_snapshot}' -> '{_current}'") - if not _drift: - return - _changes = "; ".join(_drift) - # A finite one-shot is consumed by this attempt, so "edit the job" is a dead end for it. - # Lifecycle-aware remediation (#72056, @sashmatash): a finite one-shot is consumed by this attempted - # dispatch — telling an operator to edit a spent job is a dead end. Recurring and repeatable jobs get - # the pin command instead. - _repeat = job.get("repeat") if isinstance(job.get("repeat"), dict) else {} - _finite_oneshot = ( - isinstance(job.get("schedule"), dict) - and job["schedule"].get("kind") == "once" - and _repeat.get("times") == 1 - ) - if _finite_oneshot: - _remediation = ( - "This finite one-shot job is consumed by this attempted run; " - "create a new one-shot job at a future time with an explicit provider and model." - ) - else: - _remediation = ( - "To run on the new config, on the host running Hermes pin it explicitly: " - f"`hermes cron edit {job_id} --provider " - "--model ` (or pin the original values to keep them)." - ) - logger.warning( - "Job '%s': SKIPPED — global inference config drifted since " - "creation (%s) and this job is unpinned. Skipped to prevent unintended spend. %s", - job_id, _changes, _remediation) - # Alert-once via drift_alerted bit (silent marker suppresses delivery); a successful run - # clears it and re-arms the alert. - # Alert-once (#73506 shape): persist the drift_alerted bit so only the FIRST drifted tick delivers; - # run_one_job suppresses delivery on the silent marker. mark_job_run clears the bit when a run succeeds - # (drift healed), re-arming the alert. - _drift_already_alerted = False - with contextlib.suppress(Exception): - from cron.jobs import mark_drift_alerted - - _drift_already_alerted = mark_drift_alerted(job_id) - _drift_marker = DRIFT_SKIP_SILENT_MARKER if _drift_already_alerted else DRIFT_SKIP_MARKER - raise RuntimeError( - f"{_drift_marker} Skipped to prevent unintended spend: global " - f"inference config drifted since this job was created " - f"({_changes}), and this job is unpinned. No inference call " - f"was made. {_remediation} " - f"This alert is sent once; the job stays skipped until the " - f"config is pinned or restored. See #44585." - ) - - def _load_credential_pool(runtime: dict, job_id: str): runtime_provider = str(runtime.get("provider") or "").strip().lower() if not runtime_provider: @@ -2203,7 +2116,7 @@ class _CronAgentSetup: def _resolve_cron_agent_setup(job: dict, job_id: str, job_name: str, jc) -> _CronAgentSetup: """Resolve model/runtime/reasoning/pool for the run, in the original gate order: exfil guard -> - preflight (may block) -> runtime -> drift check -> fallback chain -> credential pool -> MCP.""" + preflight (may block) -> runtime (+ fallback chain) -> credential pool -> MCP.""" _cfg = jc.cfg setup = _CronAgentSetup(model=jc.model) setup.prefill_messages = _load_prefill_messages(_cfg, job_id) @@ -2223,13 +2136,10 @@ def _resolve_cron_agent_setup(job: dict, job_id: str, job_name: str, jc) -> _Cro if setup.blocked is not None: return setup - primary_model_for_drift = setup.model - setup.runtime, setup.model, primary_provider_for_drift = _resolve_job_runtime(job, job_id, jc) + setup.runtime, setup.model = _resolve_job_runtime(job, job_id, jc) setup.reasoning_config = _resolve_job_reasoning_config( job, _cfg if isinstance(_cfg, dict) else {}, str(setup.model) ) - _check_model_drift( - job, job_id, _cfg, setup.runtime, primary_provider_for_drift, primary_model_for_drift) setup.fallback_model = get_fallback_chain(_cfg) or None setup.credential_pool = _load_credential_pool(setup.runtime, job_id) # MCP servers must be registered before AIAgent is constructed. @@ -2622,11 +2532,9 @@ def _compose_run_delivery( silent_alert, incident_acked, failure_incident_id)``; ``silent_alert``: an alert-once marker says the operator was already told, deliver nothing.""" err = str(error) if error else "" - # Failed jobs always deliver, except blocked-config / drift-skip runs, which alert exactly ONCE. + # Failed jobs always deliver, except blocked-config runs, which alert exactly ONCE. blocked_config_silent = BLOCKED_CONFIG_SILENT_MARKER in err blocked_config = blocked_config_silent or BLOCKED_CONFIG_MARKER in err - drift_skip_silent = DRIFT_SKIP_SILENT_MARKER in err - drift_skip = drift_skip_silent or DRIFT_SKIP_MARKER in err incident_acked = False failure_incident_id = None if blocked_config and not success: @@ -2646,20 +2554,13 @@ def _compose_run_delivery( incident_acked, failure_incident_id = _upsert_incident_for_failure( job, error or "", output_file=output_file ) - if incident_acked and not drift_skip: + if incident_acked: deliver_content = "" else: deliver_content = ( _summarize_cron_failure_for_delivery(job, error) + _failure_streak_nudge(job) ) - if drift_skip: - # Deliver the guard's message intact (summarizer truncation would eat the remediation - # command). NOT gated on incident ack: acks silence failure pings, not drift alerts. - _drift_text = re.sub(r"\[drift_skip[^\]]*\]\s*", "", err).strip() - deliver_content = f"⚠️ Cron '{job.get('name') or job['id']}' skipped: {_drift_text}" - return ( - deliver_content, blocked_config, blocked_config_silent or drift_skip_silent, - incident_acked, failure_incident_id) + return deliver_content, blocked_config, blocked_config_silent, incident_acked, failure_incident_id class _FireClaimLostDuringSideEffect(Exception): @@ -3876,9 +3777,8 @@ from cron.scheduler_prompt import ( # noqa: E402 _block_and_pause_job, _build_job_prompt, _guard_job_credential_exfil, _parse_wake_gate, ) from cron.scheduler_preflight import ( # noqa: E402 - BLOCKED_CONFIG_MARKER, BLOCKED_CONFIG_SILENT_MARKER, DRIFT_SKIP_MARKER, - DRIFT_SKIP_SILENT_MARKER, _cron_preflight_enabled, _is_transient_provider_resolve_error, - _preflight_job_config, + BLOCKED_CONFIG_MARKER, BLOCKED_CONFIG_SILENT_MARKER, _cron_preflight_enabled, + _is_transient_provider_resolve_error, _preflight_job_config, ) diff --git a/cron/scheduler_preflight.py b/cron/scheduler_preflight.py index d964132dc7..8712cc4f71 100644 --- a/cron/scheduler_preflight.py +++ b/cron/scheduler_preflight.py @@ -21,12 +21,6 @@ logger = logging.getLogger("cron.scheduler") # alert-once dedup. ``:silent`` = already alerted on a previous tick — do not deliver again. BLOCKED_CONFIG_MARKER = "[blocked_config]" BLOCKED_CONFIG_SILENT_MARKER = "[blocked_config:silent]" -# Drift-guard skip: same contract (drift_alerted bit on the job record). -# Same alert-once contract as blocked_config: run_one_job keys off it to record last_status and the -# ``:silent`` variant means "already alerted on a previous tick — do not deliver again" (the drift_alerted -# bit on the job record, #73506 shape). -DRIFT_SKIP_MARKER = "[drift_skip]" -DRIFT_SKIP_SILENT_MARKER = "[drift_skip:silent]" _TRANSIENT_NET_EXC_NAMES = frozenset({ "ConnectError", "ConnectTimeout", "ReadTimeout", "WriteTimeout", "PoolTimeout", "NetworkError", @@ -316,8 +310,8 @@ def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]: so the caller refuses BEFORE building agent machinery or burning an LLM call. Every check fails open — preflight blocks only on an affirmative misconfiguration verdict. - Same fail-before-spend spirit as the #44585 drift guard and the fail-loud-on-hidden-tools direction in - #27948; alert dedup follows the alert-once pattern from the dead-pin auto-pause (#73506). + Same fail-before-spend spirit as the fail-loud-on-hidden-tools direction in #27948; alert dedup + follows the alert-once pattern from the dead-pin auto-pause (#73506). """ for name, check in ( ("provider_key", lambda: _preflight_check_provider_key(job, cfg)), diff --git a/hermes_cli/config.py b/hermes_cli/config.py index b32a6da5b2..dffeb513a6 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -2949,7 +2949,7 @@ def edit_config(): subprocess.run([editor, str(config_path)]) -# ---- Cron model-drift guard helpers ---- +# ---- Cron model-drift helpers: which unpinned jobs stay on their creation snapshot ---- _CRON_DRIFT_AXIS_BY_KEY = { "model": "model", "model.default": "model", "model.model": "model", "model.name": "model", @@ -2957,7 +2957,7 @@ _CRON_DRIFT_AXIS_BY_KEY = { def _cron_model_drift_axis_for_config_key(key: str) -> Optional[str]: - """Return the cron drift guard axis affected by a config key, if any.""" + """Return the cron inference axis affected by a config key, if any.""" return _CRON_DRIFT_AXIS_BY_KEY.get(str(key or "").strip().lower()) @@ -2972,15 +2972,6 @@ def _cron_section(config: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]: return cron_config if isinstance(cron_config, dict) else None -def cron_model_drift_guard_enabled(config: Optional[Dict[str, Any]] = None) -> bool: - """Whether cron must fail closed on unpinned inference drift. - Only the literal YAML boolean ``false`` disables this spend-safety guard; missing, malformed, - or non-boolean values stay fail-closed. With *config* omitted the merged config is loaded so - CLI warnings honor the same user/managed setting as the scheduler.""" - cron_config = _cron_section(config) - return cron_config is None or cron_config.get("model_drift_guard", True) is not False - - _CRON_MODEL_IMPACT_JOB_LIMIT = 50 _CRON_MODEL_IMPACT_ID_LIMIT = 256 _CRON_MODEL_IMPACT_NAME_LIMIT = 120 @@ -2996,7 +2987,7 @@ def resolve_cron_model_drift_defaults( """Resolve the global ``(provider, model)`` cron compares against snapshots. Mirrors the scheduler's precedence: a truthy configured model wins over ``HERMES_MODEL``; the environment is only a fallback. Per-job and cron fleet defaults are handled by the caller - because they suppress a drift axis rather than changing the global assignment.""" + because they cover an axis rather than changing the global assignment.""" env = os.environ if environ is None else environ provider = "" model_config = config.get("model") if isinstance(config, dict) else None @@ -3010,15 +3001,16 @@ def resolve_cron_model_drift_defaults( def cron_model_drift_axes( job: Any, *, current_provider: Any = "", current_model: Any = "", config: Any = None ) -> List[str]: - """Return the unpinned axes that the fail-closed cron guard would block.""" - if not isinstance(job, dict) or not cron_model_drift_guard_enabled(config): + """Return the unpinned axes on which *job* will keep running on its creation snapshot rather + than the new global assignment (the scheduler treats the snapshot as the effective pin).""" + if not isinstance(job, dict): return [] current = { "provider": _model_assignment_text(current_provider).lower(), "model": _model_assignment_text(current_model).lower()} - # A cron.model / cron.model_provider fleet default covers its axis: that axis no longer follows - # the global assignment at fire time, so the guard never engages and a warning would be false. + # A cron.model / cron.model_provider fleet default covers its axis: that axis never reads the + # snapshot at fire time, so reporting it would be false. fleet = _cron_section(config) or {} drifted: List[str] = [] for axis, fleet_key in (("provider", "model_provider"), ("model", "model")): @@ -3050,35 +3042,28 @@ def _cron_impact_job_name(value: Any, job_id: str) -> str: return f"Job {job_id}"[:_CRON_MODEL_IMPACT_NAME_LIMIT].rstrip() -def _cron_model_impact_result(available: bool, guard_enabled: bool) -> Dict[str, Any]: - return { - "available": available, - "guard_enabled": guard_enabled, - "affected_count": 0, - "truncated": False, - "jobs": []} +def _cron_model_impact_result(available: bool) -> Dict[str, Any]: + return {"available": available, "affected_count": 0, "truncated": False, "jobs": []} def build_cron_model_impact( *, current_provider: Any = "", current_model: Any = "", config: Any = None, jobs: Any = None ) -> Dict[str, Any]: - """Build a bounded, profile-local summary of jobs blocked by model drift. - Job-store inspection is best effort: the model assignment has already succeeded when Desktop - requests this, so an unreadable store is reported as unavailable rather than failing.""" - guard_enabled = cron_model_drift_guard_enabled(config) + """Build a bounded, profile-local summary of unpinned jobs that stay on their creation snapshot + after a global model/provider change. Job-store inspection is best effort: the model assignment + has already succeeded when Desktop requests this, so an unreadable store is reported as + unavailable rather than failing.""" if jobs is None: try: from cron.jobs import load_jobs jobs = load_jobs() except Exception: - return _cron_model_impact_result(False, guard_enabled) + return _cron_model_impact_result(False) if not isinstance(jobs, list): - return _cron_model_impact_result(False, guard_enabled) + return _cron_model_impact_result(False) - result = _cron_model_impact_result(True, guard_enabled) - if not guard_enabled: - return result + result = _cron_model_impact_result(True) from cron.jobs import is_job_runnable @@ -3107,7 +3092,7 @@ def build_cron_model_impact( def warn_unpinned_cron_jobs_after_model_config_change( key: str, value: Any, config: Optional[Dict[str, Any]] = None) -> None: - """Warn when a global model/provider change will trip cron's drift guard.""" + """Tell the operator which unpinned cron jobs a global model/provider change does NOT move.""" axis = _cron_model_drift_axis_for_config_key(key) if axis is None: return @@ -3122,13 +3107,12 @@ def warn_unpinned_cron_jobs_after_model_config_change( if affected <= 0: return - noun, verb = ("job", "has") if affected == 1 else ("jobs", "have") + noun, verb = ("job", "keeps") if affected == 1 else ("jobs", "keep") print( - f"⚠️ {affected} enabled unpinned cron {noun} {verb} stored " - f"{axis}_snapshot values that differ from the new global {axis}. " - "They will fail closed on their next run instead of silently using the changed " - "model/provider. Inspect with `hermes cron list`, then pin the intended values with " - "`hermes cron edit --provider --model `.") + f"ℹ️ {affected} unpinned cron {noun} {verb} running on the {axis} it was created under " + f"(its {axis}_snapshot), not the new global {axis}. To move it, pin it with " + "`hermes cron edit --provider --model ` or set a fleet default " + "with `hermes config set cron.model `.") def _default_value_for_key(dotted_key: str): diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 40d8381407..df64c94219 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1628,13 +1628,10 @@ DEFAULT_CONFIG = { # platforms are configured. Failure -> last_status=blocked_config, ONE alert, no LLM call. # False = fail during the run instead. "preflight": True, - # Fail closed when an unpinned job's current global model/provider differs from its - # creation-time snapshot, so unattended jobs never silently inherit a paid default. False - # only when jobs should track changing global inference defaults. - "model_drift_guard": True, # Default model for cron jobs (WHAT model runs). Fire-time resolution: per-job pin > - # cron.model > model.default. When set, unpinned jobs follow it deliberately and the drift - # guard does not engage for the model axis. "" = fall through to model.default. + # cron.model > the job's creation-time snapshot > model.default. An unpinned job keeps + # running on the model it was created under when model.default later changes; cron.model + # is the way to move the whole fleet at once. "" = fall through. "model": "", # Inference provider paired with cron.model (NOT the scheduler provider below). "" = resolve # from global config. @@ -2340,7 +2337,7 @@ DEFAULT_CONFIG = { # Extra ports detection probes for an external llama-server (besides 8080). "detect_ports": [], }, - "_config_version": 41, # Config schema version - bump this when adding new required fields + "_config_version": 42, # Config schema version - bump this when adding new required fields } diff --git a/hermes_cli/config_migrations.py b/hermes_cli/config_migrations.py index d91983f95c..92b03cb4fc 100644 --- a/hermes_cli/config_migrations.py +++ b/hermes_cli/config_migrations.py @@ -627,6 +627,16 @@ MIGRATIONS: Tuple[Tuple[int, Callable[[Dict[str, Any], bool], None]], ...] = ( message=" ✓ Model catalog now refreshes every 20 minutes (model_catalog.ttl_minutes)", extra_guard=lambda raw: "ttl_minutes" not in raw)), (41, _migrate_to_41), + # 41 → 42: cron.model_drift_guard is gone. Unpinned jobs now run on their creation snapshot + # instead of failing closed when the global model changes, so the toggle has nothing to gate. + (42, functools.partial( + _rewrite_key, section="cron", key="model_drift_guard", new=None, + match=lambda cur: cur is not None, + added="removed cron.model_drift_guard", + message=( + " ✓ Removed cron.model_drift_guard — unpinned cron jobs now keep running on the " + "model/provider they were created under when the global default changes, instead " + "of being skipped. Pin a job or set cron.model to move it."))), )