fix(cron): unpinned jobs run on their creation-snapshot model instead of failing closed

A global model/provider change must never stop a cron job. The #44585 guard
raised [drift_skip] for every unpinned job whose provider_snapshot /
model_snapshot no longer matched the live global default, so one `hermes model`
switch silently killed whole fleets (reported by fastfinge, nitinthewiz,
Dr-ilies; 13 of 60 jobs on the project lead's box after
claude-fable-5 -> claude-fable-5.1).

The snapshot is now the job's effective pin: _load_cron_job_config prefers
job['model_snapshot'] over the global default and _resolve_job_runtime passes
job['provider_snapshot'] as `requested` when neither a per-job pin nor a
cron.model / cron.model_provider fleet default covers the axis. One INFO line
per differing axis tells the operator what the job is running on and how to
move it. Jobs without a snapshot (legacy records) still follow the global
default; the existing fallback chain still handles a snapshot provider that
fails to resolve.

Both goals of #44585 hold: no silent inherit of a paid default (the job runs on
what it was created under) and no outage. Owner decision (Teknium): "main agent
model changing should not stop crons from executing, ever".

Removed as unreachable: _check_model_drift, DRIFT_SKIP markers, the
drift_alerted alert-once bit (mark_drift_alerted + the _record_run_outcome pop),
the drift special-cases in _compose_run_delivery and
_summarize_cron_failure_for_delivery, cron_model_drift_guard_enabled and the
cron.model_drift_guard config key (v42 migration drops it from existing
configs). The PLUGIN-COMPAT clear_drift_alerted block is untouched (scheduled
revert).

The `hermes config set model.default` notice and the Desktop model-change toast
are reworded from "will fail closed / will be skipped" to "keep running on the
model they were created under"; the impact payload drops guard_enabled (all six
desktop locales updated).
This commit is contained in:
Teknium
2026-09-09 04:19:17 -07:00
parent 48465c3933
commit 7e4d02fef5
15 changed files with 107 additions and 225 deletions
+3 -2
View File
@@ -1564,8 +1564,9 @@ export const ar = defineLocale({
cron: {
close: 'إغلاق',
modelImpact: {
title: 'تحتاج المهام المجدولة إلى المراجعة',
message: count => `سيتم تخطي ${count} من المهام المجدولة حتى تراجع إعدادات النموذج الخاصة بها.`,
title: 'تبقى المهام المجدولة على نموذجها الأصلي',
message: count =>
`${count} من المهام المجدولة غير المثبتة ستواصل العمل على النموذج الذي أُنشئت به. ثبّتها أو اضبط cron.model لنقلها.`,
detailMore: (names, remaining) => `${names} و${remaining} أخرى`,
review: 'مراجعة المهام المجدولة',
saveFailed: 'لم يحفظ Hermes تغيير النموذج هذا.',
+2 -2
View File
@@ -2174,9 +2174,9 @@ export const en: Translations = {
title: 'Scheduled jobs',
count: count => `${count} ${count === 1 ? 'job' : 'jobs'}`,
modelImpact: {
title: 'Scheduled jobs need review',
title: 'Scheduled jobs stay on their original model',
message: count =>
`${count} scheduled ${count === 1 ? 'job' : 'jobs'} will be skipped until you review their model settings.`,
`${count} unpinned scheduled ${count === 1 ? 'job keeps' : 'jobs keep'} running on the model ${count === 1 ? 'it was' : 'they were'} created under. Pin ${count === 1 ? 'it' : 'them'} or set cron.model to move ${count === 1 ? 'it' : 'them'}.`,
detailMore: (names, remaining) => `${names} and ${remaining} more`,
review: 'Review scheduled jobs',
saveFailed: 'Hermes did not save that model change.',
+3 -2
View File
@@ -1846,8 +1846,9 @@ export const ja = defineLocale({
title: 'スケジュール済みジョブ',
count: count => `${count} 件のジョブ`,
modelImpact: {
title: 'スケジュール済みジョブの確認が必要です',
message: count => `モデル設定を確認するまで、${count} 件のスケジュール済みジョブがスキップされます。`,
title: 'スケジュール済みジョブは元のモデルで実行されます',
message: count =>
`ピン留めされていない ${count} 件のスケジュール済みジョブは、作成時のモデルで引き続き実行されます。移行するにはピン留めするか cron.model を設定してください。`,
detailMore: (names, remaining) => `${names}、ほか ${remaining} 件`,
review: 'スケジュール済みジョブを確認',
saveFailed: 'Hermes はモデルの変更を保存しませんでした。',
+2 -2
View File
@@ -2212,9 +2212,9 @@ export const ru = defineLocale({
title: 'Запланированные задачи',
count: count => `${count} ${RU_PLURAL(count, 'задача', 'задачи', 'задач')}`,
modelImpact: {
title: 'Запланированные задачи требуют проверки',
title: 'Запланированные задачи остаются на исходной модели',
message: count =>
`Будет пропущено ${count} ${RU_NOUN(count, 'задача', 'задачи', 'задач')} до тех пор, пока вы не проверите их настройки модели.`,
`${count} незакреплённых запланированных задач продолжат работать на модели, с которой были созданы. Закрепите их или задайте cron.model, чтобы перевести.`,
detailMore: (names, remaining) => `${names} и ещё ${remaining}`,
review: 'Проверить запланированные задачи',
saveFailed: 'Hermes не сохранил это изменение модели.'
+3 -2
View File
@@ -1773,8 +1773,9 @@ export const zhHant = defineLocale({
title: '排程工作',
count: count => `${count} 個工作`,
modelImpact: {
title: '排程工作需要檢查',
message: count => `在您檢查模型設定之前,${count} 個排程工作將被略過。`,
title: '排程工作將繼續使用原模型',
message: count =>
`${count} 個未固定的排程工作將繼續使用建立時的模型執行。固定它們或設定 cron.model 以遷移。`,
detailMore: (names, remaining) => `${names},以及另外 ${remaining} 個`,
review: '檢查排程工作',
saveFailed: 'Hermes 未儲存該模型變更。',
+3 -2
View File
@@ -2340,8 +2340,9 @@ export const zh: Translations = {
title: '定时任务',
count: count => `${count} 个任务`,
modelImpact: {
title: '定时任务需要检查',
message: count => `在您检查模型设置之前,${count} 个定时任务将被跳过。`,
title: '定时任务将继续使用原模型',
message: count =>
`${count} 个未固定的定时任务将继续使用创建时的模型运行。固定它们或设置 cron.model 以迁移。`,
detailMore: (names, remaining) => `${names},以及另外 ${remaining} 个`,
review: '检查定时任务',
saveFailed: 'Hermes 未保存该模型更改。',
+4 -3
View File
@@ -74,7 +74,6 @@ export function parseCronModelImpact(value: unknown): CronModelImpact | null {
if (
typeof impact.available !== 'boolean' ||
typeof impact.guard_enabled !== 'boolean' ||
!Number.isSafeInteger(impact.affected_count) ||
(impact.affected_count ?? -1) < 0 ||
typeof impact.truncated !== 'boolean' ||
@@ -121,15 +120,17 @@ function publishImpact(impact: CronModelImpact, profile: string, connection: str
return
}
if (!impact.guard_enabled || impact.affected_count === 0) {
if (impact.affected_count === 0) {
dismissNotification(CRON_MODEL_IMPACT_NOTIFICATION_ID)
return
}
// Informational: these jobs keep running on the model they were created under; nothing is
// skipped. The action is a read-only review so the user can pin or move them deliberately.
notify({
id: CRON_MODEL_IMPACT_NOTIFICATION_ID,
kind: 'warning',
kind: 'info',
title: translateNow('cron.modelImpact.title'),
message: translateNow('cron.modelImpact.message', impact.affected_count),
detail: detailFor(impact),
-1
View File
@@ -1494,7 +1494,6 @@ export interface CronModelImpactJob {
export interface CronModelImpact {
available: boolean
guard_enabled: boolean
affected_count: number
truncated: boolean
jobs: CronModelImpactJob[]
+1 -1
View File
@@ -2408,7 +2408,7 @@ def save_config_value(key_path: str, value: any) -> bool:
os.chmod(config_path, 0o600)
except (OSError, NotImplementedError):
pass
# Same fail-closed cron drift warning as `hermes config set` for every model switch.
# Same unpinned-cron notice as `hermes config set` for every model switch.
from hermes_cli.config import warn_unpinned_cron_jobs_after_model_config_change
warn_unpinned_cron_jobs_after_model_config_change(key_path, value)
+6 -13
View File
@@ -1494,7 +1494,7 @@ def _normalize_workdir(workdir: Optional[str]) -> Optional[str]:
def _resolve_default_model_snapshot() -> Optional[str]:
"""Default model resolved as the ticker's ``run_job`` does, so unpinned jobs can snapshot it and
detect a later swap. ``None`` on missing config or failure (fail-open: "no snapshot")."""
keep running on it after a later swap. ``None`` on missing config or failure ("no snapshot")."""
try:
from hermes_cli.config import _expand_env_vars, read_user_config_raw
@@ -1599,8 +1599,9 @@ _UPDATE_FIELD_NORMALIZERS: Dict[str, Callable[[Any], Any]] = {
def _compute_provider_model_snapshots(
*, provider: Any, model: Any, base_url: Any, no_agent: Any,
) -> Tuple[Optional[str], Optional[str]]:
"""Snapshot unpinned provider/model resolution so a later global switch fails closed at fire
time instead of silently changing spend. Pinned axes and no-agent jobs carry no snapshot."""
"""Snapshot unpinned provider/model resolution: the scheduler runs the job on this snapshot after
a later global switch instead of silently changing spend. Pinned axes and no-agent jobs carry no
snapshot."""
normalized_provider = _normalize_job_optional_text(provider)
normalized_model = _normalize_job_optional_text(model)
normalized_base_url = _normalize_base_url(base_url)
@@ -2108,13 +2109,11 @@ def remove_job(job_id: str) -> bool:
def _set_alert_flag(job_id: str, field: str, value: bool) -> bool:
"""Set/clear a persisted alert-dedup marker (alert exactly once until the condition heals;
survives restarts) and return the PRIOR value. Fields: ``preflight_alerted``,
``drift_alerted``.
survives restarts) and return the PRIOR value. Field: ``preflight_alerted`` (blocked config).
The marker records that the operator was already alerted about this job's condition, so the scheduler
alerts exactly once and stays silent on subsequent ticks until the condition heals (same alert-once
shape as the dead-pin auto-pause in #73506). Fields: ``preflight_alerted`` (blocked config, T1-26) and
``drift_alerted`` (#44585 drift-guard skip).
shape as the dead-pin auto-pause in #73506).
"""
def apply(jobs, _i, job):
prior = bool(job.get(field))
@@ -2139,11 +2138,6 @@ def clear_preflight_alerted(job_id: str) -> None:
_set_alert_flag(job_id, "preflight_alerted", False)
def mark_drift_alerted(job_id: str) -> bool:
"""Mark the job as drift-alerted; return True if it already was."""
return _set_alert_flag(job_id, "drift_alerted", True)
def note_fire_forward_failure(job_id: str, detail: str) -> bool:
"""Durably record (as ``last_fire_error``) that a scheduled fire could not be handed to the
runner — written by the dashboard fire webhook when the loopback forward fails. Without it
@@ -2175,7 +2169,6 @@ def _record_run_outcome(
# Healthy run: drop the alert-once dedup markers so a FUTURE break re-alerts, and clear
# the forward-failure stamp so it only describes CURRENT auto-fire health.
job.pop("preflight_alerted", None)
job.pop("drift_alerted", None)
job.pop("last_fire_error", None)
job["failure_streak"] = 0
else:
+41 -141
View File
@@ -38,8 +38,7 @@ sys.path.insert(0, str(Path(__file__).parent.parent))
from hermes_constants import get_hermes_home
from hermes_cli._subprocess_compat import windows_hide_flags
from hermes_cli.config import (
_expand_env_vars, cron_model_drift_axes, cron_model_drift_guard_enabled, load_config,
resolve_cron_model_drift_defaults)
_expand_env_vars, load_config, resolve_cron_model_drift_defaults)
from hermes_cli.fallback_config import get_fallback_chain
from hermes_time import now as _hermes_now
from agent.interrupt_compat import request_hard_interrupt
@@ -220,24 +219,6 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
text = (error or "unknown error").strip()
lower = text.lower()
if "skipped to prevent unintended spend: global inference config drifted" in lower:
if "finite one-shot job is consumed" in lower:
remediation = (
"This finite one-shot is consumed; create a new one-shot job at "
"a future time with an explicit provider and model."
)
else:
job_id = job.get("id") or "<job_id>"
remediation = (
"On the host running Hermes, pin it explicitly: "
f"`hermes cron edit {job_id} --provider <provider> "
"--model <model>`."
)
return (
f"⚠️ Cron '{job_name}' skipped before inference to prevent "
f"unintended spend. {remediation}"
)
# no_agent jobs never reach a model, so provider errors are structurally impossible for them.
# Gate on job MODE before substring matching, or a script's own wording ("timed out", "429")
# would blame the wrong subsystem; the generic cleaner below reports what actually happened.
@@ -1354,10 +1335,26 @@ class _CronJobConfig:
cron_default_provider: str
def _snapshot_pin(job: dict, axis: str, current: str, job_id: str) -> str:
"""The creation snapshot is an unpinned axis's effective pin: return it, logging once when it
differs from *current* (the live global default); ``""`` for legacy jobs without one, which keep
following the global default. A global model/provider change must never stop a cron job; a job
keeps running on what it was created under until the operator pins it or sets a cron.* fleet
default (#44585)."""
snapshot = str(job.get(f"{axis}_snapshot") or "").strip()
if snapshot and current and snapshot.lower() != current.lower():
logger.info(
"Job '%s': running on creation-snapshot %s %r (global default is now %r); "
"`hermes cron edit %s --%s <value>` or cron.%s in config.yaml moves it.",
job_id, axis, snapshot, current, job_id, axis,
"model" if axis == "model" else "model_provider")
return snapshot
def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConfig:
"""Load config.yaml and resolve the run's model: per-job override > cron.model (fleet default) >
HERMES_MODEL > config ``model:``. Re-read every tick (no cache) so ``hermes cron edit --model``
applies next tick. An axis resolved from cron.model/model_provider is explicit (no drift guard)."""
creation snapshot > HERMES_MODEL > config ``model:``. Re-read every tick (no cache) so
``hermes cron edit --model`` applies next tick."""
model = job.get("model") or os.getenv("HERMES_MODEL") or ""
_cron_default_provider = ""
_cfg: dict = {}
@@ -1383,10 +1380,8 @@ def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConf
if _cron_default_model:
model = _cron_default_model
else:
# Shared with Desktop's impact summary so both compare against the same model.
_, _global_model = resolve_cron_model_drift_defaults(_cfg)
if _global_model:
model = _global_model
model = _snapshot_pin(job, "model", _global_model, job_id) or _global_model or model
except Exception as e:
logger.warning("Job '%s': failed to load config.yaml, using defaults: %s", job_id, e)
@@ -1489,42 +1484,34 @@ def _preflight_or_block(job: dict, job_id: str, job_name: str, cfg: dict) -> Opt
return False, blocked_doc, "", f"{marker} {_pf_reason}"
def _resolve_job_runtime(
job: dict, job_id: str, jc: _CronJobConfig,
) -> tuple[dict, str, Optional[str]]:
def _resolve_job_runtime(job: dict, job_id: str, jc: _CronJobConfig) -> tuple[dict, str]:
"""Resolve the runtime, walking the fallback chain on auth/transient-network errors. Returns
``(runtime, model, primary_provider_for_drift)``; provider+model swap atomically (never swap
only the provider while keeping a paid primary model)."""
``(runtime, model)``; provider+model swap atomically (never swap only the provider while keeping
a paid primary model). Provider precedence: per-job pin > cron.model_provider > creation
snapshot > persisted global config."""
from hermes_cli.runtime_provider import (
resolve_runtime_provider, format_runtime_provider_error)
from hermes_cli.auth import AuthError
model = jc.model
configured_provider_for_drift = (
str(jc.model_cfg.get("provider") or "").strip().lower()
if isinstance(jc.model_cfg, dict)
else ""
)
primary_provider_for_drift = (
str(job.get("provider") or "").strip().lower()
or configured_provider_for_drift
or None
)
requested = job.get("provider") or jc.cron_default_provider or None
if not requested:
global_provider = (
str(jc.model_cfg.get("provider") or "").strip() if isinstance(jc.model_cfg, dict) else "")
# None (not the config provider) keeps the legacy no-snapshot path resolving from persisted
# config exactly as before.
requested = _snapshot_pin(job, "provider", global_provider, job_id) or None
try:
# Do NOT pass HERMES_INFERENCE_PROVIDER as `requested`: it would override persisted config
# and resurrect stale providers for unpinned jobs.
runtime_kwargs = {
"requested": job.get("provider") or jc.cron_default_provider or None,
"requested": requested,
# api_mode must derive from the model actually run, not the stale persisted default.
"target_model": model,
}
if job.get("base_url"):
runtime_kwargs["explicit_base_url"] = job.get("base_url")
runtime = resolve_runtime_provider(**runtime_kwargs)
primary_provider_for_drift = (
str(runtime.get("provider") or "").strip().lower() or primary_provider_for_drift
)
return runtime, model, primary_provider_for_drift
return resolve_runtime_provider(**runtime_kwargs), model
except Exception as resolve_exc:
# Walk the fallback chain on AuthError AND transient network/DNS failures (e.g. during
# OAuth refresh); anything else re-raises.
@@ -1533,10 +1520,6 @@ def _resolve_job_runtime(
if not (is_auth or is_transient_net):
raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
primary_provider_for_drift = (
str(getattr(resolve_exc, "provider", "") or "").strip().lower()
or primary_provider_for_drift
)
logger.warning(
"Job '%s': primary provider resolve failed (%s: %s), trying fallback",
job_id, "auth" if is_auth else "transient network", resolve_exc)
@@ -1560,82 +1543,12 @@ def _resolve_job_runtime(
logger.info(
"Job '%s': fallback resolved to %s model %s",
job_id, runtime.get("provider"), fb_model)
return runtime, fb_model, primary_provider_for_drift
return runtime, fb_model
except Exception as fb_exc:
logger.debug("Job '%s': fallback %s failed: %s", job_id, fb_provider, fb_exc)
raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
def _check_model_drift(
job: dict, job_id: str, cfg: dict, runtime: dict,
primary_provider_for_drift: Optional[str], primary_model_for_drift: str,
) -> None:
"""Fail-closed provider/model drift guard; raises RuntimeError (with drift marker) on drift.
An unpinned job follows the global default, which may have switched to a paid provider/model:
each unpinned axis whose creation snapshot (job["<axis>_snapshot"]) now resolves differently
skips the run and alerts to pin. No snapshot, pinned axes, or the cron.model fleet default
never count as drift."""
if not cron_model_drift_guard_enabled(cfg):
return
_current_provider = str(
primary_provider_for_drift or runtime.get("provider") or ""
).strip().lower()
_current_model = str(primary_model_for_drift or "").strip().lower()
_drift: list[str] = []
for _axis in cron_model_drift_axes(
job, current_provider=_current_provider, current_model=_current_model, config=cfg):
_snapshot = str(job.get(f"{_axis}_snapshot") or "").strip().lower()
_current = _current_provider if _axis == "provider" else _current_model
_drift.append(f"{_axis} '{_snapshot}' -> '{_current}'")
if not _drift:
return
_changes = "; ".join(_drift)
# A finite one-shot is consumed by this attempt, so "edit the job" is a dead end for it.
# Lifecycle-aware remediation (#72056, @sashmatash): a finite one-shot is consumed by this attempted
# dispatch — telling an operator to edit a spent job is a dead end. Recurring and repeatable jobs get
# the pin command instead.
_repeat = job.get("repeat") if isinstance(job.get("repeat"), dict) else {}
_finite_oneshot = (
isinstance(job.get("schedule"), dict)
and job["schedule"].get("kind") == "once"
and _repeat.get("times") == 1
)
if _finite_oneshot:
_remediation = (
"This finite one-shot job is consumed by this attempted run; "
"create a new one-shot job at a future time with an explicit provider and model."
)
else:
_remediation = (
"To run on the new config, on the host running Hermes pin it explicitly: "
f"`hermes cron edit {job_id} --provider <provider> "
"--model <model>` (or pin the original values to keep them)."
)
logger.warning(
"Job '%s': SKIPPED — global inference config drifted since "
"creation (%s) and this job is unpinned. Skipped to prevent unintended spend. %s",
job_id, _changes, _remediation)
# Alert-once via drift_alerted bit (silent marker suppresses delivery); a successful run
# clears it and re-arms the alert.
# Alert-once (#73506 shape): persist the drift_alerted bit so only the FIRST drifted tick delivers;
# run_one_job suppresses delivery on the silent marker. mark_job_run clears the bit when a run succeeds
# (drift healed), re-arming the alert.
_drift_already_alerted = False
with contextlib.suppress(Exception):
from cron.jobs import mark_drift_alerted
_drift_already_alerted = mark_drift_alerted(job_id)
_drift_marker = DRIFT_SKIP_SILENT_MARKER if _drift_already_alerted else DRIFT_SKIP_MARKER
raise RuntimeError(
f"{_drift_marker} Skipped to prevent unintended spend: global "
f"inference config drifted since this job was created "
f"({_changes}), and this job is unpinned. No inference call "
f"was made. {_remediation} "
f"This alert is sent once; the job stays skipped until the "
f"config is pinned or restored. See #44585."
)
def _load_credential_pool(runtime: dict, job_id: str):
runtime_provider = str(runtime.get("provider") or "").strip().lower()
if not runtime_provider:
@@ -2203,7 +2116,7 @@ class _CronAgentSetup:
def _resolve_cron_agent_setup(job: dict, job_id: str, job_name: str, jc) -> _CronAgentSetup:
"""Resolve model/runtime/reasoning/pool for the run, in the original gate order: exfil guard ->
preflight (may block) -> runtime -> drift check -> fallback chain -> credential pool -> MCP."""
preflight (may block) -> runtime (+ fallback chain) -> credential pool -> MCP."""
_cfg = jc.cfg
setup = _CronAgentSetup(model=jc.model)
setup.prefill_messages = _load_prefill_messages(_cfg, job_id)
@@ -2223,13 +2136,10 @@ def _resolve_cron_agent_setup(job: dict, job_id: str, job_name: str, jc) -> _Cro
if setup.blocked is not None:
return setup
primary_model_for_drift = setup.model
setup.runtime, setup.model, primary_provider_for_drift = _resolve_job_runtime(job, job_id, jc)
setup.runtime, setup.model = _resolve_job_runtime(job, job_id, jc)
setup.reasoning_config = _resolve_job_reasoning_config(
job, _cfg if isinstance(_cfg, dict) else {}, str(setup.model)
)
_check_model_drift(
job, job_id, _cfg, setup.runtime, primary_provider_for_drift, primary_model_for_drift)
setup.fallback_model = get_fallback_chain(_cfg) or None
setup.credential_pool = _load_credential_pool(setup.runtime, job_id)
# MCP servers must be registered before AIAgent is constructed.
@@ -2622,11 +2532,9 @@ def _compose_run_delivery(
silent_alert, incident_acked, failure_incident_id)``; ``silent_alert``: an alert-once marker
says the operator was already told, deliver nothing."""
err = str(error) if error else ""
# Failed jobs always deliver, except blocked-config / drift-skip runs, which alert exactly ONCE.
# Failed jobs always deliver, except blocked-config runs, which alert exactly ONCE.
blocked_config_silent = BLOCKED_CONFIG_SILENT_MARKER in err
blocked_config = blocked_config_silent or BLOCKED_CONFIG_MARKER in err
drift_skip_silent = DRIFT_SKIP_SILENT_MARKER in err
drift_skip = drift_skip_silent or DRIFT_SKIP_MARKER in err
incident_acked = False
failure_incident_id = None
if blocked_config and not success:
@@ -2646,20 +2554,13 @@ def _compose_run_delivery(
incident_acked, failure_incident_id = _upsert_incident_for_failure(
job, error or "", output_file=output_file
)
if incident_acked and not drift_skip:
if incident_acked:
deliver_content = ""
else:
deliver_content = (
_summarize_cron_failure_for_delivery(job, error) + _failure_streak_nudge(job)
)
if drift_skip:
# Deliver the guard's message intact (summarizer truncation would eat the remediation
# command). NOT gated on incident ack: acks silence failure pings, not drift alerts.
_drift_text = re.sub(r"\[drift_skip[^\]]*\]\s*", "", err).strip()
deliver_content = f"⚠️ Cron '{job.get('name') or job['id']}' skipped: {_drift_text}"
return (
deliver_content, blocked_config, blocked_config_silent or drift_skip_silent,
incident_acked, failure_incident_id)
return deliver_content, blocked_config, blocked_config_silent, incident_acked, failure_incident_id
class _FireClaimLostDuringSideEffect(Exception):
@@ -3876,9 +3777,8 @@ from cron.scheduler_prompt import ( # noqa: E402
_block_and_pause_job, _build_job_prompt, _guard_job_credential_exfil, _parse_wake_gate,
)
from cron.scheduler_preflight import ( # noqa: E402
BLOCKED_CONFIG_MARKER, BLOCKED_CONFIG_SILENT_MARKER, DRIFT_SKIP_MARKER,
DRIFT_SKIP_SILENT_MARKER, _cron_preflight_enabled, _is_transient_provider_resolve_error,
_preflight_job_config,
BLOCKED_CONFIG_MARKER, BLOCKED_CONFIG_SILENT_MARKER, _cron_preflight_enabled,
_is_transient_provider_resolve_error, _preflight_job_config,
)
+2 -8
View File
@@ -21,12 +21,6 @@ logger = logging.getLogger("cron.scheduler")
# alert-once dedup. ``:silent`` = already alerted on a previous tick — do not deliver again.
BLOCKED_CONFIG_MARKER = "[blocked_config]"
BLOCKED_CONFIG_SILENT_MARKER = "[blocked_config:silent]"
# Drift-guard skip: same contract (drift_alerted bit on the job record).
# Same alert-once contract as blocked_config: run_one_job keys off it to record last_status and the
# ``:silent`` variant means "already alerted on a previous tick — do not deliver again" (the drift_alerted
# bit on the job record, #73506 shape).
DRIFT_SKIP_MARKER = "[drift_skip]"
DRIFT_SKIP_SILENT_MARKER = "[drift_skip:silent]"
_TRANSIENT_NET_EXC_NAMES = frozenset({
"ConnectError", "ConnectTimeout", "ReadTimeout", "WriteTimeout", "PoolTimeout", "NetworkError",
@@ -316,8 +310,8 @@ def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]:
so the caller refuses BEFORE building agent machinery or burning an LLM call. Every check fails
open — preflight blocks only on an affirmative misconfiguration verdict.
Same fail-before-spend spirit as the #44585 drift guard and the fail-loud-on-hidden-tools direction in
#27948; alert dedup follows the alert-once pattern from the dead-pin auto-pause (#73506).
Same fail-before-spend spirit as the fail-loud-on-hidden-tools direction in #27948; alert dedup
follows the alert-once pattern from the dead-pin auto-pause (#73506).
"""
for name, check in (
("provider_key", lambda: _preflight_check_provider_key(job, cfg)),
+23 -39
View File
@@ -2949,7 +2949,7 @@ def edit_config():
subprocess.run([editor, str(config_path)])
# ---- Cron model-drift guard helpers ----
# ---- Cron model-drift helpers: which unpinned jobs stay on their creation snapshot ----
_CRON_DRIFT_AXIS_BY_KEY = {
"model": "model", "model.default": "model", "model.model": "model", "model.name": "model",
@@ -2957,7 +2957,7 @@ _CRON_DRIFT_AXIS_BY_KEY = {
def _cron_model_drift_axis_for_config_key(key: str) -> Optional[str]:
"""Return the cron drift guard axis affected by a config key, if any."""
"""Return the cron inference axis affected by a config key, if any."""
return _CRON_DRIFT_AXIS_BY_KEY.get(str(key or "").strip().lower())
@@ -2972,15 +2972,6 @@ def _cron_section(config: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
return cron_config if isinstance(cron_config, dict) else None
def cron_model_drift_guard_enabled(config: Optional[Dict[str, Any]] = None) -> bool:
"""Whether cron must fail closed on unpinned inference drift.
Only the literal YAML boolean ``false`` disables this spend-safety guard; missing, malformed,
or non-boolean values stay fail-closed. With *config* omitted the merged config is loaded so
CLI warnings honor the same user/managed setting as the scheduler."""
cron_config = _cron_section(config)
return cron_config is None or cron_config.get("model_drift_guard", True) is not False
_CRON_MODEL_IMPACT_JOB_LIMIT = 50
_CRON_MODEL_IMPACT_ID_LIMIT = 256
_CRON_MODEL_IMPACT_NAME_LIMIT = 120
@@ -2996,7 +2987,7 @@ def resolve_cron_model_drift_defaults(
"""Resolve the global ``(provider, model)`` cron compares against snapshots.
Mirrors the scheduler's precedence: a truthy configured model wins over ``HERMES_MODEL``; the
environment is only a fallback. Per-job and cron fleet defaults are handled by the caller
because they suppress a drift axis rather than changing the global assignment."""
because they cover an axis rather than changing the global assignment."""
env = os.environ if environ is None else environ
provider = ""
model_config = config.get("model") if isinstance(config, dict) else None
@@ -3010,15 +3001,16 @@ def resolve_cron_model_drift_defaults(
def cron_model_drift_axes(
job: Any, *, current_provider: Any = "", current_model: Any = "", config: Any = None
) -> List[str]:
"""Return the unpinned axes that the fail-closed cron guard would block."""
if not isinstance(job, dict) or not cron_model_drift_guard_enabled(config):
"""Return the unpinned axes on which *job* will keep running on its creation snapshot rather
than the new global assignment (the scheduler treats the snapshot as the effective pin)."""
if not isinstance(job, dict):
return []
current = {
"provider": _model_assignment_text(current_provider).lower(),
"model": _model_assignment_text(current_model).lower()}
# A cron.model / cron.model_provider fleet default covers its axis: that axis no longer follows
# the global assignment at fire time, so the guard never engages and a warning would be false.
# A cron.model / cron.model_provider fleet default covers its axis: that axis never reads the
# snapshot at fire time, so reporting it would be false.
fleet = _cron_section(config) or {}
drifted: List[str] = []
for axis, fleet_key in (("provider", "model_provider"), ("model", "model")):
@@ -3050,35 +3042,28 @@ def _cron_impact_job_name(value: Any, job_id: str) -> str:
return f"Job {job_id}"[:_CRON_MODEL_IMPACT_NAME_LIMIT].rstrip()
def _cron_model_impact_result(available: bool, guard_enabled: bool) -> Dict[str, Any]:
return {
"available": available,
"guard_enabled": guard_enabled,
"affected_count": 0,
"truncated": False,
"jobs": []}
def _cron_model_impact_result(available: bool) -> Dict[str, Any]:
return {"available": available, "affected_count": 0, "truncated": False, "jobs": []}
def build_cron_model_impact(
*, current_provider: Any = "", current_model: Any = "", config: Any = None, jobs: Any = None
) -> Dict[str, Any]:
"""Build a bounded, profile-local summary of jobs blocked by model drift.
Job-store inspection is best effort: the model assignment has already succeeded when Desktop
requests this, so an unreadable store is reported as unavailable rather than failing."""
guard_enabled = cron_model_drift_guard_enabled(config)
"""Build a bounded, profile-local summary of unpinned jobs that stay on their creation snapshot
after a global model/provider change. Job-store inspection is best effort: the model assignment
has already succeeded when Desktop requests this, so an unreadable store is reported as
unavailable rather than failing."""
if jobs is None:
try:
from cron.jobs import load_jobs
jobs = load_jobs()
except Exception:
return _cron_model_impact_result(False, guard_enabled)
return _cron_model_impact_result(False)
if not isinstance(jobs, list):
return _cron_model_impact_result(False, guard_enabled)
return _cron_model_impact_result(False)
result = _cron_model_impact_result(True, guard_enabled)
if not guard_enabled:
return result
result = _cron_model_impact_result(True)
from cron.jobs import is_job_runnable
@@ -3107,7 +3092,7 @@ def build_cron_model_impact(
def warn_unpinned_cron_jobs_after_model_config_change(
key: str, value: Any, config: Optional[Dict[str, Any]] = None) -> None:
"""Warn when a global model/provider change will trip cron's drift guard."""
"""Tell the operator which unpinned cron jobs a global model/provider change does NOT move."""
axis = _cron_model_drift_axis_for_config_key(key)
if axis is None:
return
@@ -3122,13 +3107,12 @@ def warn_unpinned_cron_jobs_after_model_config_change(
if affected <= 0:
return
noun, verb = ("job", "has") if affected == 1 else ("jobs", "have")
noun, verb = ("job", "keeps") if affected == 1 else ("jobs", "keep")
print(
f"⚠️ {affected} enabled unpinned cron {noun} {verb} stored "
f"{axis}_snapshot values that differ from the new global {axis}. "
"They will fail closed on their next run instead of silently using the changed "
"model/provider. Inspect with `hermes cron list`, then pin the intended values with "
"`hermes cron edit <job_id> --provider <provider> --model <model>`.")
f"ℹ️ {affected} unpinned cron {noun} {verb} running on the {axis} it was created under "
f"(its {axis}_snapshot), not the new global {axis}. To move it, pin it with "
"`hermes cron edit <job_id> --provider <provider> --model <model>` or set a fleet default "
"with `hermes config set cron.model <model>`.")
def _default_value_for_key(dotted_key: str):
+4 -7
View File
@@ -1628,13 +1628,10 @@ DEFAULT_CONFIG = {
# platforms are configured. Failure -> last_status=blocked_config, ONE alert, no LLM call.
# False = fail during the run instead.
"preflight": True,
# Fail closed when an unpinned job's current global model/provider differs from its
# creation-time snapshot, so unattended jobs never silently inherit a paid default. False
# only when jobs should track changing global inference defaults.
"model_drift_guard": True,
# Default model for cron jobs (WHAT model runs). Fire-time resolution: per-job pin >
# cron.model > model.default. When set, unpinned jobs follow it deliberately and the drift
# guard does not engage for the model axis. "" = fall through to model.default.
# cron.model > the job's creation-time snapshot > model.default. An unpinned job keeps
# running on the model it was created under when model.default later changes; cron.model
# is the way to move the whole fleet at once. "" = fall through.
"model": "",
# Inference provider paired with cron.model (NOT the scheduler provider below). "" = resolve
# from global config.
@@ -2340,7 +2337,7 @@ DEFAULT_CONFIG = {
# Extra ports detection probes for an external llama-server (besides 8080).
"detect_ports": [],
},
"_config_version": 41, # Config schema version - bump this when adding new required fields
"_config_version": 42, # Config schema version - bump this when adding new required fields
}
+10
View File
@@ -627,6 +627,16 @@ MIGRATIONS: Tuple[Tuple[int, Callable[[Dict[str, Any], bool], None]], ...] = (
message=" ✓ Model catalog now refreshes every 20 minutes (model_catalog.ttl_minutes)",
extra_guard=lambda raw: "ttl_minutes" not in raw)),
(41, _migrate_to_41),
# 41 → 42: cron.model_drift_guard is gone. Unpinned jobs now run on their creation snapshot
# instead of failing closed when the global model changes, so the toggle has nothing to gate.
(42, functools.partial(
_rewrite_key, section="cron", key="model_drift_guard", new=None,
match=lambda cur: cur is not None,
added="removed cron.model_drift_guard",
message=(
" ✓ Removed cron.model_drift_guard — unpinned cron jobs now keep running on the "
"model/provider they were created under when the global default changes, instead "
"of being skipped. Pin a job or set cron.model to move it."))),
)