fix(cron): unpinned jobs run on their creation-snapshot model instead of failing closed
A global model/provider change must never stop a cron job. The #44585 guard raised [drift_skip] for every unpinned job whose provider_snapshot / model_snapshot no longer matched the live global default, so one `hermes model` switch silently killed whole fleets (reported by fastfinge, nitinthewiz, Dr-ilies; 13 of 60 jobs on the project lead's box after claude-fable-5 -> claude-fable-5.1). The snapshot is now the job's effective pin: _load_cron_job_config prefers job['model_snapshot'] over the global default and _resolve_job_runtime passes job['provider_snapshot'] as `requested` when neither a per-job pin nor a cron.model / cron.model_provider fleet default covers the axis. One INFO line per differing axis tells the operator what the job is running on and how to move it. Jobs without a snapshot (legacy records) still follow the global default; the existing fallback chain still handles a snapshot provider that fails to resolve. Both goals of #44585 hold: no silent inherit of a paid default (the job runs on what it was created under) and no outage. Owner decision (Teknium): "main agent model changing should not stop crons from executing, ever". Removed as unreachable: _check_model_drift, DRIFT_SKIP markers, the drift_alerted alert-once bit (mark_drift_alerted + the _record_run_outcome pop), the drift special-cases in _compose_run_delivery and _summarize_cron_failure_for_delivery, cron_model_drift_guard_enabled and the cron.model_drift_guard config key (v42 migration drops it from existing configs). The PLUGIN-COMPAT clear_drift_alerted block is untouched (scheduled revert). The `hermes config set model.default` notice and the Desktop model-change toast are reworded from "will fail closed / will be skipped" to "keep running on the model they were created under"; the impact payload drops guard_enabled (all six desktop locales updated).
This commit is contained in:
@@ -1564,8 +1564,9 @@ export const ar = defineLocale({
|
||||
cron: {
|
||||
close: 'إغلاق',
|
||||
modelImpact: {
|
||||
title: 'تحتاج المهام المجدولة إلى المراجعة',
|
||||
message: count => `سيتم تخطي ${count} من المهام المجدولة حتى تراجع إعدادات النموذج الخاصة بها.`,
|
||||
title: 'تبقى المهام المجدولة على نموذجها الأصلي',
|
||||
message: count =>
|
||||
`${count} من المهام المجدولة غير المثبتة ستواصل العمل على النموذج الذي أُنشئت به. ثبّتها أو اضبط cron.model لنقلها.`,
|
||||
detailMore: (names, remaining) => `${names} و${remaining} أخرى`,
|
||||
review: 'مراجعة المهام المجدولة',
|
||||
saveFailed: 'لم يحفظ Hermes تغيير النموذج هذا.',
|
||||
|
||||
@@ -2174,9 +2174,9 @@ export const en: Translations = {
|
||||
title: 'Scheduled jobs',
|
||||
count: count => `${count} ${count === 1 ? 'job' : 'jobs'}`,
|
||||
modelImpact: {
|
||||
title: 'Scheduled jobs need review',
|
||||
title: 'Scheduled jobs stay on their original model',
|
||||
message: count =>
|
||||
`${count} scheduled ${count === 1 ? 'job' : 'jobs'} will be skipped until you review their model settings.`,
|
||||
`${count} unpinned scheduled ${count === 1 ? 'job keeps' : 'jobs keep'} running on the model ${count === 1 ? 'it was' : 'they were'} created under. Pin ${count === 1 ? 'it' : 'them'} or set cron.model to move ${count === 1 ? 'it' : 'them'}.`,
|
||||
detailMore: (names, remaining) => `${names} and ${remaining} more`,
|
||||
review: 'Review scheduled jobs',
|
||||
saveFailed: 'Hermes did not save that model change.',
|
||||
|
||||
@@ -1846,8 +1846,9 @@ export const ja = defineLocale({
|
||||
title: 'スケジュール済みジョブ',
|
||||
count: count => `${count} 件のジョブ`,
|
||||
modelImpact: {
|
||||
title: 'スケジュール済みジョブの確認が必要です',
|
||||
message: count => `モデル設定を確認するまで、${count} 件のスケジュール済みジョブがスキップされます。`,
|
||||
title: 'スケジュール済みジョブは元のモデルで実行されます',
|
||||
message: count =>
|
||||
`ピン留めされていない ${count} 件のスケジュール済みジョブは、作成時のモデルで引き続き実行されます。移行するにはピン留めするか cron.model を設定してください。`,
|
||||
detailMore: (names, remaining) => `${names}、ほか ${remaining} 件`,
|
||||
review: 'スケジュール済みジョブを確認',
|
||||
saveFailed: 'Hermes はモデルの変更を保存しませんでした。',
|
||||
|
||||
@@ -2212,9 +2212,9 @@ export const ru = defineLocale({
|
||||
title: 'Запланированные задачи',
|
||||
count: count => `${count} ${RU_PLURAL(count, 'задача', 'задачи', 'задач')}`,
|
||||
modelImpact: {
|
||||
title: 'Запланированные задачи требуют проверки',
|
||||
title: 'Запланированные задачи остаются на исходной модели',
|
||||
message: count =>
|
||||
`Будет пропущено ${count} ${RU_NOUN(count, 'задача', 'задачи', 'задач')} до тех пор, пока вы не проверите их настройки модели.`,
|
||||
`${count} незакреплённых запланированных задач продолжат работать на модели, с которой были созданы. Закрепите их или задайте cron.model, чтобы перевести.`,
|
||||
detailMore: (names, remaining) => `${names} и ещё ${remaining}`,
|
||||
review: 'Проверить запланированные задачи',
|
||||
saveFailed: 'Hermes не сохранил это изменение модели.'
|
||||
|
||||
@@ -1773,8 +1773,9 @@ export const zhHant = defineLocale({
|
||||
title: '排程工作',
|
||||
count: count => `${count} 個工作`,
|
||||
modelImpact: {
|
||||
title: '排程工作需要檢查',
|
||||
message: count => `在您檢查模型設定之前,${count} 個排程工作將被略過。`,
|
||||
title: '排程工作將繼續使用原模型',
|
||||
message: count =>
|
||||
`${count} 個未固定的排程工作將繼續使用建立時的模型執行。固定它們或設定 cron.model 以遷移。`,
|
||||
detailMore: (names, remaining) => `${names},以及另外 ${remaining} 個`,
|
||||
review: '檢查排程工作',
|
||||
saveFailed: 'Hermes 未儲存該模型變更。',
|
||||
|
||||
@@ -2340,8 +2340,9 @@ export const zh: Translations = {
|
||||
title: '定时任务',
|
||||
count: count => `${count} 个任务`,
|
||||
modelImpact: {
|
||||
title: '定时任务需要检查',
|
||||
message: count => `在您检查模型设置之前,${count} 个定时任务将被跳过。`,
|
||||
title: '定时任务将继续使用原模型',
|
||||
message: count =>
|
||||
`${count} 个未固定的定时任务将继续使用创建时的模型运行。固定它们或设置 cron.model 以迁移。`,
|
||||
detailMore: (names, remaining) => `${names},以及另外 ${remaining} 个`,
|
||||
review: '检查定时任务',
|
||||
saveFailed: 'Hermes 未保存该模型更改。',
|
||||
|
||||
@@ -74,7 +74,6 @@ export function parseCronModelImpact(value: unknown): CronModelImpact | null {
|
||||
|
||||
if (
|
||||
typeof impact.available !== 'boolean' ||
|
||||
typeof impact.guard_enabled !== 'boolean' ||
|
||||
!Number.isSafeInteger(impact.affected_count) ||
|
||||
(impact.affected_count ?? -1) < 0 ||
|
||||
typeof impact.truncated !== 'boolean' ||
|
||||
@@ -121,15 +120,17 @@ function publishImpact(impact: CronModelImpact, profile: string, connection: str
|
||||
return
|
||||
}
|
||||
|
||||
if (!impact.guard_enabled || impact.affected_count === 0) {
|
||||
if (impact.affected_count === 0) {
|
||||
dismissNotification(CRON_MODEL_IMPACT_NOTIFICATION_ID)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
// Informational: these jobs keep running on the model they were created under; nothing is
|
||||
// skipped. The action is a read-only review so the user can pin or move them deliberately.
|
||||
notify({
|
||||
id: CRON_MODEL_IMPACT_NOTIFICATION_ID,
|
||||
kind: 'warning',
|
||||
kind: 'info',
|
||||
title: translateNow('cron.modelImpact.title'),
|
||||
message: translateNow('cron.modelImpact.message', impact.affected_count),
|
||||
detail: detailFor(impact),
|
||||
|
||||
@@ -1494,7 +1494,6 @@ export interface CronModelImpactJob {
|
||||
|
||||
export interface CronModelImpact {
|
||||
available: boolean
|
||||
guard_enabled: boolean
|
||||
affected_count: number
|
||||
truncated: boolean
|
||||
jobs: CronModelImpactJob[]
|
||||
|
||||
@@ -2408,7 +2408,7 @@ def save_config_value(key_path: str, value: any) -> bool:
|
||||
os.chmod(config_path, 0o600)
|
||||
except (OSError, NotImplementedError):
|
||||
pass
|
||||
# Same fail-closed cron drift warning as `hermes config set` for every model switch.
|
||||
# Same unpinned-cron notice as `hermes config set` for every model switch.
|
||||
from hermes_cli.config import warn_unpinned_cron_jobs_after_model_config_change
|
||||
|
||||
warn_unpinned_cron_jobs_after_model_config_change(key_path, value)
|
||||
|
||||
+6
-13
@@ -1494,7 +1494,7 @@ def _normalize_workdir(workdir: Optional[str]) -> Optional[str]:
|
||||
|
||||
def _resolve_default_model_snapshot() -> Optional[str]:
|
||||
"""Default model resolved as the ticker's ``run_job`` does, so unpinned jobs can snapshot it and
|
||||
detect a later swap. ``None`` on missing config or failure (fail-open: "no snapshot")."""
|
||||
keep running on it after a later swap. ``None`` on missing config or failure ("no snapshot")."""
|
||||
try:
|
||||
from hermes_cli.config import _expand_env_vars, read_user_config_raw
|
||||
|
||||
@@ -1599,8 +1599,9 @@ _UPDATE_FIELD_NORMALIZERS: Dict[str, Callable[[Any], Any]] = {
|
||||
def _compute_provider_model_snapshots(
|
||||
*, provider: Any, model: Any, base_url: Any, no_agent: Any,
|
||||
) -> Tuple[Optional[str], Optional[str]]:
|
||||
"""Snapshot unpinned provider/model resolution so a later global switch fails closed at fire
|
||||
time instead of silently changing spend. Pinned axes and no-agent jobs carry no snapshot."""
|
||||
"""Snapshot unpinned provider/model resolution: the scheduler runs the job on this snapshot after
|
||||
a later global switch instead of silently changing spend. Pinned axes and no-agent jobs carry no
|
||||
snapshot."""
|
||||
normalized_provider = _normalize_job_optional_text(provider)
|
||||
normalized_model = _normalize_job_optional_text(model)
|
||||
normalized_base_url = _normalize_base_url(base_url)
|
||||
@@ -2108,13 +2109,11 @@ def remove_job(job_id: str) -> bool:
|
||||
|
||||
def _set_alert_flag(job_id: str, field: str, value: bool) -> bool:
|
||||
"""Set/clear a persisted alert-dedup marker (alert exactly once until the condition heals;
|
||||
survives restarts) and return the PRIOR value. Fields: ``preflight_alerted``,
|
||||
``drift_alerted``.
|
||||
survives restarts) and return the PRIOR value. Field: ``preflight_alerted`` (blocked config).
|
||||
|
||||
The marker records that the operator was already alerted about this job's condition, so the scheduler
|
||||
alerts exactly once and stays silent on subsequent ticks until the condition heals (same alert-once
|
||||
shape as the dead-pin auto-pause in #73506). Fields: ``preflight_alerted`` (blocked config, T1-26) and
|
||||
``drift_alerted`` (#44585 drift-guard skip).
|
||||
shape as the dead-pin auto-pause in #73506).
|
||||
"""
|
||||
def apply(jobs, _i, job):
|
||||
prior = bool(job.get(field))
|
||||
@@ -2139,11 +2138,6 @@ def clear_preflight_alerted(job_id: str) -> None:
|
||||
_set_alert_flag(job_id, "preflight_alerted", False)
|
||||
|
||||
|
||||
def mark_drift_alerted(job_id: str) -> bool:
|
||||
"""Mark the job as drift-alerted; return True if it already was."""
|
||||
return _set_alert_flag(job_id, "drift_alerted", True)
|
||||
|
||||
|
||||
def note_fire_forward_failure(job_id: str, detail: str) -> bool:
|
||||
"""Durably record (as ``last_fire_error``) that a scheduled fire could not be handed to the
|
||||
runner — written by the dashboard fire webhook when the loopback forward fails. Without it
|
||||
@@ -2175,7 +2169,6 @@ def _record_run_outcome(
|
||||
# Healthy run: drop the alert-once dedup markers so a FUTURE break re-alerts, and clear
|
||||
# the forward-failure stamp so it only describes CURRENT auto-fire health.
|
||||
job.pop("preflight_alerted", None)
|
||||
job.pop("drift_alerted", None)
|
||||
job.pop("last_fire_error", None)
|
||||
job["failure_streak"] = 0
|
||||
else:
|
||||
|
||||
+41
-141
@@ -38,8 +38,7 @@ sys.path.insert(0, str(Path(__file__).parent.parent))
|
||||
from hermes_constants import get_hermes_home
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
from hermes_cli.config import (
|
||||
_expand_env_vars, cron_model_drift_axes, cron_model_drift_guard_enabled, load_config,
|
||||
resolve_cron_model_drift_defaults)
|
||||
_expand_env_vars, load_config, resolve_cron_model_drift_defaults)
|
||||
from hermes_cli.fallback_config import get_fallback_chain
|
||||
from hermes_time import now as _hermes_now
|
||||
from agent.interrupt_compat import request_hard_interrupt
|
||||
@@ -220,24 +219,6 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
|
||||
text = (error or "unknown error").strip()
|
||||
lower = text.lower()
|
||||
|
||||
if "skipped to prevent unintended spend: global inference config drifted" in lower:
|
||||
if "finite one-shot job is consumed" in lower:
|
||||
remediation = (
|
||||
"This finite one-shot is consumed; create a new one-shot job at "
|
||||
"a future time with an explicit provider and model."
|
||||
)
|
||||
else:
|
||||
job_id = job.get("id") or "<job_id>"
|
||||
remediation = (
|
||||
"On the host running Hermes, pin it explicitly: "
|
||||
f"`hermes cron edit {job_id} --provider <provider> "
|
||||
"--model <model>`."
|
||||
)
|
||||
return (
|
||||
f"⚠️ Cron '{job_name}' skipped before inference to prevent "
|
||||
f"unintended spend. {remediation}"
|
||||
)
|
||||
|
||||
# no_agent jobs never reach a model, so provider errors are structurally impossible for them.
|
||||
# Gate on job MODE before substring matching, or a script's own wording ("timed out", "429")
|
||||
# would blame the wrong subsystem; the generic cleaner below reports what actually happened.
|
||||
@@ -1354,10 +1335,26 @@ class _CronJobConfig:
|
||||
cron_default_provider: str
|
||||
|
||||
|
||||
def _snapshot_pin(job: dict, axis: str, current: str, job_id: str) -> str:
|
||||
"""The creation snapshot is an unpinned axis's effective pin: return it, logging once when it
|
||||
differs from *current* (the live global default); ``""`` for legacy jobs without one, which keep
|
||||
following the global default. A global model/provider change must never stop a cron job; a job
|
||||
keeps running on what it was created under until the operator pins it or sets a cron.* fleet
|
||||
default (#44585)."""
|
||||
snapshot = str(job.get(f"{axis}_snapshot") or "").strip()
|
||||
if snapshot and current and snapshot.lower() != current.lower():
|
||||
logger.info(
|
||||
"Job '%s': running on creation-snapshot %s %r (global default is now %r); "
|
||||
"`hermes cron edit %s --%s <value>` or cron.%s in config.yaml moves it.",
|
||||
job_id, axis, snapshot, current, job_id, axis,
|
||||
"model" if axis == "model" else "model_provider")
|
||||
return snapshot
|
||||
|
||||
|
||||
def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConfig:
|
||||
"""Load config.yaml and resolve the run's model: per-job override > cron.model (fleet default) >
|
||||
HERMES_MODEL > config ``model:``. Re-read every tick (no cache) so ``hermes cron edit --model``
|
||||
applies next tick. An axis resolved from cron.model/model_provider is explicit (no drift guard)."""
|
||||
creation snapshot > HERMES_MODEL > config ``model:``. Re-read every tick (no cache) so
|
||||
``hermes cron edit --model`` applies next tick."""
|
||||
model = job.get("model") or os.getenv("HERMES_MODEL") or ""
|
||||
_cron_default_provider = ""
|
||||
_cfg: dict = {}
|
||||
@@ -1383,10 +1380,8 @@ def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConf
|
||||
if _cron_default_model:
|
||||
model = _cron_default_model
|
||||
else:
|
||||
# Shared with Desktop's impact summary so both compare against the same model.
|
||||
_, _global_model = resolve_cron_model_drift_defaults(_cfg)
|
||||
if _global_model:
|
||||
model = _global_model
|
||||
model = _snapshot_pin(job, "model", _global_model, job_id) or _global_model or model
|
||||
except Exception as e:
|
||||
logger.warning("Job '%s': failed to load config.yaml, using defaults: %s", job_id, e)
|
||||
|
||||
@@ -1489,42 +1484,34 @@ def _preflight_or_block(job: dict, job_id: str, job_name: str, cfg: dict) -> Opt
|
||||
return False, blocked_doc, "", f"{marker} {_pf_reason}"
|
||||
|
||||
|
||||
def _resolve_job_runtime(
|
||||
job: dict, job_id: str, jc: _CronJobConfig,
|
||||
) -> tuple[dict, str, Optional[str]]:
|
||||
def _resolve_job_runtime(job: dict, job_id: str, jc: _CronJobConfig) -> tuple[dict, str]:
|
||||
"""Resolve the runtime, walking the fallback chain on auth/transient-network errors. Returns
|
||||
``(runtime, model, primary_provider_for_drift)``; provider+model swap atomically (never swap
|
||||
only the provider while keeping a paid primary model)."""
|
||||
``(runtime, model)``; provider+model swap atomically (never swap only the provider while keeping
|
||||
a paid primary model). Provider precedence: per-job pin > cron.model_provider > creation
|
||||
snapshot > persisted global config."""
|
||||
from hermes_cli.runtime_provider import (
|
||||
resolve_runtime_provider, format_runtime_provider_error)
|
||||
from hermes_cli.auth import AuthError
|
||||
|
||||
model = jc.model
|
||||
configured_provider_for_drift = (
|
||||
str(jc.model_cfg.get("provider") or "").strip().lower()
|
||||
if isinstance(jc.model_cfg, dict)
|
||||
else ""
|
||||
)
|
||||
primary_provider_for_drift = (
|
||||
str(job.get("provider") or "").strip().lower()
|
||||
or configured_provider_for_drift
|
||||
or None
|
||||
)
|
||||
requested = job.get("provider") or jc.cron_default_provider or None
|
||||
if not requested:
|
||||
global_provider = (
|
||||
str(jc.model_cfg.get("provider") or "").strip() if isinstance(jc.model_cfg, dict) else "")
|
||||
# None (not the config provider) keeps the legacy no-snapshot path resolving from persisted
|
||||
# config exactly as before.
|
||||
requested = _snapshot_pin(job, "provider", global_provider, job_id) or None
|
||||
try:
|
||||
# Do NOT pass HERMES_INFERENCE_PROVIDER as `requested`: it would override persisted config
|
||||
# and resurrect stale providers for unpinned jobs.
|
||||
runtime_kwargs = {
|
||||
"requested": job.get("provider") or jc.cron_default_provider or None,
|
||||
"requested": requested,
|
||||
# api_mode must derive from the model actually run, not the stale persisted default.
|
||||
"target_model": model,
|
||||
}
|
||||
if job.get("base_url"):
|
||||
runtime_kwargs["explicit_base_url"] = job.get("base_url")
|
||||
runtime = resolve_runtime_provider(**runtime_kwargs)
|
||||
primary_provider_for_drift = (
|
||||
str(runtime.get("provider") or "").strip().lower() or primary_provider_for_drift
|
||||
)
|
||||
return runtime, model, primary_provider_for_drift
|
||||
return resolve_runtime_provider(**runtime_kwargs), model
|
||||
except Exception as resolve_exc:
|
||||
# Walk the fallback chain on AuthError AND transient network/DNS failures (e.g. during
|
||||
# OAuth refresh); anything else re-raises.
|
||||
@@ -1533,10 +1520,6 @@ def _resolve_job_runtime(
|
||||
if not (is_auth or is_transient_net):
|
||||
raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
|
||||
|
||||
primary_provider_for_drift = (
|
||||
str(getattr(resolve_exc, "provider", "") or "").strip().lower()
|
||||
or primary_provider_for_drift
|
||||
)
|
||||
logger.warning(
|
||||
"Job '%s': primary provider resolve failed (%s: %s), trying fallback",
|
||||
job_id, "auth" if is_auth else "transient network", resolve_exc)
|
||||
@@ -1560,82 +1543,12 @@ def _resolve_job_runtime(
|
||||
logger.info(
|
||||
"Job '%s': fallback resolved to %s model %s",
|
||||
job_id, runtime.get("provider"), fb_model)
|
||||
return runtime, fb_model, primary_provider_for_drift
|
||||
return runtime, fb_model
|
||||
except Exception as fb_exc:
|
||||
logger.debug("Job '%s': fallback %s failed: %s", job_id, fb_provider, fb_exc)
|
||||
raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
|
||||
|
||||
|
||||
def _check_model_drift(
|
||||
job: dict, job_id: str, cfg: dict, runtime: dict,
|
||||
primary_provider_for_drift: Optional[str], primary_model_for_drift: str,
|
||||
) -> None:
|
||||
"""Fail-closed provider/model drift guard; raises RuntimeError (with drift marker) on drift.
|
||||
An unpinned job follows the global default, which may have switched to a paid provider/model:
|
||||
each unpinned axis whose creation snapshot (job["<axis>_snapshot"]) now resolves differently
|
||||
skips the run and alerts to pin. No snapshot, pinned axes, or the cron.model fleet default
|
||||
never count as drift."""
|
||||
if not cron_model_drift_guard_enabled(cfg):
|
||||
return
|
||||
_current_provider = str(
|
||||
primary_provider_for_drift or runtime.get("provider") or ""
|
||||
).strip().lower()
|
||||
_current_model = str(primary_model_for_drift or "").strip().lower()
|
||||
_drift: list[str] = []
|
||||
for _axis in cron_model_drift_axes(
|
||||
job, current_provider=_current_provider, current_model=_current_model, config=cfg):
|
||||
_snapshot = str(job.get(f"{_axis}_snapshot") or "").strip().lower()
|
||||
_current = _current_provider if _axis == "provider" else _current_model
|
||||
_drift.append(f"{_axis} '{_snapshot}' -> '{_current}'")
|
||||
if not _drift:
|
||||
return
|
||||
_changes = "; ".join(_drift)
|
||||
# A finite one-shot is consumed by this attempt, so "edit the job" is a dead end for it.
|
||||
# Lifecycle-aware remediation (#72056, @sashmatash): a finite one-shot is consumed by this attempted
|
||||
# dispatch — telling an operator to edit a spent job is a dead end. Recurring and repeatable jobs get
|
||||
# the pin command instead.
|
||||
_repeat = job.get("repeat") if isinstance(job.get("repeat"), dict) else {}
|
||||
_finite_oneshot = (
|
||||
isinstance(job.get("schedule"), dict)
|
||||
and job["schedule"].get("kind") == "once"
|
||||
and _repeat.get("times") == 1
|
||||
)
|
||||
if _finite_oneshot:
|
||||
_remediation = (
|
||||
"This finite one-shot job is consumed by this attempted run; "
|
||||
"create a new one-shot job at a future time with an explicit provider and model."
|
||||
)
|
||||
else:
|
||||
_remediation = (
|
||||
"To run on the new config, on the host running Hermes pin it explicitly: "
|
||||
f"`hermes cron edit {job_id} --provider <provider> "
|
||||
"--model <model>` (or pin the original values to keep them)."
|
||||
)
|
||||
logger.warning(
|
||||
"Job '%s': SKIPPED — global inference config drifted since "
|
||||
"creation (%s) and this job is unpinned. Skipped to prevent unintended spend. %s",
|
||||
job_id, _changes, _remediation)
|
||||
# Alert-once via drift_alerted bit (silent marker suppresses delivery); a successful run
|
||||
# clears it and re-arms the alert.
|
||||
# Alert-once (#73506 shape): persist the drift_alerted bit so only the FIRST drifted tick delivers;
|
||||
# run_one_job suppresses delivery on the silent marker. mark_job_run clears the bit when a run succeeds
|
||||
# (drift healed), re-arming the alert.
|
||||
_drift_already_alerted = False
|
||||
with contextlib.suppress(Exception):
|
||||
from cron.jobs import mark_drift_alerted
|
||||
|
||||
_drift_already_alerted = mark_drift_alerted(job_id)
|
||||
_drift_marker = DRIFT_SKIP_SILENT_MARKER if _drift_already_alerted else DRIFT_SKIP_MARKER
|
||||
raise RuntimeError(
|
||||
f"{_drift_marker} Skipped to prevent unintended spend: global "
|
||||
f"inference config drifted since this job was created "
|
||||
f"({_changes}), and this job is unpinned. No inference call "
|
||||
f"was made. {_remediation} "
|
||||
f"This alert is sent once; the job stays skipped until the "
|
||||
f"config is pinned or restored. See #44585."
|
||||
)
|
||||
|
||||
|
||||
def _load_credential_pool(runtime: dict, job_id: str):
|
||||
runtime_provider = str(runtime.get("provider") or "").strip().lower()
|
||||
if not runtime_provider:
|
||||
@@ -2203,7 +2116,7 @@ class _CronAgentSetup:
|
||||
|
||||
def _resolve_cron_agent_setup(job: dict, job_id: str, job_name: str, jc) -> _CronAgentSetup:
|
||||
"""Resolve model/runtime/reasoning/pool for the run, in the original gate order: exfil guard ->
|
||||
preflight (may block) -> runtime -> drift check -> fallback chain -> credential pool -> MCP."""
|
||||
preflight (may block) -> runtime (+ fallback chain) -> credential pool -> MCP."""
|
||||
_cfg = jc.cfg
|
||||
setup = _CronAgentSetup(model=jc.model)
|
||||
setup.prefill_messages = _load_prefill_messages(_cfg, job_id)
|
||||
@@ -2223,13 +2136,10 @@ def _resolve_cron_agent_setup(job: dict, job_id: str, job_name: str, jc) -> _Cro
|
||||
if setup.blocked is not None:
|
||||
return setup
|
||||
|
||||
primary_model_for_drift = setup.model
|
||||
setup.runtime, setup.model, primary_provider_for_drift = _resolve_job_runtime(job, job_id, jc)
|
||||
setup.runtime, setup.model = _resolve_job_runtime(job, job_id, jc)
|
||||
setup.reasoning_config = _resolve_job_reasoning_config(
|
||||
job, _cfg if isinstance(_cfg, dict) else {}, str(setup.model)
|
||||
)
|
||||
_check_model_drift(
|
||||
job, job_id, _cfg, setup.runtime, primary_provider_for_drift, primary_model_for_drift)
|
||||
setup.fallback_model = get_fallback_chain(_cfg) or None
|
||||
setup.credential_pool = _load_credential_pool(setup.runtime, job_id)
|
||||
# MCP servers must be registered before AIAgent is constructed.
|
||||
@@ -2622,11 +2532,9 @@ def _compose_run_delivery(
|
||||
silent_alert, incident_acked, failure_incident_id)``; ``silent_alert``: an alert-once marker
|
||||
says the operator was already told, deliver nothing."""
|
||||
err = str(error) if error else ""
|
||||
# Failed jobs always deliver, except blocked-config / drift-skip runs, which alert exactly ONCE.
|
||||
# Failed jobs always deliver, except blocked-config runs, which alert exactly ONCE.
|
||||
blocked_config_silent = BLOCKED_CONFIG_SILENT_MARKER in err
|
||||
blocked_config = blocked_config_silent or BLOCKED_CONFIG_MARKER in err
|
||||
drift_skip_silent = DRIFT_SKIP_SILENT_MARKER in err
|
||||
drift_skip = drift_skip_silent or DRIFT_SKIP_MARKER in err
|
||||
incident_acked = False
|
||||
failure_incident_id = None
|
||||
if blocked_config and not success:
|
||||
@@ -2646,20 +2554,13 @@ def _compose_run_delivery(
|
||||
incident_acked, failure_incident_id = _upsert_incident_for_failure(
|
||||
job, error or "", output_file=output_file
|
||||
)
|
||||
if incident_acked and not drift_skip:
|
||||
if incident_acked:
|
||||
deliver_content = ""
|
||||
else:
|
||||
deliver_content = (
|
||||
_summarize_cron_failure_for_delivery(job, error) + _failure_streak_nudge(job)
|
||||
)
|
||||
if drift_skip:
|
||||
# Deliver the guard's message intact (summarizer truncation would eat the remediation
|
||||
# command). NOT gated on incident ack: acks silence failure pings, not drift alerts.
|
||||
_drift_text = re.sub(r"\[drift_skip[^\]]*\]\s*", "", err).strip()
|
||||
deliver_content = f"⚠️ Cron '{job.get('name') or job['id']}' skipped: {_drift_text}"
|
||||
return (
|
||||
deliver_content, blocked_config, blocked_config_silent or drift_skip_silent,
|
||||
incident_acked, failure_incident_id)
|
||||
return deliver_content, blocked_config, blocked_config_silent, incident_acked, failure_incident_id
|
||||
|
||||
|
||||
class _FireClaimLostDuringSideEffect(Exception):
|
||||
@@ -3876,9 +3777,8 @@ from cron.scheduler_prompt import ( # noqa: E402
|
||||
_block_and_pause_job, _build_job_prompt, _guard_job_credential_exfil, _parse_wake_gate,
|
||||
)
|
||||
from cron.scheduler_preflight import ( # noqa: E402
|
||||
BLOCKED_CONFIG_MARKER, BLOCKED_CONFIG_SILENT_MARKER, DRIFT_SKIP_MARKER,
|
||||
DRIFT_SKIP_SILENT_MARKER, _cron_preflight_enabled, _is_transient_provider_resolve_error,
|
||||
_preflight_job_config,
|
||||
BLOCKED_CONFIG_MARKER, BLOCKED_CONFIG_SILENT_MARKER, _cron_preflight_enabled,
|
||||
_is_transient_provider_resolve_error, _preflight_job_config,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -21,12 +21,6 @@ logger = logging.getLogger("cron.scheduler")
|
||||
# alert-once dedup. ``:silent`` = already alerted on a previous tick — do not deliver again.
|
||||
BLOCKED_CONFIG_MARKER = "[blocked_config]"
|
||||
BLOCKED_CONFIG_SILENT_MARKER = "[blocked_config:silent]"
|
||||
# Drift-guard skip: same contract (drift_alerted bit on the job record).
|
||||
# Same alert-once contract as blocked_config: run_one_job keys off it to record last_status and the
|
||||
# ``:silent`` variant means "already alerted on a previous tick — do not deliver again" (the drift_alerted
|
||||
# bit on the job record, #73506 shape).
|
||||
DRIFT_SKIP_MARKER = "[drift_skip]"
|
||||
DRIFT_SKIP_SILENT_MARKER = "[drift_skip:silent]"
|
||||
|
||||
_TRANSIENT_NET_EXC_NAMES = frozenset({
|
||||
"ConnectError", "ConnectTimeout", "ReadTimeout", "WriteTimeout", "PoolTimeout", "NetworkError",
|
||||
@@ -316,8 +310,8 @@ def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]:
|
||||
so the caller refuses BEFORE building agent machinery or burning an LLM call. Every check fails
|
||||
open — preflight blocks only on an affirmative misconfiguration verdict.
|
||||
|
||||
Same fail-before-spend spirit as the #44585 drift guard and the fail-loud-on-hidden-tools direction in
|
||||
#27948; alert dedup follows the alert-once pattern from the dead-pin auto-pause (#73506).
|
||||
Same fail-before-spend spirit as the fail-loud-on-hidden-tools direction in #27948; alert dedup
|
||||
follows the alert-once pattern from the dead-pin auto-pause (#73506).
|
||||
"""
|
||||
for name, check in (
|
||||
("provider_key", lambda: _preflight_check_provider_key(job, cfg)),
|
||||
|
||||
+23
-39
@@ -2949,7 +2949,7 @@ def edit_config():
|
||||
subprocess.run([editor, str(config_path)])
|
||||
|
||||
|
||||
# ---- Cron model-drift guard helpers ----
|
||||
# ---- Cron model-drift helpers: which unpinned jobs stay on their creation snapshot ----
|
||||
|
||||
_CRON_DRIFT_AXIS_BY_KEY = {
|
||||
"model": "model", "model.default": "model", "model.model": "model", "model.name": "model",
|
||||
@@ -2957,7 +2957,7 @@ _CRON_DRIFT_AXIS_BY_KEY = {
|
||||
|
||||
|
||||
def _cron_model_drift_axis_for_config_key(key: str) -> Optional[str]:
|
||||
"""Return the cron drift guard axis affected by a config key, if any."""
|
||||
"""Return the cron inference axis affected by a config key, if any."""
|
||||
return _CRON_DRIFT_AXIS_BY_KEY.get(str(key or "").strip().lower())
|
||||
|
||||
|
||||
@@ -2972,15 +2972,6 @@ def _cron_section(config: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
||||
return cron_config if isinstance(cron_config, dict) else None
|
||||
|
||||
|
||||
def cron_model_drift_guard_enabled(config: Optional[Dict[str, Any]] = None) -> bool:
|
||||
"""Whether cron must fail closed on unpinned inference drift.
|
||||
Only the literal YAML boolean ``false`` disables this spend-safety guard; missing, malformed,
|
||||
or non-boolean values stay fail-closed. With *config* omitted the merged config is loaded so
|
||||
CLI warnings honor the same user/managed setting as the scheduler."""
|
||||
cron_config = _cron_section(config)
|
||||
return cron_config is None or cron_config.get("model_drift_guard", True) is not False
|
||||
|
||||
|
||||
_CRON_MODEL_IMPACT_JOB_LIMIT = 50
|
||||
_CRON_MODEL_IMPACT_ID_LIMIT = 256
|
||||
_CRON_MODEL_IMPACT_NAME_LIMIT = 120
|
||||
@@ -2996,7 +2987,7 @@ def resolve_cron_model_drift_defaults(
|
||||
"""Resolve the global ``(provider, model)`` cron compares against snapshots.
|
||||
Mirrors the scheduler's precedence: a truthy configured model wins over ``HERMES_MODEL``; the
|
||||
environment is only a fallback. Per-job and cron fleet defaults are handled by the caller
|
||||
because they suppress a drift axis rather than changing the global assignment."""
|
||||
because they cover an axis rather than changing the global assignment."""
|
||||
env = os.environ if environ is None else environ
|
||||
provider = ""
|
||||
model_config = config.get("model") if isinstance(config, dict) else None
|
||||
@@ -3010,15 +3001,16 @@ def resolve_cron_model_drift_defaults(
|
||||
def cron_model_drift_axes(
|
||||
job: Any, *, current_provider: Any = "", current_model: Any = "", config: Any = None
|
||||
) -> List[str]:
|
||||
"""Return the unpinned axes that the fail-closed cron guard would block."""
|
||||
if not isinstance(job, dict) or not cron_model_drift_guard_enabled(config):
|
||||
"""Return the unpinned axes on which *job* will keep running on its creation snapshot rather
|
||||
than the new global assignment (the scheduler treats the snapshot as the effective pin)."""
|
||||
if not isinstance(job, dict):
|
||||
return []
|
||||
|
||||
current = {
|
||||
"provider": _model_assignment_text(current_provider).lower(),
|
||||
"model": _model_assignment_text(current_model).lower()}
|
||||
# A cron.model / cron.model_provider fleet default covers its axis: that axis no longer follows
|
||||
# the global assignment at fire time, so the guard never engages and a warning would be false.
|
||||
# A cron.model / cron.model_provider fleet default covers its axis: that axis never reads the
|
||||
# snapshot at fire time, so reporting it would be false.
|
||||
fleet = _cron_section(config) or {}
|
||||
drifted: List[str] = []
|
||||
for axis, fleet_key in (("provider", "model_provider"), ("model", "model")):
|
||||
@@ -3050,35 +3042,28 @@ def _cron_impact_job_name(value: Any, job_id: str) -> str:
|
||||
return f"Job {job_id}"[:_CRON_MODEL_IMPACT_NAME_LIMIT].rstrip()
|
||||
|
||||
|
||||
def _cron_model_impact_result(available: bool, guard_enabled: bool) -> Dict[str, Any]:
|
||||
return {
|
||||
"available": available,
|
||||
"guard_enabled": guard_enabled,
|
||||
"affected_count": 0,
|
||||
"truncated": False,
|
||||
"jobs": []}
|
||||
def _cron_model_impact_result(available: bool) -> Dict[str, Any]:
|
||||
return {"available": available, "affected_count": 0, "truncated": False, "jobs": []}
|
||||
|
||||
|
||||
def build_cron_model_impact(
|
||||
*, current_provider: Any = "", current_model: Any = "", config: Any = None, jobs: Any = None
|
||||
) -> Dict[str, Any]:
|
||||
"""Build a bounded, profile-local summary of jobs blocked by model drift.
|
||||
Job-store inspection is best effort: the model assignment has already succeeded when Desktop
|
||||
requests this, so an unreadable store is reported as unavailable rather than failing."""
|
||||
guard_enabled = cron_model_drift_guard_enabled(config)
|
||||
"""Build a bounded, profile-local summary of unpinned jobs that stay on their creation snapshot
|
||||
after a global model/provider change. Job-store inspection is best effort: the model assignment
|
||||
has already succeeded when Desktop requests this, so an unreadable store is reported as
|
||||
unavailable rather than failing."""
|
||||
if jobs is None:
|
||||
try:
|
||||
from cron.jobs import load_jobs
|
||||
|
||||
jobs = load_jobs()
|
||||
except Exception:
|
||||
return _cron_model_impact_result(False, guard_enabled)
|
||||
return _cron_model_impact_result(False)
|
||||
if not isinstance(jobs, list):
|
||||
return _cron_model_impact_result(False, guard_enabled)
|
||||
return _cron_model_impact_result(False)
|
||||
|
||||
result = _cron_model_impact_result(True, guard_enabled)
|
||||
if not guard_enabled:
|
||||
return result
|
||||
result = _cron_model_impact_result(True)
|
||||
|
||||
from cron.jobs import is_job_runnable
|
||||
|
||||
@@ -3107,7 +3092,7 @@ def build_cron_model_impact(
|
||||
|
||||
def warn_unpinned_cron_jobs_after_model_config_change(
|
||||
key: str, value: Any, config: Optional[Dict[str, Any]] = None) -> None:
|
||||
"""Warn when a global model/provider change will trip cron's drift guard."""
|
||||
"""Tell the operator which unpinned cron jobs a global model/provider change does NOT move."""
|
||||
axis = _cron_model_drift_axis_for_config_key(key)
|
||||
if axis is None:
|
||||
return
|
||||
@@ -3122,13 +3107,12 @@ def warn_unpinned_cron_jobs_after_model_config_change(
|
||||
if affected <= 0:
|
||||
return
|
||||
|
||||
noun, verb = ("job", "has") if affected == 1 else ("jobs", "have")
|
||||
noun, verb = ("job", "keeps") if affected == 1 else ("jobs", "keep")
|
||||
print(
|
||||
f"⚠️ {affected} enabled unpinned cron {noun} {verb} stored "
|
||||
f"{axis}_snapshot values that differ from the new global {axis}. "
|
||||
"They will fail closed on their next run instead of silently using the changed "
|
||||
"model/provider. Inspect with `hermes cron list`, then pin the intended values with "
|
||||
"`hermes cron edit <job_id> --provider <provider> --model <model>`.")
|
||||
f"ℹ️ {affected} unpinned cron {noun} {verb} running on the {axis} it was created under "
|
||||
f"(its {axis}_snapshot), not the new global {axis}. To move it, pin it with "
|
||||
"`hermes cron edit <job_id> --provider <provider> --model <model>` or set a fleet default "
|
||||
"with `hermes config set cron.model <model>`.")
|
||||
|
||||
|
||||
def _default_value_for_key(dotted_key: str):
|
||||
|
||||
@@ -1628,13 +1628,10 @@ DEFAULT_CONFIG = {
|
||||
# platforms are configured. Failure -> last_status=blocked_config, ONE alert, no LLM call.
|
||||
# False = fail during the run instead.
|
||||
"preflight": True,
|
||||
# Fail closed when an unpinned job's current global model/provider differs from its
|
||||
# creation-time snapshot, so unattended jobs never silently inherit a paid default. False
|
||||
# only when jobs should track changing global inference defaults.
|
||||
"model_drift_guard": True,
|
||||
# Default model for cron jobs (WHAT model runs). Fire-time resolution: per-job pin >
|
||||
# cron.model > model.default. When set, unpinned jobs follow it deliberately and the drift
|
||||
# guard does not engage for the model axis. "" = fall through to model.default.
|
||||
# cron.model > the job's creation-time snapshot > model.default. An unpinned job keeps
|
||||
# running on the model it was created under when model.default later changes; cron.model
|
||||
# is the way to move the whole fleet at once. "" = fall through.
|
||||
"model": "",
|
||||
# Inference provider paired with cron.model (NOT the scheduler provider below). "" = resolve
|
||||
# from global config.
|
||||
@@ -2340,7 +2337,7 @@ DEFAULT_CONFIG = {
|
||||
# Extra ports detection probes for an external llama-server (besides 8080).
|
||||
"detect_ports": [],
|
||||
},
|
||||
"_config_version": 41, # Config schema version - bump this when adding new required fields
|
||||
"_config_version": 42, # Config schema version - bump this when adding new required fields
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -627,6 +627,16 @@ MIGRATIONS: Tuple[Tuple[int, Callable[[Dict[str, Any], bool], None]], ...] = (
|
||||
message=" ✓ Model catalog now refreshes every 20 minutes (model_catalog.ttl_minutes)",
|
||||
extra_guard=lambda raw: "ttl_minutes" not in raw)),
|
||||
(41, _migrate_to_41),
|
||||
# 41 → 42: cron.model_drift_guard is gone. Unpinned jobs now run on their creation snapshot
|
||||
# instead of failing closed when the global model changes, so the toggle has nothing to gate.
|
||||
(42, functools.partial(
|
||||
_rewrite_key, section="cron", key="model_drift_guard", new=None,
|
||||
match=lambda cur: cur is not None,
|
||||
added="removed cron.model_drift_guard",
|
||||
message=(
|
||||
" ✓ Removed cron.model_drift_guard — unpinned cron jobs now keep running on the "
|
||||
"model/provider they were created under when the global default changes, instead "
|
||||
"of being skipped. Pin a job or set cron.model to move it."))),
|
||||
)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user