[verified] fix(cron): preserve active runs across gateway restart
This commit is contained in:
committed by
kshitij
parent
68cbe484a4
commit
3373e97693
+276
-1
@@ -3203,6 +3203,16 @@ def _deliver_result(
|
||||
logger.warning("Job '%s': %s", job["id"], msg)
|
||||
return msg
|
||||
|
||||
# Restart-safe workers intentionally have no live gateway adapter objects.
|
||||
# Hand the send back through a durable queue so the current or replacement
|
||||
# gateway performs it with relay/E2EE parity. The execution id is the
|
||||
# idempotency key; the queue never retries an uncertain claimed send.
|
||||
external_execution = os.environ.get("_HERMES_CRON_EXTERNAL_WORKER", "")
|
||||
if external_execution and adapters is None:
|
||||
from cron.delivery_queue import enqueue_and_wait
|
||||
|
||||
return enqueue_and_wait(external_execution, job, content)
|
||||
|
||||
from tools.send_message_tool import _send_to_platform
|
||||
from gateway.config import load_gateway_config, Platform
|
||||
|
||||
@@ -4118,6 +4128,20 @@ def _deliver_result(
|
||||
return None
|
||||
|
||||
|
||||
def drain_delivery_queue(adapters, loop) -> int:
|
||||
"""Send queued worker results through this gateway's live adapters."""
|
||||
from cron.delivery_queue import drain
|
||||
|
||||
return drain(
|
||||
lambda queued_job, queued_content: _deliver_result(
|
||||
queued_job,
|
||||
queued_content,
|
||||
adapters=adapters,
|
||||
loop=loop,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
_DEFAULT_SCRIPT_TIMEOUT = 3600 # seconds (1 hour)
|
||||
# Backward-compatible module override used by tests and emergency monkeypatches.
|
||||
_SCRIPT_TIMEOUT = _DEFAULT_SCRIPT_TIMEOUT
|
||||
@@ -7368,6 +7392,34 @@ def run_one_job(
|
||||
run cooperatively — agent interruption AND script process-tree kill —
|
||||
through the single fenced completion path.
|
||||
"""
|
||||
# Every gateway path (built-in scheduler, external providers, and direct
|
||||
# API fires) crosses this seam. Ensure the detached worker has a durable
|
||||
# attempt to adopt before any launch can occur.
|
||||
if adapters is not None and not job.get("execution_id"):
|
||||
execution = create_execution(job["id"], source="direct")
|
||||
job["execution_id"] = execution["id"]
|
||||
|
||||
if adapters is not None:
|
||||
try:
|
||||
if _launch_external_cron_worker(job):
|
||||
return True
|
||||
except Exception as handoff_error:
|
||||
error = f"Restart-safe cron worker dispatch failed: {handoff_error}"
|
||||
logger.error("Job '%s': %s", job["id"], error)
|
||||
claim = job.get("fire_claim")
|
||||
owner = str(claim.get("by") or "") if isinstance(claim, dict) else ""
|
||||
try:
|
||||
mark_job_run(
|
||||
job["id"],
|
||||
False,
|
||||
error,
|
||||
**({"expected_fire_owner": owner} if owner else {}),
|
||||
)
|
||||
finally:
|
||||
execution_id = job.get("execution_id")
|
||||
if execution_id:
|
||||
finish_execution(execution_id, success=False, error=error)
|
||||
return True
|
||||
if extra_prompt is None:
|
||||
# A gateway-forwarded manual run (`hermes cron run --prompt` /
|
||||
# cronjob(action='run', prompt=...) on a relay-fronted target) stamps
|
||||
@@ -7491,7 +7543,16 @@ def _run_one_job_body(
|
||||
|
||||
# The attempt is claimed durably before executor/provider dispatch and
|
||||
# becomes running only immediately before the actual run.
|
||||
mark_execution_running(execution_id)
|
||||
# Detached workers atomically transition the attempt to running while
|
||||
# adopting it. In-process paths must win the claimed->running CAS
|
||||
# here before any user script or agent side effect may begin.
|
||||
external_owner = os.environ.get("_HERMES_CRON_EXTERNAL_WORKER") == execution_id
|
||||
if not external_owner and mark_execution_running(execution_id) is None:
|
||||
logger.warning(
|
||||
"Cron job %s lost execution ownership before start; skipping",
|
||||
job["id"],
|
||||
)
|
||||
return True
|
||||
|
||||
# Run and deliver under the profile's secret scope. get_secret() fails
|
||||
# closed outside a scope once profile isolation is active, and cron
|
||||
@@ -7967,6 +8028,210 @@ def _run_one_job_body(
|
||||
reset_terminal_scope(_terminal_scope_token)
|
||||
|
||||
|
||||
def _launch_external_cron_worker(job: dict) -> bool:
|
||||
"""Launch *job* outside a managed gateway cgroup when required.
|
||||
|
||||
Returns ``False`` when the caller is not a managed systemd gateway and the
|
||||
existing in-process path should be used. In managed topology, failure to
|
||||
establish the transient scope raises: falling back would recreate the
|
||||
restart interruption this handoff exists to prevent.
|
||||
"""
|
||||
execution_id = str(job["execution_id"])
|
||||
job_id = str(job["id"])
|
||||
handoff_dir = _get_hermes_home() / "cron" / "external-workers"
|
||||
payload_path = handoff_dir / f"{execution_id}.json"
|
||||
ack_path = handoff_dir / f"{execution_id}.ready"
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"cron.scheduler",
|
||||
"--external-worker-file",
|
||||
str(payload_path),
|
||||
"--ack-file",
|
||||
str(ack_path),
|
||||
]
|
||||
|
||||
from agent.secret_scope import is_multiplex_active
|
||||
from tools.environments.local import build_subprocess_env
|
||||
from tools.process_registry import restart_safe_gateway_child_argv
|
||||
|
||||
multiplex_active = is_multiplex_active()
|
||||
scoped_command = restart_safe_gateway_child_argv(
|
||||
command,
|
||||
unit_suffix=f"cron-{job_id}-exec-{execution_id}",
|
||||
)
|
||||
if scoped_command == command:
|
||||
return False
|
||||
|
||||
_ensure_cron_dir(handoff_dir)
|
||||
try:
|
||||
handoff_dir.chmod(0o700)
|
||||
except OSError:
|
||||
pass
|
||||
fd = os.open(payload_path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as payload_file:
|
||||
json.dump(
|
||||
{
|
||||
"job": job,
|
||||
"profile_home": str(_get_hermes_home().resolve()),
|
||||
"multiplex_active": multiplex_active,
|
||||
},
|
||||
payload_file,
|
||||
)
|
||||
payload_file.flush()
|
||||
os.fsync(payload_file.fileno())
|
||||
except BaseException:
|
||||
payload_path.unlink(missing_ok=True)
|
||||
raise
|
||||
|
||||
worker_env = build_subprocess_env(
|
||||
scrub_secrets=multiplex_active,
|
||||
inherit_profile_home=True,
|
||||
extra={"HERMES_HOME": str(_get_hermes_home().resolve())},
|
||||
)
|
||||
try:
|
||||
process = subprocess.Popen(
|
||||
scoped_command,
|
||||
cwd=str(Path(__file__).resolve().parent.parent),
|
||||
env=worker_env,
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
start_new_session=True,
|
||||
creationflags=windows_hide_flags(),
|
||||
)
|
||||
except BaseException:
|
||||
payload_path.unlink(missing_ok=True)
|
||||
raise
|
||||
|
||||
deadline = time.monotonic() + 5.0
|
||||
while time.monotonic() < deadline:
|
||||
if ack_path.exists():
|
||||
try:
|
||||
acknowledgement = json.loads(ack_path.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
logger.exception(
|
||||
"Cron external worker %s published an unreadable acknowledgement; "
|
||||
"treating handoff as ownership-uncertain",
|
||||
execution_id,
|
||||
)
|
||||
return True
|
||||
finally:
|
||||
ack_path.unlink(missing_ok=True)
|
||||
if acknowledgement.get("execution_id") != execution_id:
|
||||
logger.error(
|
||||
"Cron external worker acknowledgement mismatch for %s; "
|
||||
"treating handoff as ownership-uncertain",
|
||||
execution_id,
|
||||
)
|
||||
return True
|
||||
logger.info(
|
||||
"Cron job '%s' handed to restart-safe worker pid=%s execution=%s",
|
||||
job_id,
|
||||
acknowledgement.get("pid"),
|
||||
execution_id,
|
||||
)
|
||||
return True
|
||||
returncode = process.poll()
|
||||
if returncode is not None:
|
||||
payload_path.unlink(missing_ok=True)
|
||||
raise RuntimeError(
|
||||
f"cron external worker exited before ownership acknowledgement "
|
||||
f"(exit {returncode})"
|
||||
)
|
||||
time.sleep(0.05)
|
||||
|
||||
# The child may have adopted the durable row just before publishing its
|
||||
# acknowledgement. Never fall back to in-process execution on an uncertain
|
||||
# handoff: that could duplicate side effects. The execution owner/dead-owner
|
||||
# recovery ledger remains the authority.
|
||||
logger.warning(
|
||||
"Cron external worker for job '%s' did not acknowledge within 5s; "
|
||||
"leaving the durable execution claim untouched",
|
||||
job_id,
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
def _run_external_worker_payload(payload_path: Path, ack_path: Path) -> bool:
|
||||
"""Adopt and execute one gateway-dispatched cron payload.
|
||||
|
||||
The execution row is created by the gateway before spawn, then transferred
|
||||
here before the ready acknowledgement is published. No side effect runs
|
||||
unless that durable ownership transfer succeeds.
|
||||
"""
|
||||
try:
|
||||
payload = json.loads(payload_path.read_text(encoding="utf-8"))
|
||||
job = payload["job"]
|
||||
profile_home = Path(payload["profile_home"]).resolve()
|
||||
execution_id = str(job["execution_id"])
|
||||
except Exception:
|
||||
logger.exception("Cron external worker could not load payload %s", payload_path)
|
||||
return False
|
||||
finally:
|
||||
try:
|
||||
payload_path.unlink(missing_ok=True)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
from agent.secret_scope import (
|
||||
build_profile_secret_scope,
|
||||
is_multiplex_active,
|
||||
reset_secret_scope,
|
||||
set_multiplex_active,
|
||||
set_secret_scope,
|
||||
)
|
||||
from cron.executions import adopt_claimed_execution
|
||||
from hermes_cli.env_loader import hydrate_profile_secret_sources
|
||||
from hermes_constants import (
|
||||
reset_hermes_home_override,
|
||||
set_hermes_home_override,
|
||||
)
|
||||
|
||||
home_token = set_hermes_home_override(profile_home)
|
||||
previous_multiplex = is_multiplex_active()
|
||||
multiplex_active = bool(payload.get("multiplex_active", False))
|
||||
set_multiplex_active(multiplex_active)
|
||||
hydrate_profile_secret_sources(profile_home)
|
||||
secret_token = set_secret_scope(build_profile_secret_scope(profile_home))
|
||||
try:
|
||||
with use_cron_store(profile_home):
|
||||
if adopt_claimed_execution(execution_id) is None:
|
||||
logger.error(
|
||||
"Cron external worker refused execution %s: durable ownership "
|
||||
"could not be established",
|
||||
execution_id,
|
||||
)
|
||||
return False
|
||||
try:
|
||||
ack_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
fd = os.open(ack_path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as ack_file:
|
||||
json.dump({"pid": os.getpid(), "execution_id": execution_id}, ack_file)
|
||||
ack_file.flush()
|
||||
os.fsync(ack_file.fileno())
|
||||
except Exception:
|
||||
logger.exception(
|
||||
"Cron external worker could not publish ready acknowledgement for %s",
|
||||
execution_id,
|
||||
)
|
||||
return False
|
||||
old_external_execution = os.environ.get("_HERMES_CRON_EXTERNAL_WORKER")
|
||||
os.environ["_HERMES_CRON_EXTERNAL_WORKER"] = execution_id
|
||||
try:
|
||||
return run_one_job(job, adapters=None, loop=None, verbose=False)
|
||||
finally:
|
||||
if old_external_execution is None:
|
||||
os.environ.pop("_HERMES_CRON_EXTERNAL_WORKER", None)
|
||||
else:
|
||||
os.environ["_HERMES_CRON_EXTERNAL_WORKER"] = old_external_execution
|
||||
finally:
|
||||
reset_secret_scope(secret_token)
|
||||
set_multiplex_active(previous_multiplex)
|
||||
reset_hermes_home_override(home_token)
|
||||
|
||||
|
||||
def _notify_provider_jobs_changed() -> None:
|
||||
"""Best-effort: tell the active scheduler provider the job set changed.
|
||||
|
||||
@@ -8582,4 +8847,14 @@ def tick(
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if "--external-worker-file" in sys.argv:
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(add_help=False)
|
||||
parser.add_argument("--external-worker-file", type=Path, required=True)
|
||||
parser.add_argument("--ack-file", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
raise SystemExit(
|
||||
0 if _run_external_worker_payload(args.external_worker_file, args.ack_file) else 1
|
||||
)
|
||||
tick(verbose=True)
|
||||
|
||||
Reference in New Issue
Block a user