chore: anchor fresh-start history to upstream
This commit is contained in:
+19
-137
@@ -81,18 +81,16 @@ WATCH_GLOBAL_COOLDOWN_SECONDS = 30
|
||||
# Under a systemd gateway with MemoryMax, local background commands inherit the gateway's
|
||||
# cgroup, so a memory-heavy executor can get the ENTIRE gateway killed by systemd-oomd;
|
||||
# ``systemd-run --user --scope`` gives the worker its own transient cgroup. Usability is
|
||||
# probed and cached for a bounded TTL (binary present but user D-Bus absent in system services/containers).
|
||||
# probed once (binary present but user D-Bus absent in system services/containers).
|
||||
# A memory-heavy executor (Codex, tests, Node) can push the whole cgroup past MemoryMax and trigger
|
||||
# systemd-oomd to kill the ENTIRE gateway — taking down the messaging control plane and silently losing the
|
||||
# active turn. We probe whether ``systemd-run --user --scope`` is actually usable (the binary can
|
||||
# active turn. We probe *once* whether ``systemd-run --user --scope`` is actually usable (the binary can
|
||||
# exist on the PATH while the user D-Bus session is unavailable — common for system services and
|
||||
# containers), and cache the verdict for a bounded TTL. See #70716.
|
||||
# containers), and cache the result for the process lifetime. See #70716.
|
||||
_SYSTEMD_SCOPE_AVAILABLE: Optional[bool] = None
|
||||
_SYSTEMD_SCOPE_PROBE_LOCK = threading.Lock()
|
||||
_SYSTEMD_SCOPE_PROBED_AT = 0.0
|
||||
# Both verdicts expire: the user bus can vanish after a True (session logout without linger,
|
||||
# #110803) and reappear after a False (linger enabled later, #104893).
|
||||
_SYSTEMD_SCOPE_PROBE_TTL_SECONDS = 60.0
|
||||
_SYSTEMD_SCOPE_FAILURE_TTL_SECONDS = 60.0
|
||||
_MIN_WORKER_MEMORY_MAX_BYTES = 64 * 1024 * 1024
|
||||
_DEFAULT_WORKER_MEMORY_MAX_BYTES = 1024 * 1024 * 1024
|
||||
_WORKER_MEMORY_MAX_CAP_BYTES = 4 * 1024 * 1024 * 1024
|
||||
@@ -206,11 +204,12 @@ def systemd_user_bus_env(base_env: Optional[Dict[str, str]] = None) -> Dict[str,
|
||||
|
||||
|
||||
def _systemd_scope_cached() -> Optional[bool]:
|
||||
"""Cached probe verdict, or None when a (re)probe is due."""
|
||||
if _SYSTEMD_SCOPE_AVAILABLE is None:
|
||||
return None
|
||||
stale = time.monotonic() - _SYSTEMD_SCOPE_PROBED_AT >= _SYSTEMD_SCOPE_PROBE_TTL_SECONDS
|
||||
return None if stale else _SYSTEMD_SCOPE_AVAILABLE
|
||||
"""Cached probe verdict, or None when a (re)probe is due. True is permanent; False
|
||||
expires after ``_SYSTEMD_SCOPE_FAILURE_TTL_SECONDS`` so a D-Bus blip isn't sticky."""
|
||||
if _SYSTEMD_SCOPE_AVAILABLE is True:
|
||||
return True
|
||||
stale = time.monotonic() - _SYSTEMD_SCOPE_PROBED_AT >= _SYSTEMD_SCOPE_FAILURE_TTL_SECONDS
|
||||
return None if _SYSTEMD_SCOPE_AVAILABLE is None or stale else False
|
||||
|
||||
|
||||
def _systemd_run_user_scope_available() -> bool:
|
||||
@@ -325,27 +324,6 @@ class GatewayChildDispatch(NamedTuple):
|
||||
argv: List[str]
|
||||
|
||||
|
||||
def scoped_spawn_lost_user_bus(spawn_env: Dict[str, str]) -> bool:
|
||||
"""After a ``systemd-run --user --scope`` wrapper exits before its child could start: True
|
||||
when the user bus is gone (:func:`systemd_user_bus_env` derives nothing), in which case the
|
||||
cached True verdict is replaced so the next dispatch re-probes and degrades instead of
|
||||
consuming another occurrence on the same dead wrapper (#110803).
|
||||
|
||||
*spawn_env* is the environment the wrapper was launched with: re-deriving from it (minus the
|
||||
bus address it carried) honours a configured ``XDG_RUNTIME_DIR`` exactly as the spawn did, so
|
||||
an unrelated wrapper exit on a host whose bus lives outside ``/run/user/<uid>`` is not
|
||||
misread as a lost bus."""
|
||||
global _SYSTEMD_SCOPE_AVAILABLE, _SYSTEMD_SCOPE_PROBED_AT
|
||||
base_env = dict(spawn_env)
|
||||
base_env.pop("DBUS_SESSION_BUS_ADDRESS", None)
|
||||
if "DBUS_SESSION_BUS_ADDRESS" in systemd_user_bus_env(base_env):
|
||||
return False
|
||||
with _SYSTEMD_SCOPE_PROBE_LOCK:
|
||||
_SYSTEMD_SCOPE_AVAILABLE = False
|
||||
_SYSTEMD_SCOPE_PROBED_AT = time.monotonic()
|
||||
return True
|
||||
|
||||
|
||||
def restart_safe_gateway_child_argv(
|
||||
command: List[str], *, unit_suffix: str, require_restart_safe_scope: bool,
|
||||
) -> GatewayChildDispatch:
|
||||
@@ -1183,16 +1161,11 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
finally:
|
||||
self._finish_reader(
|
||||
session, decoder, _append_chunk, "Process",
|
||||
session.process.wait, lambda: session.process.returncode)
|
||||
lambda: session.process.wait(timeout=5), lambda: session.process.returncode)
|
||||
|
||||
def _finish_reader(self, session, decoder, append, label, wait, exit_code) -> None:
|
||||
"""Reader-thread teardown: flush the decoder (a truncated multibyte tail becomes
|
||||
one U+FFFD instead of vanishing), reap the child (no zombies), record the exit.
|
||||
|
||||
A process may close stdout long before it exits. The reader owns a dedicated
|
||||
daemon thread, so it must keep waiting rather than publish a false completion
|
||||
and discard the only ``Popen`` handle that can reap the child.
|
||||
"""
|
||||
one U+FFFD instead of vanishing), reap the child (no zombies), record the exit."""
|
||||
with suppress(Exception):
|
||||
tail = decoder.decode(b"", final=True)
|
||||
if tail:
|
||||
@@ -1200,12 +1173,7 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
try:
|
||||
wait()
|
||||
except Exception as e:
|
||||
# A PTY child reaped by isalive() already has its exitstatus; only an
|
||||
# unknown status must stay tracked for later reconciliation.
|
||||
if exit_code() is None:
|
||||
logger.warning("%s wait failed; leaving process tracked: %s", label, e)
|
||||
return
|
||||
logger.warning("%s wait failed; recording known exit status: %s", label, e)
|
||||
logger.debug("%s wait timed out or failed: %s", label, e)
|
||||
self._finish_exited(session, exit_code())
|
||||
|
||||
@staticmethod
|
||||
@@ -1304,30 +1272,12 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
# PTY reads can split a multibyte UTF-8 character across chunks just like pipe reads — hold partial
|
||||
# sequences until the rest arrives. (Ported from openclaw/openclaw#112325.)
|
||||
decoder = codecs.getincrementaldecoder("utf-8")(errors="replace")
|
||||
# Programs in a PTY can block waiting for replies to device-status / window-size /
|
||||
# cursor-position / DEC private-mode queries. Answer the bounded set and strip the
|
||||
# queries from captured output. POSIX only: Windows ConPTY is a real console host that
|
||||
# answers itself (and pywinpty yields str chunks, not bytes).
|
||||
responder = None
|
||||
if not _IS_WINDOWS:
|
||||
from tools.pty_query_responder import PtyQueryResponder
|
||||
responder = PtyQueryResponder(rows=30, cols=120)
|
||||
try:
|
||||
while pty.isalive():
|
||||
try:
|
||||
chunk = pty.read(4096)
|
||||
if chunk:
|
||||
# ptyprocess returns bytes; pywinpty returns str
|
||||
if responder is not None and isinstance(chunk, bytes):
|
||||
chunk, replies = responder.process(chunk)
|
||||
if replies:
|
||||
try:
|
||||
pty.write(replies)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"PTY query response write failed",
|
||||
exc_info=True,
|
||||
)
|
||||
text = chunk if isinstance(chunk, str) else decoder.decode(chunk)
|
||||
if text:
|
||||
self._ingest_output(session, text)
|
||||
@@ -1335,11 +1285,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
break
|
||||
except Exception as e:
|
||||
logger.debug("PTY stdout reader ended: %s", e)
|
||||
if responder is not None:
|
||||
# A query prefix split across the final reads is plain output after all.
|
||||
tail = decoder.decode(responder.flush())
|
||||
if tail:
|
||||
self._ingest_output(session, tail)
|
||||
self._finish_reader(
|
||||
session, decoder, lambda t: self._ingest_output(session, t), "PTY",
|
||||
pty.wait, lambda: pty.exitstatus if hasattr(pty, 'exitstatus') else -1)
|
||||
@@ -1355,11 +1300,10 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
session.mark_exited(exit_code)
|
||||
self._move_to_finished(session)
|
||||
|
||||
def _move_to_finished(self, session: ProcessSession) -> bool:
|
||||
def _move_to_finished(self, session: ProcessSession):
|
||||
"""Move a session from running to finished.
|
||||
Idempotent: kill_process() and the reader thread can both call this; only
|
||||
the FIRST move enqueues the completion notification, so no duplicates.
|
||||
Returns True when this call is the one that persisted the session."""
|
||||
the FIRST move enqueues the completion notification, so no duplicates."""
|
||||
with self._lock:
|
||||
was_running = session.id in self._running
|
||||
if was_running:
|
||||
@@ -1368,16 +1312,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
save_completed_result(session)
|
||||
self._running.pop(session.id)
|
||||
self._finished[session.id] = session
|
||||
# Release the retained Popen/PTY handles now: otherwise every
|
||||
# finished-but-unpruned session keeps its stdout pipe (or PTY master)
|
||||
# FD open until FINISHED_TTL_SECONDS elapses, and heavy background
|
||||
# churn can exhaust the gateway's FD limit. On the reader-thread path
|
||||
# the pipe is already at EOF; on the kill/reconcile paths the reader
|
||||
# may still be draining — its next read raises on the closed stream
|
||||
# and the loop exits, dropping at most the unread tail of a process
|
||||
# that was just killed. poll()/wait()/read_log() serve from the
|
||||
# buffered ``output_buffer``, never from the pipe.
|
||||
self._release_finished_handles(session)
|
||||
self._write_checkpoint()
|
||||
if was_running and session.notify_on_complete:
|
||||
notification = {
|
||||
@@ -1397,7 +1331,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
_redact_process_result(notification)
|
||||
self.completion_queue.put(notification)
|
||||
session._completion_event.set()
|
||||
return was_running
|
||||
|
||||
@staticmethod
|
||||
def _exit_fields(session: ProcessSession) -> dict:
|
||||
@@ -1407,28 +1340,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
"termination_source": session.termination_source,
|
||||
}
|
||||
|
||||
def _release_finished_handles(self, session: ProcessSession):
|
||||
"""Close a finished session's OS handles (Popen pipes / PTY master).
|
||||
|
||||
Best-effort and idempotent: the session may have no local Popen (env
|
||||
backends, detached recovery), or the handles may already be closed by
|
||||
the reader loop / kill path. Closing a Popen's stream objects does not
|
||||
kill anything — the child has already exited — it only releases the
|
||||
parent's pipe FDs, which is exactly the retained-resource leak.
|
||||
"""
|
||||
proc = session.process
|
||||
if proc is not None:
|
||||
for stream in (proc.stdout, proc.stderr, proc.stdin):
|
||||
if stream is not None:
|
||||
with suppress(OSError, ValueError): # a stdin flush can hit EPIPE
|
||||
stream.close()
|
||||
if session._pty is not None:
|
||||
# ptyprocess/pywinpty close() is idempotent (``closed`` flag) and
|
||||
# closes the master fd exactly once; it raises only if the child
|
||||
# ignores SIGKILL, which we don't want to surface on the finish path.
|
||||
with suppress(Exception):
|
||||
session._pty.close()
|
||||
|
||||
# ----- Query Methods -----
|
||||
|
||||
def is_completion_consumed(self, session_id: str) -> bool:
|
||||
@@ -1472,11 +1383,8 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
timeout = self._oneshot_completion_wait_seconds()
|
||||
result: dict = {"waited": [], "completed": [], "timed_out": []}
|
||||
with self._lock:
|
||||
# `_finished` too: `_move_to_finished` pops a session from `_running` and enqueues its completion
|
||||
# only after releasing handles and writing the checkpoint. A parent whose turn ends inside that
|
||||
# window would otherwise see nothing pending, drain nothing and exit without the follow-up turn.
|
||||
pending = [
|
||||
s for store in (self._running, self._finished) for s in store.values()
|
||||
s for s in self._running.values()
|
||||
if s.notify_on_complete and not s._completion_event.is_set() and (task_id is None or s.task_id == task_id)
|
||||
]
|
||||
if not pending or timeout <= 0:
|
||||
@@ -1765,11 +1673,10 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
return result
|
||||
|
||||
def wait(self, session_id: str, timeout: int = None) -> dict:
|
||||
"""Block until the process exits, the timeout elapses, the user interrupts, or a
|
||||
mid-turn user message (steer/redirect → ``request_yield``) releases the wait.
|
||||
"""Block until the process exits, the timeout elapses, or the user interrupts.
|
||||
``timeout`` defaults to (and is clamped by) TERMINAL_TIMEOUT. Returns a dict
|
||||
with status exited|timeout|interrupted|not_found|error and an output snapshot."""
|
||||
from tools.interrupt import consume_yield as _consume_yield, is_interrupted as _is_interrupted
|
||||
from tools.interrupt import is_interrupted as _is_interrupted
|
||||
|
||||
try:
|
||||
max_timeout = int(os.getenv("TERMINAL_TIMEOUT", "180"))
|
||||
@@ -1801,16 +1708,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
result = {
|
||||
"status": "interrupted", "command": session.command, "output": _output_tail(session, 1000),
|
||||
"note": "User sent a new message -- wait interrupted"}
|
||||
elif _consume_yield(threading.current_thread().ident):
|
||||
# A steer/redirect landed mid-turn: redirect() asks tool workers to YIELD so
|
||||
# the user's message is delivered instead of parked behind this wait. The
|
||||
# process is untouched and still notify-tracked; the model should read the
|
||||
# steer text and respond, not re-issue the wait (kimi-code#3697 class).
|
||||
result = {
|
||||
"status": "interrupted", "command": session.command, "output": _output_tail(session, 1000),
|
||||
"process_running": True,
|
||||
"note": ("User sent a new message -- wait released; the process is still "
|
||||
"running and you will be notified on exit. Respond to the user now.")}
|
||||
if result is not None:
|
||||
if timeout_note:
|
||||
result["timeout_note"] = timeout_note
|
||||
@@ -1889,12 +1786,7 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
session.exit_code = -15 # SIGTERM
|
||||
session.completion_reason = "killed"
|
||||
session.termination_source = source
|
||||
# The reader thread can finalise the session while the signal path
|
||||
# blocks in the SIGKILL grace window: its ``save_completed_result``
|
||||
# then persists this kill as a plain ``exited``. Re-write the receipt
|
||||
# so the durable record matches what the caller was told.
|
||||
if not self._move_to_finished(session):
|
||||
save_completed_result(session)
|
||||
self._move_to_finished(session)
|
||||
self._write_checkpoint()
|
||||
return {
|
||||
"status": "killed", "session_id": session.id, "completion_reason": session.completion_reason,
|
||||
@@ -2047,9 +1939,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
]
|
||||
result = []
|
||||
for s in all_sessions:
|
||||
# List-only refreshes must observe child exit even while descendants
|
||||
# keep the capture pipe open; retain the existing completion owner.
|
||||
self._reconcile_local_exit(s)
|
||||
entry = {
|
||||
"session_id": s.id,
|
||||
"command": s.command[:200],
|
||||
@@ -2175,13 +2064,6 @@ class ProcessRegistry(ProcessCheckpointMixin):
|
||||
if over_cap and (survivors := [sid for sid in self._finished if sid not in expired]):
|
||||
expired.append(min(survivors, key=lambda sid: self._finished[sid].started_at))
|
||||
for sid in expired:
|
||||
# Belt-and-suspenders handle release: sessions normally arrive in
|
||||
# _finished via _move_to_finished(), which already released their
|
||||
# Popen/PTY handles — but any session inserted into _finished
|
||||
# directly (defensive paths, historical checkpoints) would
|
||||
# otherwise carry its OS handles to the grave unreleased. The
|
||||
# release is idempotent, so double-closing is safe.
|
||||
self._release_finished_handles(self._finished[sid])
|
||||
del self._finished[sid]
|
||||
# Belt-and-suspenders against module-lifetime growth: forget consumed /
|
||||
# poll-observed marks for any session no longer tracked at all.
|
||||
|
||||
Reference in New Issue
Block a user