refactor(state): split SessionDB into domain mixins and free-function modules; unify SQL boilerplate
hermes_state.py 17,220 -> 6,442 LOC. Behavior-neutral: every moved body is AST-identical to the original, verified per extraction. SessionDB core - _write_sql / _write_rowcount / _read_one / _read_all replace ~120 copies of the `def _do(conn): conn.execute(...)` + `_execute_write(_do)` and `with self._read_ctx() as conn: row = conn.execute(...).fetchone()` shapes. - _set_lineage_column replaces four copies of the recursive compression-lineage UPDATE (archived / pinned / hidden / last_read_at). - _read_session_number unifies the three compression counter readers. - Dead (zero refs repo-wide): restore_rewound, delete_gateway_routing_entries, _is_duplicate_replayed_user_message, SessionPortabilityMixin.get_first_assistant_text. New mixins bound onto SessionDB via the MRO (logger name stays "hermes_state"): hermes_state_messages SessionMessagesMixin 48 methods hermes_state_compression SessionCompressionMixin 30 hermes_state_gateway SessionGatewayMixin 26 hermes_state_maintenance SessionMaintenanceMixin 13 hermes_state_usage SessionUsageMixin 12 hermes_state_titles SessionTitlesMixin 13 hermes_state_telegram SessionTelegramTopicsMixin 11 Origin-internal symbols resolve through a lazy `from hermes_state import ...` inside the few methods that need them (no import cycle). New free-function modules, every name re-imported into hermes_state so `hermes_state.<name>` (and test monkeypatches on it) keep working; intra-module calls to patched helpers go through the lazy origin import: hermes_state_repair repair/backup/preflight (43 defs) hermes_state_wal journal-mode / PRAGMA policy (33 defs) hermes_state_dbfile header probes, zeroed-db quarantine, stats, holders (21 defs) Existing mixins: search — shared FTS MATCH/LIKE builders, unified rebuild status/step/finish engines, state_meta helpers; schema — one legacy/v23 FTS init branch, shared _live_pk_columns, Row/tuple dual access dropped; portability — shared _PREVIEW_RAW_SUBQUERY_SQL and _rich_row; common — single stat_db_file_identity (was 3 copies), AUTO_VACUUM_MIN_FREELIST_RATIO. Docstrings/comments hand-compacted (AST-identical) keeping every invariant, ordering rule, failure mode and WHY. Schema SQL, migration order and PRAGMAs untouched. test_repair_path_has_no_bare_connects repointed to hermes_state_repair.
This commit is contained in:
+119
-300
@@ -1,11 +1,9 @@
|
||||
"""Session listing/rich rows, export, and import (portability) for SessionDB.
|
||||
|
||||
Mixin contract: this is a plain mixin class consumed by
|
||||
``hermes_state.SessionDB``. It defines no ``__init__`` and no state of its
|
||||
own; methods access the host's attributes (``self._conn``, ``self.db_path``,
|
||||
``self._execute_write`` and other SessionDB methods) established by
|
||||
``SessionDB.__init__``. It must never import hermes_state (cycle) — shared
|
||||
module-level constants live in hermes_state_common.
|
||||
Plain mixin consumed by ``hermes_state.SessionDB``: no ``__init__``, no state
|
||||
of its own; methods use host attributes established by ``SessionDB.__init__``.
|
||||
Must never import hermes_state (cycle) — shared constants live in
|
||||
hermes_state_common.
|
||||
"""
|
||||
|
||||
import logging
|
||||
@@ -16,14 +14,12 @@ from typing import Any, Dict, List, Optional
|
||||
from agent.skill_commands import SKILL_SCAFFOLD_SQL_LIKE
|
||||
from hermes_state_common import (
|
||||
SCHEMA_SQL,
|
||||
_PREVIEW_ELIGIBLE_SQL,
|
||||
_PREVIEW_RAW_SELECT,
|
||||
_PREVIEW_RAW_SUBQUERY_SQL,
|
||||
_shape_preview,
|
||||
_sql_session_last_active,
|
||||
)
|
||||
|
||||
# Moved methods logged under the "hermes_state" logger before the split;
|
||||
# keep that logger identity so log filtering/capture behavior is unchanged.
|
||||
# Keep the pre-split logger identity so log filtering/capture is unchanged.
|
||||
logger = logging.getLogger("hermes_state")
|
||||
|
||||
|
||||
@@ -32,9 +28,8 @@ class SessionPortabilityMixin:
|
||||
|
||||
@classmethod
|
||||
def _compact_session_cols(cls) -> str:
|
||||
"""SELECT list for compact_rows: every ``sessions`` column declared in
|
||||
SCHEMA_SQL except prompt storage internals, aliased with the ``s``
|
||||
prefix used by list_sessions_rich/_get_session_rich_row queries."""
|
||||
"""``s.``-prefixed SELECT list of every SCHEMA_SQL ``sessions`` column
|
||||
except prompt storage internals (the compact_rows projection)."""
|
||||
if cls._session_compact_cols_sql is None:
|
||||
declared = cls._parse_schema_columns(SCHEMA_SQL)["sessions"]
|
||||
cls._session_compact_cols_sql = ", ".join(
|
||||
@@ -43,13 +38,19 @@ class SessionPortabilityMixin:
|
||||
)
|
||||
return cls._session_compact_cols_sql
|
||||
|
||||
@classmethod
|
||||
def _rich_row(cls, row) -> Dict[str, Any]:
|
||||
"""Session row dict with ``_preview_raw`` shaped into ``preview``."""
|
||||
s = cls._session_row_dict(row)
|
||||
s["preview"] = _shape_preview(s.pop("_preview_raw", ""))
|
||||
return s
|
||||
|
||||
def distinct_session_cwds(self, include_archived: bool = False) -> List[Dict[str, Any]]:
|
||||
"""Distinct non-empty session cwds with usage stats, for repo discovery.
|
||||
|
||||
Aggregates across ALL session history (not a single page), so the desktop
|
||||
can surface every git repo the user has worked in — not just the repos
|
||||
that happen to be in the currently-loaded recents. Children/branches
|
||||
count: a worktree session is still a real workspace signal.
|
||||
Aggregates across ALL history (not one page) so every repo the user
|
||||
worked in surfaces. Children/branches count: a worktree session is a
|
||||
real workspace signal.
|
||||
"""
|
||||
where = "cwd IS NOT NULL AND TRIM(cwd) != ''"
|
||||
if not include_archived:
|
||||
@@ -77,41 +78,24 @@ class SessionPortabilityMixin:
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""List the run sessions produced by a single cron job, newest first.
|
||||
|
||||
Cron runs are flat, independent sessions whose id is
|
||||
``cron_{job_id}_{timestamp}`` (see ``cron/scheduler.run_job``). They are
|
||||
never compression roots and never branch, so this deliberately skips the
|
||||
``list_sessions_rich`` recursive compression-chain CTE / leading-wildcard
|
||||
``id_query`` path — that path seeds from *every* ``source='cron'`` row in
|
||||
the DB and only filters to one job's runs after the scan, so it scales
|
||||
with the whole cron pile (a heavy history makes the desktop run-history
|
||||
endpoint time out before it eventually populates).
|
||||
Cron runs are flat sessions with id ``cron_{job_id}_{timestamp}``; they
|
||||
never compress or branch, so this skips ``list_sessions_rich``'s
|
||||
compression-chain CTE / leading-wildcard ``id_query`` path, which seeds
|
||||
from EVERY ``source='cron'`` row and scales with the whole cron pile.
|
||||
Instead: a ``[prefix, prefix_hi)`` index range scan on id, filtered to
|
||||
``source='cron'``, so work scales with the requested window.
|
||||
|
||||
Instead this binds to one job with a ``[prefix, prefix_hi)`` range over
|
||||
the id (an index range scan, not a ``%...%`` substring), filters
|
||||
``source='cron'``, and orders by ``started_at DESC``. Work scales with
|
||||
the requested window, not the total cron history.
|
||||
|
||||
Returns the same enriched row shape as ``list_sessions_rich`` (adds
|
||||
``preview`` + ``last_active``) so callers can reuse it.
|
||||
Returns the ``list_sessions_rich`` row shape (``preview`` + ``last_active``).
|
||||
"""
|
||||
prefix = f"cron_{job_id}_"
|
||||
# Half-open upper bound for an index range scan: increment the final
|
||||
# byte of the prefix so the range covers exactly the ids that start
|
||||
# with ``prefix`` and nothing else. ``prefix`` always ends in '_', but
|
||||
# compute it generically rather than hardcoding the successor char.
|
||||
# Half-open upper bound: bump the final byte so the range covers exactly
|
||||
# the ids starting with ``prefix``.
|
||||
prefix_hi = prefix[:-1] + chr(ord(prefix[-1]) + 1)
|
||||
|
||||
query = f"""
|
||||
SELECT s.*,
|
||||
COALESCE(sp.prompt, s.system_prompt) AS _system_prompt_resolved,
|
||||
COALESCE(
|
||||
(SELECT {_PREVIEW_RAW_SELECT}
|
||||
FROM messages m
|
||||
WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL
|
||||
AND {_PREVIEW_ELIGIBLE_SQL}
|
||||
ORDER BY m.timestamp, m.id LIMIT 1),
|
||||
''
|
||||
) AS _preview_raw,
|
||||
{_PREVIEW_RAW_SUBQUERY_SQL},
|
||||
{_sql_session_last_active("s")} AS last_active
|
||||
FROM sessions s
|
||||
LEFT JOIN system_prompts sp ON sp.hash = s.system_prompt_hash
|
||||
@@ -120,27 +104,12 @@ class SessionPortabilityMixin:
|
||||
LIMIT ? OFFSET ?
|
||||
"""
|
||||
with self._lock:
|
||||
cursor = self._conn.execute(query, (prefix, prefix_hi, limit, offset))
|
||||
rows = cursor.fetchall()
|
||||
|
||||
runs: List[Dict[str, Any]] = []
|
||||
for row in rows:
|
||||
s = self._session_row_dict(row)
|
||||
s["preview"] = _shape_preview(s.pop("_preview_raw", ""))
|
||||
runs.append(s)
|
||||
return runs
|
||||
rows = self._conn.execute(query, (prefix, prefix_hi, limit, offset)).fetchall()
|
||||
return [self._rich_row(row) for row in rows]
|
||||
|
||||
def _get_session_rich_row(self, session_id: str, compact_rows: bool = False) -> Optional[Dict[str, Any]]:
|
||||
"""Fetch a single session with the same enriched columns as
|
||||
``list_sessions_rich`` (preview + last_active). Returns None if the
|
||||
session doesn't exist.
|
||||
|
||||
Pass ``compact_rows=True`` to omit the ``system_prompt`` blob (see
|
||||
``list_sessions_rich`` for details).
|
||||
|
||||
Thin wrapper over ``_get_session_rich_rows_batch`` so the enriched
|
||||
SELECT lives in exactly one place.
|
||||
"""
|
||||
"""One session with the ``list_sessions_rich`` enriched columns, or
|
||||
None. ``compact_rows=True`` omits the ``system_prompt`` blob."""
|
||||
return self._get_session_rich_rows_batch(
|
||||
[session_id], compact_rows=compact_rows
|
||||
).get(session_id)
|
||||
@@ -148,23 +117,15 @@ class SessionPortabilityMixin:
|
||||
def _get_session_rich_rows_batch(
|
||||
self, session_ids, compact_rows: bool = False
|
||||
) -> Dict[str, Dict[str, Any]]:
|
||||
"""Fetch multiple sessions with the same enriched columns as
|
||||
``_get_session_rich_row``, in a single query.
|
||||
|
||||
Used by ``list_sessions_rich``'s compression-tip projection to resolve
|
||||
every tip row for a page in one round trip instead of one query per
|
||||
compression-root row. Returns a dict keyed by session id; ids that
|
||||
don't exist are simply absent from the result (same as
|
||||
``_get_session_rich_row`` returning ``None`` for them).
|
||||
"""Enriched rows for many sessions in one query, keyed by id; missing
|
||||
ids are simply absent. Resolves a page of compression tips in one
|
||||
round trip instead of one query per root row.
|
||||
"""
|
||||
ids = [sid for sid in session_ids if sid]
|
||||
if not ids:
|
||||
return {}
|
||||
# Old SQLite builds cap bound variables at 999
|
||||
# (SQLITE_MAX_VARIABLE_NUMBER); large pages (limit=10000 callers
|
||||
# exist) could exceed it. Chunk the IN list so the helper is safe at
|
||||
# any page size — this is the single choke point for the enriched
|
||||
# multi-row fetch, so the bound lives here, not at call sites.
|
||||
# Old SQLite caps bound variables at 999 (SQLITE_MAX_VARIABLE_NUMBER);
|
||||
# limit=10000 callers exist. Chunk here — the single choke point.
|
||||
_CHUNK = 900
|
||||
if len(ids) > _CHUNK:
|
||||
result: Dict[str, Dict[str, Any]] = {}
|
||||
@@ -189,45 +150,26 @@ class SessionPortabilityMixin:
|
||||
)
|
||||
query = f"""
|
||||
SELECT {_sel}{prompt_select},
|
||||
COALESCE(
|
||||
(SELECT {_PREVIEW_RAW_SELECT}
|
||||
FROM messages m
|
||||
WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL
|
||||
AND {_PREVIEW_ELIGIBLE_SQL}
|
||||
ORDER BY m.timestamp, m.id LIMIT 1),
|
||||
''
|
||||
) AS _preview_raw,
|
||||
{_PREVIEW_RAW_SUBQUERY_SQL},
|
||||
{_sql_session_last_active("s")} AS last_active
|
||||
FROM sessions s
|
||||
{prompt_join}
|
||||
WHERE s.id IN ({placeholders})
|
||||
"""
|
||||
with self._lock:
|
||||
cursor = self._conn.execute(query, ids)
|
||||
rows = cursor.fetchall()
|
||||
result: Dict[str, Dict[str, Any]] = {}
|
||||
for row in rows:
|
||||
s = self._session_row_dict(row)
|
||||
s["preview"] = _shape_preview(s.pop("_preview_raw", ""))
|
||||
result[s["id"]] = s
|
||||
return result
|
||||
rows = self._conn.execute(query, ids).fetchall()
|
||||
return {s["id"]: s for s in map(self._rich_row, rows)}
|
||||
|
||||
def get_session_rich_row(self, session_id: str, compact_rows: bool = False) -> Optional[Dict[str, Any]]:
|
||||
"""Public wrapper for :meth:`_get_session_rich_row`.
|
||||
|
||||
Exposes the single-session enriched row (same columns as
|
||||
``list_sessions_rich``: preview + last_active) for callers outside
|
||||
this module, e.g. the web server's session-search hydration.
|
||||
"""
|
||||
"""Public wrapper for :meth:`_get_session_rich_row` (web server hydration)."""
|
||||
return self._get_session_rich_row(session_id, compact_rows=compact_rows)
|
||||
|
||||
def list_skill_scaffolded_sessions(self, limit: int = 200) -> List[Dict[str, Any]]:
|
||||
"""Titled sessions whose first user turn was a ``/skill`` invocation.
|
||||
|
||||
Those titles were generated from the expanded message, which embeds the
|
||||
whole skill body — so they describe the skill rather than the request.
|
||||
Returns ``id``, ``title``, and the full first-turn ``content`` so a
|
||||
caller can re-derive what the user typed. Newest first.
|
||||
Their titles were generated from the expanded skill body, so they
|
||||
describe the skill, not the request. Returns ``id``, ``title`` and the
|
||||
first-turn ``content`` so callers can re-derive what was typed. Newest first.
|
||||
"""
|
||||
with self._lock:
|
||||
rows = self._conn.execute(
|
||||
@@ -248,63 +190,36 @@ class SessionPortabilityMixin:
|
||||
).fetchall()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
def get_first_assistant_text(self, session_id: str) -> str:
|
||||
"""The session's first assistant reply as plain text ('' when none).
|
||||
|
||||
Pairs with :meth:`list_skill_scaffolded_sessions` so a re-title can feed
|
||||
the titler the same (request, reply) shape the live path uses.
|
||||
"""
|
||||
with self._lock:
|
||||
row = self._conn.execute(
|
||||
"SELECT content FROM messages "
|
||||
"WHERE session_id = ? AND role = 'assistant' AND content IS NOT NULL "
|
||||
"ORDER BY timestamp, id LIMIT 1",
|
||||
(session_id,),
|
||||
).fetchone()
|
||||
if not row:
|
||||
return ""
|
||||
decoded = self._decode_content(row["content"])
|
||||
return decoded if isinstance(decoded, str) else ""
|
||||
|
||||
def export_session(self, session_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Export a single session with all its messages as a dict."""
|
||||
session = self.get_session(session_id)
|
||||
if not session:
|
||||
return None
|
||||
messages = self.get_messages(session_id)
|
||||
return {**session, "messages": messages}
|
||||
return {**session, "messages": self.get_messages(session_id)}
|
||||
|
||||
def export_session_lineage(self, session_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Export a compression lineage as one logical session dict."""
|
||||
lineage_ids = self.get_compression_lineage(session_id)
|
||||
if not lineage_ids:
|
||||
return None
|
||||
segments = []
|
||||
for sid in lineage_ids:
|
||||
segment = self.export_session(sid)
|
||||
if segment:
|
||||
segments.append(segment)
|
||||
segments = [seg for seg in map(self.export_session, lineage_ids) if seg]
|
||||
if not segments:
|
||||
return None
|
||||
base = dict(segments[-1])
|
||||
total_messages = sum(len(seg.get("messages") or []) for seg in segments)
|
||||
base["segments"] = segments
|
||||
base["lineage_session_ids"] = [seg["id"] for seg in segments]
|
||||
base["message_count"] = total_messages
|
||||
base["messages"] = [msg for seg in segments for msg in (seg.get("messages") or [])]
|
||||
return base
|
||||
messages = [msg for seg in segments for msg in (seg.get("messages") or [])]
|
||||
return {
|
||||
**segments[-1],
|
||||
"segments": segments,
|
||||
"lineage_session_ids": [seg["id"] for seg in segments],
|
||||
"message_count": len(messages),
|
||||
"messages": messages,
|
||||
}
|
||||
|
||||
def export_all(self, source: str = None) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Export all sessions (with messages) as a list of dicts.
|
||||
Suitable for writing to a JSONL file for backup/analysis.
|
||||
"""
|
||||
sessions = self.search_sessions(source=source, limit=100000)
|
||||
results = []
|
||||
for session in sessions:
|
||||
messages = self.get_messages(session["id"])
|
||||
results.append({**session, "messages": messages})
|
||||
return results
|
||||
"""Export all sessions (with messages) as dicts, e.g. for JSONL backup."""
|
||||
return [
|
||||
{**session, "messages": self.get_messages(session["id"])}
|
||||
for session in self.search_sessions(source=source, limit=100000)
|
||||
]
|
||||
|
||||
def adopt_session_lineage_from(
|
||||
self,
|
||||
@@ -313,52 +228,36 @@ class SessionPortabilityMixin:
|
||||
*,
|
||||
retire_donor: bool = True,
|
||||
) -> Dict[str, Any]:
|
||||
"""Adopt *session_id*'s full compression lineage from *donor_db* into
|
||||
this store.
|
||||
"""Adopt *session_id*'s full compression lineage from *donor_db*.
|
||||
|
||||
The stranded-bot-session heal (#93091 follow-up to #93296): before the
|
||||
desktop routed session RPCs by their target session, a profile bot's
|
||||
turns executed on whichever backend held window focus — usually the
|
||||
default one — so the bot's canonical session rows and messages
|
||||
accumulated in the DEFAULT profile's state.db. Once routing was fixed,
|
||||
the profile backend correctly received the RPCs but had no such
|
||||
session, so the same chat 4001'd for the opposite reason. This method
|
||||
moves the conversation to where routing now looks for it.
|
||||
Stranded-bot-session heal: before the desktop routed session RPCs by
|
||||
target session, a profile bot's rows accumulated in the DEFAULT
|
||||
profile's state.db; this moves the conversation to where routing now
|
||||
looks. Pure composition: ``donor_db.export_session_lineage()`` ->
|
||||
``self.import_sessions()`` — routing/handoff/activity fields reset,
|
||||
already-present ids skipped (idempotent re-adoption).
|
||||
|
||||
Composition of existing primitives (no new import/export machinery):
|
||||
``donor_db.export_session_lineage()`` -> ``self.import_sessions()``.
|
||||
Import semantics apply unchanged: gateway routing, handoff, and live
|
||||
activity fields are reset; already-present ids are skipped
|
||||
(idempotent re-adoption after a partial run).
|
||||
With ``retire_donor`` and a complete adoption, donor rows are ARCHIVED
|
||||
(never deleted) with ``end_reason='adopted_by_profile'``. That
|
||||
end_reason is deliberately NOT in the recoverable set
|
||||
(agent_close/ws_orphan_reap): resurrection must not undo an adoption.
|
||||
|
||||
When ``retire_donor`` is True and at least one segment was imported
|
||||
(or every segment already exists here), the donor rows are ARCHIVED —
|
||||
never deleted — with ``end_reason='adopted_by_profile'`` so the
|
||||
default profile's list stops advertising a conversation that now
|
||||
lives elsewhere, while the bytes stay recoverable. The archive is
|
||||
deliberately NOT in the recoverable set (agent_close/ws_orphan_reap):
|
||||
canonical-lookup resurrection must not undo an adoption.
|
||||
|
||||
Returns the ``import_sessions`` result dict, plus ``adopted`` (bool)
|
||||
and ``donor_retired`` (bool — True only when EVERY segment's
|
||||
retirement actually applied).
|
||||
Returns the ``import_sessions`` dict plus ``adopted`` and
|
||||
``donor_retired`` (True only when EVERY segment's retirement applied).
|
||||
"""
|
||||
payload = donor_db.export_session_lineage(session_id)
|
||||
if not payload:
|
||||
return {
|
||||
"ok": False,
|
||||
"adopted": False,
|
||||
"donor_retired": False,
|
||||
"ok": False, "adopted": False, "donor_retired": False,
|
||||
"error": f"session {session_id!r} not found in donor store",
|
||||
}
|
||||
|
||||
segments = payload.get("segments") or [payload]
|
||||
|
||||
# Divergence guard: a segment we are about to SKIP (already present
|
||||
# here) may have kept accumulating messages in the donor store after
|
||||
# a partial earlier adoption. Retiring it would strand those newer
|
||||
# messages behind a non-recoverable archive. Compare counts up front
|
||||
# and refuse to retire (still adopt/import) when the donor is ahead.
|
||||
# Divergence guard: a segment we will SKIP (already here) may have kept
|
||||
# growing in the donor after a partial adoption; retiring it would strand
|
||||
# those messages behind a non-recoverable archive. Still import, but
|
||||
# refuse to retire when the donor is ahead.
|
||||
donor_ahead = False
|
||||
for seg in segments:
|
||||
seg_id = seg.get("id")
|
||||
@@ -394,15 +293,11 @@ class SessionPortabilityMixin:
|
||||
if not seg_id:
|
||||
continue
|
||||
try:
|
||||
# TOCTOU close-out: the guard above compared EXPORT-TIME
|
||||
# counts, but another backend can append donor messages
|
||||
# between export and this loop. Re-read both stores right
|
||||
# before stamping; a donor-ahead signal here skips the
|
||||
# stamp so growth never lands behind a non-recoverable
|
||||
# archive. (Count comparison cannot see equal-count
|
||||
# CONTENT divergence — e.g. a donor rewind+rewrite; that
|
||||
# residual case is accepted: bytes stay in the donor
|
||||
# store either way, only reachability differs.)
|
||||
# TOCTOU close-out: the guard above used EXPORT-TIME counts;
|
||||
# re-read both stores right before stamping so donor growth
|
||||
# never lands behind a non-recoverable archive. (Equal-count
|
||||
# CONTENT divergence is accepted: bytes stay in the donor
|
||||
# either way, only reachability differs.)
|
||||
donor_now = len(donor_db.get_messages(seg_id))
|
||||
local_now = len(self.get_messages(seg_id))
|
||||
if donor_now > local_now:
|
||||
@@ -414,17 +309,15 @@ class SessionPortabilityMixin:
|
||||
seg_id, donor_now, local_now,
|
||||
)
|
||||
continue
|
||||
# First end_reason wins in end_session(); reopen first so
|
||||
# the adoption boundary is stamped even on ended segments
|
||||
# (e.g. 'compression' parents).
|
||||
# First end_reason wins in end_session(); reopen so the
|
||||
# adoption boundary is stamped even on ended segments.
|
||||
donor_db.reopen_session(seg_id)
|
||||
donor_db.end_session(seg_id, "adopted_by_profile")
|
||||
donor_db.set_session_archived(seg_id, True)
|
||||
except Exception:
|
||||
# Best-effort by design: a retirement failure must not
|
||||
# fail the adoption (the profile copy is already whole;
|
||||
# a later resume retries retirement idempotently). But
|
||||
# never claim success we didn't have.
|
||||
# Best-effort: a retirement failure must not fail the adoption
|
||||
# (a later resume retries idempotently) — but never claim
|
||||
# success we didn't have.
|
||||
retire_ok = False
|
||||
logger.warning(
|
||||
"failed to retire donor segment %s after adoption",
|
||||
@@ -507,21 +400,16 @@ class SessionPortabilityMixin:
|
||||
def import_sessions(self, sessions: List[Dict[str, Any]]) -> Dict[str, Any]:
|
||||
"""Import sessions exported by :meth:`export_session` or ``export_all``.
|
||||
|
||||
Existing session IDs are skipped. Imported child sessions keep their
|
||||
parent only when that parent already exists or is included in the same
|
||||
import payload; otherwise the child is detached so partial imports don't
|
||||
fail foreign-key validation. Gateway routing, handoff, rewind, and other
|
||||
live runtime state are intentionally reset: this restores conversation
|
||||
history, not ownership of a live channel or process.
|
||||
Existing ids are skipped. A child keeps its parent only when the parent
|
||||
exists or is in the same payload; otherwise it is detached so partial
|
||||
imports pass FK validation. Gateway routing, handoff, rewind and other
|
||||
live runtime state are reset: this restores history, not ownership of
|
||||
a live channel or process.
|
||||
|
||||
Activity contract (#76354 review S4): export INCLUDES the live
|
||||
activity fields (``last_activity_at`` / ``last_activity_description``
|
||||
/ ``last_activity_provenance``) because they are part of the durable
|
||||
row, but import deliberately RESETS them to NULL. Resurrecting a
|
||||
stale "working ..." label on a machine where no agent is running
|
||||
would fabricate activity the watchdog and session listings act on.
|
||||
This asymmetry is intentional and covered by regression
|
||||
(tests/gateway/test_watchdog_review_76354.py::test_s4_export_includes_activity_import_resets_it).
|
||||
Activity contract: export INCLUDES ``last_activity_*`` (durable row
|
||||
fields) but import RESETS them to NULL — resurrecting a stale
|
||||
"working ..." label would fabricate activity the watchdog and listings
|
||||
act on. Intentional asymmetry, pinned by regression test.
|
||||
"""
|
||||
if not isinstance(sessions, list):
|
||||
raise ValueError("sessions must be a list")
|
||||
@@ -536,32 +424,14 @@ class SessionPortabilityMixin:
|
||||
total_messages = 0
|
||||
total_bytes = 0
|
||||
session_text_fields = (
|
||||
"source",
|
||||
"user_id",
|
||||
"model",
|
||||
"system_prompt",
|
||||
"end_reason",
|
||||
"cwd",
|
||||
"git_branch",
|
||||
"git_repo_root",
|
||||
"billing_provider",
|
||||
"billing_base_url",
|
||||
"billing_mode",
|
||||
"cost_status",
|
||||
"cost_source",
|
||||
"pricing_version",
|
||||
"title",
|
||||
"source", "user_id", "model", "system_prompt", "end_reason", "cwd",
|
||||
"git_branch", "git_repo_root", "billing_provider", "billing_base_url",
|
||||
"billing_mode", "cost_status", "cost_source", "pricing_version", "title",
|
||||
)
|
||||
# ``role`` is validated separately below (non-empty string).
|
||||
message_text_fields = (
|
||||
"role",
|
||||
"tool_call_id",
|
||||
"tool_name",
|
||||
"effect_disposition",
|
||||
"finish_reason",
|
||||
"reasoning",
|
||||
"reasoning_content",
|
||||
"platform_message_id",
|
||||
"message_id",
|
||||
"tool_call_id", "tool_name", "effect_disposition", "finish_reason",
|
||||
"reasoning", "reasoning_content", "platform_message_id", "message_id",
|
||||
)
|
||||
|
||||
for index, raw in enumerate(sessions):
|
||||
@@ -580,43 +450,23 @@ class SessionPortabilityMixin:
|
||||
errors.append(self._import_error(index, session_id, "messages must be a list"))
|
||||
continue
|
||||
if len(messages) > self._IMPORT_MAX_MESSAGES_PER_SESSION:
|
||||
errors.append(
|
||||
self._import_error(
|
||||
index,
|
||||
session_id,
|
||||
"messages exceeds the per-session import limit",
|
||||
)
|
||||
)
|
||||
errors.append(self._import_error(index, session_id, "messages exceeds the per-session import limit"))
|
||||
continue
|
||||
if any(not isinstance(msg, dict) for msg in messages):
|
||||
errors.append(
|
||||
self._import_error(
|
||||
index,
|
||||
session_id,
|
||||
"messages must contain only objects",
|
||||
)
|
||||
)
|
||||
errors.append(self._import_error(index, session_id, "messages must contain only objects"))
|
||||
continue
|
||||
|
||||
try:
|
||||
session_bytes = len(
|
||||
json.dumps(raw, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
|
||||
)
|
||||
session_bytes = len(json.dumps(raw, ensure_ascii=False, separators=(",", ":")).encode("utf-8"))
|
||||
except (TypeError, ValueError):
|
||||
errors.append(
|
||||
self._import_error(index, session_id, "session must be JSON serializable")
|
||||
)
|
||||
errors.append(self._import_error(index, session_id, "session must be JSON serializable"))
|
||||
continue
|
||||
if session_bytes > self._IMPORT_MAX_SESSION_BYTES:
|
||||
errors.append(
|
||||
self._import_error(index, session_id, "session exceeds the import size limit")
|
||||
)
|
||||
errors.append(self._import_error(index, session_id, "session exceeds the import size limit"))
|
||||
continue
|
||||
total_bytes += session_bytes
|
||||
if total_bytes > self._IMPORT_MAX_TOTAL_BYTES:
|
||||
errors.append(
|
||||
self._import_error(index, session_id, "import exceeds the total size limit")
|
||||
)
|
||||
errors.append(self._import_error(index, session_id, "import exceeds the total size limit"))
|
||||
continue
|
||||
|
||||
try:
|
||||
@@ -640,8 +490,6 @@ class SessionPortabilityMixin:
|
||||
if not isinstance(role, str) or not role:
|
||||
raise ValueError(f"messages[{message_index}].role must be a non-empty string")
|
||||
for field in message_text_fields:
|
||||
if field == "role":
|
||||
continue
|
||||
clean_message[field] = self._import_text_or_none(
|
||||
clean_message.get(field), field
|
||||
)
|
||||
@@ -655,13 +503,7 @@ class SessionPortabilityMixin:
|
||||
|
||||
total_messages += len(clean_messages)
|
||||
if total_messages > self._IMPORT_MAX_TOTAL_MESSAGES:
|
||||
errors.append(
|
||||
self._import_error(
|
||||
index,
|
||||
session_id,
|
||||
"messages exceeds the total import limit",
|
||||
)
|
||||
)
|
||||
errors.append(self._import_error(index, session_id, "messages exceeds the total import limit"))
|
||||
continue
|
||||
seen_ids.add(session_id)
|
||||
normalized.append(
|
||||
@@ -669,13 +511,7 @@ class SessionPortabilityMixin:
|
||||
)
|
||||
|
||||
if errors:
|
||||
return {
|
||||
"ok": False,
|
||||
"imported": 0,
|
||||
"skipped": 0,
|
||||
"detached": 0,
|
||||
"errors": errors,
|
||||
}
|
||||
return {"ok": False, "imported": 0, "skipped": 0, "detached": 0, "errors": errors}
|
||||
|
||||
def _do(conn):
|
||||
imported_ids: List[str] = []
|
||||
@@ -738,27 +574,17 @@ class SessionPortabilityMixin:
|
||||
"end_reason": raw.get("end_reason"),
|
||||
"input_tokens": self._int_or_default(raw.get("input_tokens")),
|
||||
"output_tokens": self._int_or_default(raw.get("output_tokens")),
|
||||
"cache_read_tokens": self._int_or_default(
|
||||
raw.get("cache_read_tokens")
|
||||
),
|
||||
"cache_write_tokens": self._int_or_default(
|
||||
raw.get("cache_write_tokens")
|
||||
),
|
||||
"reasoning_tokens": self._int_or_default(
|
||||
raw.get("reasoning_tokens")
|
||||
),
|
||||
"cache_read_tokens": self._int_or_default(raw.get("cache_read_tokens")),
|
||||
"cache_write_tokens": self._int_or_default(raw.get("cache_write_tokens")),
|
||||
"reasoning_tokens": self._int_or_default(raw.get("reasoning_tokens")),
|
||||
"cwd": raw.get("cwd"),
|
||||
"git_branch": raw.get("git_branch"),
|
||||
"git_repo_root": raw.get("git_repo_root"),
|
||||
"billing_provider": raw.get("billing_provider"),
|
||||
"billing_base_url": raw.get("billing_base_url"),
|
||||
"billing_mode": raw.get("billing_mode"),
|
||||
"estimated_cost_usd": self._float_or_none(
|
||||
raw.get("estimated_cost_usd")
|
||||
),
|
||||
"actual_cost_usd": self._float_or_none(
|
||||
raw.get("actual_cost_usd")
|
||||
),
|
||||
"estimated_cost_usd": self._float_or_none(raw.get("estimated_cost_usd")),
|
||||
"actual_cost_usd": self._float_or_none(raw.get("actual_cost_usd")),
|
||||
"cost_status": raw.get("cost_status"),
|
||||
"cost_source": raw.get("cost_source"),
|
||||
"pricing_version": raw.get("pricing_version"),
|
||||
@@ -771,18 +597,12 @@ class SessionPortabilityMixin:
|
||||
sanitized_messages: List[Dict[str, Any]] = []
|
||||
for msg in messages:
|
||||
clean = dict(msg)
|
||||
for key in (
|
||||
"reasoning_details",
|
||||
"codex_reasoning_items",
|
||||
"codex_message_items",
|
||||
):
|
||||
for key in ("reasoning_details", "codex_reasoning_items", "codex_message_items"):
|
||||
clean[key] = self._reasoning_json_value(clean.get(key))
|
||||
sanitized_messages.append(clean)
|
||||
|
||||
total_messages, total_tool_calls = self._insert_message_rows(
|
||||
conn,
|
||||
session_id,
|
||||
sanitized_messages,
|
||||
conn, session_id, sanitized_messages
|
||||
)
|
||||
conn.execute(
|
||||
"UPDATE sessions SET message_count = ?, tool_call_count = ? WHERE id = ?",
|
||||
@@ -826,9 +646,8 @@ class SessionPortabilityMixin:
|
||||
(parent_id, session_id),
|
||||
)
|
||||
else:
|
||||
# Drop only the closing edge. Later entries can still attach
|
||||
# to this now-root session, preserving the acyclic portion
|
||||
# of a malformed imported lineage.
|
||||
# Drop only the closing edge; later entries can still attach
|
||||
# to this now-root session.
|
||||
parent_by_child.pop(session_id, None)
|
||||
detached += 1
|
||||
|
||||
|
||||
Reference in New Issue
Block a user