refactor(state): split SessionDB into domain mixins and free-function modules; unify SQL boilerplate

hermes_state.py 17,220 -> 6,442 LOC. Behavior-neutral: every moved body is
AST-identical to the original, verified per extraction.

SessionDB core
- _write_sql / _write_rowcount / _read_one / _read_all replace ~120 copies of
  the `def _do(conn): conn.execute(...)` + `_execute_write(_do)` and
  `with self._read_ctx() as conn: row = conn.execute(...).fetchone()` shapes.
- _set_lineage_column replaces four copies of the recursive compression-lineage
  UPDATE (archived / pinned / hidden / last_read_at).
- _read_session_number unifies the three compression counter readers.
- Dead (zero refs repo-wide): restore_rewound, delete_gateway_routing_entries,
  _is_duplicate_replayed_user_message, SessionPortabilityMixin.get_first_assistant_text.

New mixins bound onto SessionDB via the MRO (logger name stays "hermes_state"):
  hermes_state_messages    SessionMessagesMixin       48 methods
  hermes_state_compression SessionCompressionMixin    30
  hermes_state_gateway     SessionGatewayMixin        26
  hermes_state_maintenance SessionMaintenanceMixin    13
  hermes_state_usage       SessionUsageMixin          12
  hermes_state_titles      SessionTitlesMixin         13
  hermes_state_telegram    SessionTelegramTopicsMixin 11
Origin-internal symbols resolve through a lazy `from hermes_state import ...`
inside the few methods that need them (no import cycle).

New free-function modules, every name re-imported into hermes_state so
`hermes_state.<name>` (and test monkeypatches on it) keep working; intra-module
calls to patched helpers go through the lazy origin import:
  hermes_state_repair   repair/backup/preflight (43 defs)
  hermes_state_wal      journal-mode / PRAGMA policy (33 defs)
  hermes_state_dbfile   header probes, zeroed-db quarantine, stats, holders (21 defs)

Existing mixins: search — shared FTS MATCH/LIKE builders, unified rebuild
status/step/finish engines, state_meta helpers; schema — one legacy/v23 FTS init
branch, shared _live_pk_columns, Row/tuple dual access dropped; portability —
shared _PREVIEW_RAW_SUBQUERY_SQL and _rich_row; common — single
stat_db_file_identity (was 3 copies), AUTO_VACUUM_MIN_FREELIST_RATIO.

Docstrings/comments hand-compacted (AST-identical) keeping every invariant,
ordering rule, failure mode and WHY. Schema SQL, migration order and PRAGMAs
untouched. test_repair_path_has_no_bare_connects repointed to hermes_state_repair.
This commit is contained in:
Teknium
2026-09-02 12:09:00 -07:00
parent a3d33fe22f
commit d15c61b5dc
17 changed files with 11604 additions and 13788 deletions
+119 -300
View File
@@ -1,11 +1,9 @@
"""Session listing/rich rows, export, and import (portability) for SessionDB.
Mixin contract: this is a plain mixin class consumed by
``hermes_state.SessionDB``. It defines no ``__init__`` and no state of its
own; methods access the host's attributes (``self._conn``, ``self.db_path``,
``self._execute_write`` and other SessionDB methods) established by
``SessionDB.__init__``. It must never import hermes_state (cycle) — shared
module-level constants live in hermes_state_common.
Plain mixin consumed by ``hermes_state.SessionDB``: no ``__init__``, no state
of its own; methods use host attributes established by ``SessionDB.__init__``.
Must never import hermes_state (cycle) — shared constants live in
hermes_state_common.
"""
import logging
@@ -16,14 +14,12 @@ from typing import Any, Dict, List, Optional
from agent.skill_commands import SKILL_SCAFFOLD_SQL_LIKE
from hermes_state_common import (
SCHEMA_SQL,
_PREVIEW_ELIGIBLE_SQL,
_PREVIEW_RAW_SELECT,
_PREVIEW_RAW_SUBQUERY_SQL,
_shape_preview,
_sql_session_last_active,
)
# Moved methods logged under the "hermes_state" logger before the split;
# keep that logger identity so log filtering/capture behavior is unchanged.
# Keep the pre-split logger identity so log filtering/capture is unchanged.
logger = logging.getLogger("hermes_state")
@@ -32,9 +28,8 @@ class SessionPortabilityMixin:
@classmethod
def _compact_session_cols(cls) -> str:
"""SELECT list for compact_rows: every ``sessions`` column declared in
SCHEMA_SQL except prompt storage internals, aliased with the ``s``
prefix used by list_sessions_rich/_get_session_rich_row queries."""
"""``s.``-prefixed SELECT list of every SCHEMA_SQL ``sessions`` column
except prompt storage internals (the compact_rows projection)."""
if cls._session_compact_cols_sql is None:
declared = cls._parse_schema_columns(SCHEMA_SQL)["sessions"]
cls._session_compact_cols_sql = ", ".join(
@@ -43,13 +38,19 @@ class SessionPortabilityMixin:
)
return cls._session_compact_cols_sql
@classmethod
def _rich_row(cls, row) -> Dict[str, Any]:
"""Session row dict with ``_preview_raw`` shaped into ``preview``."""
s = cls._session_row_dict(row)
s["preview"] = _shape_preview(s.pop("_preview_raw", ""))
return s
def distinct_session_cwds(self, include_archived: bool = False) -> List[Dict[str, Any]]:
"""Distinct non-empty session cwds with usage stats, for repo discovery.
Aggregates across ALL session history (not a single page), so the desktop
can surface every git repo the user has worked in — not just the repos
that happen to be in the currently-loaded recents. Children/branches
count: a worktree session is still a real workspace signal.
Aggregates across ALL history (not one page) so every repo the user
worked in surfaces. Children/branches count: a worktree session is a
real workspace signal.
"""
where = "cwd IS NOT NULL AND TRIM(cwd) != ''"
if not include_archived:
@@ -77,41 +78,24 @@ class SessionPortabilityMixin:
) -> List[Dict[str, Any]]:
"""List the run sessions produced by a single cron job, newest first.
Cron runs are flat, independent sessions whose id is
``cron_{job_id}_{timestamp}`` (see ``cron/scheduler.run_job``). They are
never compression roots and never branch, so this deliberately skips the
``list_sessions_rich`` recursive compression-chain CTE / leading-wildcard
``id_query`` path — that path seeds from *every* ``source='cron'`` row in
the DB and only filters to one job's runs after the scan, so it scales
with the whole cron pile (a heavy history makes the desktop run-history
endpoint time out before it eventually populates).
Cron runs are flat sessions with id ``cron_{job_id}_{timestamp}``; they
never compress or branch, so this skips ``list_sessions_rich``'s
compression-chain CTE / leading-wildcard ``id_query`` path, which seeds
from EVERY ``source='cron'`` row and scales with the whole cron pile.
Instead: a ``[prefix, prefix_hi)`` index range scan on id, filtered to
``source='cron'``, so work scales with the requested window.
Instead this binds to one job with a ``[prefix, prefix_hi)`` range over
the id (an index range scan, not a ``%...%`` substring), filters
``source='cron'``, and orders by ``started_at DESC``. Work scales with
the requested window, not the total cron history.
Returns the same enriched row shape as ``list_sessions_rich`` (adds
``preview`` + ``last_active``) so callers can reuse it.
Returns the ``list_sessions_rich`` row shape (``preview`` + ``last_active``).
"""
prefix = f"cron_{job_id}_"
# Half-open upper bound for an index range scan: increment the final
# byte of the prefix so the range covers exactly the ids that start
# with ``prefix`` and nothing else. ``prefix`` always ends in '_', but
# compute it generically rather than hardcoding the successor char.
# Half-open upper bound: bump the final byte so the range covers exactly
# the ids starting with ``prefix``.
prefix_hi = prefix[:-1] + chr(ord(prefix[-1]) + 1)
query = f"""
SELECT s.*,
COALESCE(sp.prompt, s.system_prompt) AS _system_prompt_resolved,
COALESCE(
(SELECT {_PREVIEW_RAW_SELECT}
FROM messages m
WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL
AND {_PREVIEW_ELIGIBLE_SQL}
ORDER BY m.timestamp, m.id LIMIT 1),
''
) AS _preview_raw,
{_PREVIEW_RAW_SUBQUERY_SQL},
{_sql_session_last_active("s")} AS last_active
FROM sessions s
LEFT JOIN system_prompts sp ON sp.hash = s.system_prompt_hash
@@ -120,27 +104,12 @@ class SessionPortabilityMixin:
LIMIT ? OFFSET ?
"""
with self._lock:
cursor = self._conn.execute(query, (prefix, prefix_hi, limit, offset))
rows = cursor.fetchall()
runs: List[Dict[str, Any]] = []
for row in rows:
s = self._session_row_dict(row)
s["preview"] = _shape_preview(s.pop("_preview_raw", ""))
runs.append(s)
return runs
rows = self._conn.execute(query, (prefix, prefix_hi, limit, offset)).fetchall()
return [self._rich_row(row) for row in rows]
def _get_session_rich_row(self, session_id: str, compact_rows: bool = False) -> Optional[Dict[str, Any]]:
"""Fetch a single session with the same enriched columns as
``list_sessions_rich`` (preview + last_active). Returns None if the
session doesn't exist.
Pass ``compact_rows=True`` to omit the ``system_prompt`` blob (see
``list_sessions_rich`` for details).
Thin wrapper over ``_get_session_rich_rows_batch`` so the enriched
SELECT lives in exactly one place.
"""
"""One session with the ``list_sessions_rich`` enriched columns, or
None. ``compact_rows=True`` omits the ``system_prompt`` blob."""
return self._get_session_rich_rows_batch(
[session_id], compact_rows=compact_rows
).get(session_id)
@@ -148,23 +117,15 @@ class SessionPortabilityMixin:
def _get_session_rich_rows_batch(
self, session_ids, compact_rows: bool = False
) -> Dict[str, Dict[str, Any]]:
"""Fetch multiple sessions with the same enriched columns as
``_get_session_rich_row``, in a single query.
Used by ``list_sessions_rich``'s compression-tip projection to resolve
every tip row for a page in one round trip instead of one query per
compression-root row. Returns a dict keyed by session id; ids that
don't exist are simply absent from the result (same as
``_get_session_rich_row`` returning ``None`` for them).
"""Enriched rows for many sessions in one query, keyed by id; missing
ids are simply absent. Resolves a page of compression tips in one
round trip instead of one query per root row.
"""
ids = [sid for sid in session_ids if sid]
if not ids:
return {}
# Old SQLite builds cap bound variables at 999
# (SQLITE_MAX_VARIABLE_NUMBER); large pages (limit=10000 callers
# exist) could exceed it. Chunk the IN list so the helper is safe at
# any page size — this is the single choke point for the enriched
# multi-row fetch, so the bound lives here, not at call sites.
# Old SQLite caps bound variables at 999 (SQLITE_MAX_VARIABLE_NUMBER);
# limit=10000 callers exist. Chunk here — the single choke point.
_CHUNK = 900
if len(ids) > _CHUNK:
result: Dict[str, Dict[str, Any]] = {}
@@ -189,45 +150,26 @@ class SessionPortabilityMixin:
)
query = f"""
SELECT {_sel}{prompt_select},
COALESCE(
(SELECT {_PREVIEW_RAW_SELECT}
FROM messages m
WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL
AND {_PREVIEW_ELIGIBLE_SQL}
ORDER BY m.timestamp, m.id LIMIT 1),
''
) AS _preview_raw,
{_PREVIEW_RAW_SUBQUERY_SQL},
{_sql_session_last_active("s")} AS last_active
FROM sessions s
{prompt_join}
WHERE s.id IN ({placeholders})
"""
with self._lock:
cursor = self._conn.execute(query, ids)
rows = cursor.fetchall()
result: Dict[str, Dict[str, Any]] = {}
for row in rows:
s = self._session_row_dict(row)
s["preview"] = _shape_preview(s.pop("_preview_raw", ""))
result[s["id"]] = s
return result
rows = self._conn.execute(query, ids).fetchall()
return {s["id"]: s for s in map(self._rich_row, rows)}
def get_session_rich_row(self, session_id: str, compact_rows: bool = False) -> Optional[Dict[str, Any]]:
"""Public wrapper for :meth:`_get_session_rich_row`.
Exposes the single-session enriched row (same columns as
``list_sessions_rich``: preview + last_active) for callers outside
this module, e.g. the web server's session-search hydration.
"""
"""Public wrapper for :meth:`_get_session_rich_row` (web server hydration)."""
return self._get_session_rich_row(session_id, compact_rows=compact_rows)
def list_skill_scaffolded_sessions(self, limit: int = 200) -> List[Dict[str, Any]]:
"""Titled sessions whose first user turn was a ``/skill`` invocation.
Those titles were generated from the expanded message, which embeds the
whole skill body — so they describe the skill rather than the request.
Returns ``id``, ``title``, and the full first-turn ``content`` so a
caller can re-derive what the user typed. Newest first.
Their titles were generated from the expanded skill body, so they
describe the skill, not the request. Returns ``id``, ``title`` and the
first-turn ``content`` so callers can re-derive what was typed. Newest first.
"""
with self._lock:
rows = self._conn.execute(
@@ -248,63 +190,36 @@ class SessionPortabilityMixin:
).fetchall()
return [dict(row) for row in rows]
def get_first_assistant_text(self, session_id: str) -> str:
"""The session's first assistant reply as plain text ('' when none).
Pairs with :meth:`list_skill_scaffolded_sessions` so a re-title can feed
the titler the same (request, reply) shape the live path uses.
"""
with self._lock:
row = self._conn.execute(
"SELECT content FROM messages "
"WHERE session_id = ? AND role = 'assistant' AND content IS NOT NULL "
"ORDER BY timestamp, id LIMIT 1",
(session_id,),
).fetchone()
if not row:
return ""
decoded = self._decode_content(row["content"])
return decoded if isinstance(decoded, str) else ""
def export_session(self, session_id: str) -> Optional[Dict[str, Any]]:
"""Export a single session with all its messages as a dict."""
session = self.get_session(session_id)
if not session:
return None
messages = self.get_messages(session_id)
return {**session, "messages": messages}
return {**session, "messages": self.get_messages(session_id)}
def export_session_lineage(self, session_id: str) -> Optional[Dict[str, Any]]:
"""Export a compression lineage as one logical session dict."""
lineage_ids = self.get_compression_lineage(session_id)
if not lineage_ids:
return None
segments = []
for sid in lineage_ids:
segment = self.export_session(sid)
if segment:
segments.append(segment)
segments = [seg for seg in map(self.export_session, lineage_ids) if seg]
if not segments:
return None
base = dict(segments[-1])
total_messages = sum(len(seg.get("messages") or []) for seg in segments)
base["segments"] = segments
base["lineage_session_ids"] = [seg["id"] for seg in segments]
base["message_count"] = total_messages
base["messages"] = [msg for seg in segments for msg in (seg.get("messages") or [])]
return base
messages = [msg for seg in segments for msg in (seg.get("messages") or [])]
return {
**segments[-1],
"segments": segments,
"lineage_session_ids": [seg["id"] for seg in segments],
"message_count": len(messages),
"messages": messages,
}
def export_all(self, source: str = None) -> List[Dict[str, Any]]:
"""
Export all sessions (with messages) as a list of dicts.
Suitable for writing to a JSONL file for backup/analysis.
"""
sessions = self.search_sessions(source=source, limit=100000)
results = []
for session in sessions:
messages = self.get_messages(session["id"])
results.append({**session, "messages": messages})
return results
"""Export all sessions (with messages) as dicts, e.g. for JSONL backup."""
return [
{**session, "messages": self.get_messages(session["id"])}
for session in self.search_sessions(source=source, limit=100000)
]
def adopt_session_lineage_from(
self,
@@ -313,52 +228,36 @@ class SessionPortabilityMixin:
*,
retire_donor: bool = True,
) -> Dict[str, Any]:
"""Adopt *session_id*'s full compression lineage from *donor_db* into
this store.
"""Adopt *session_id*'s full compression lineage from *donor_db*.
The stranded-bot-session heal (#93091 follow-up to #93296): before the
desktop routed session RPCs by their target session, a profile bot's
turns executed on whichever backend held window focus — usually the
default one — so the bot's canonical session rows and messages
accumulated in the DEFAULT profile's state.db. Once routing was fixed,
the profile backend correctly received the RPCs but had no such
session, so the same chat 4001'd for the opposite reason. This method
moves the conversation to where routing now looks for it.
Stranded-bot-session heal: before the desktop routed session RPCs by
target session, a profile bot's rows accumulated in the DEFAULT
profile's state.db; this moves the conversation to where routing now
looks. Pure composition: ``donor_db.export_session_lineage()`` ->
``self.import_sessions()`` — routing/handoff/activity fields reset,
already-present ids skipped (idempotent re-adoption).
Composition of existing primitives (no new import/export machinery):
``donor_db.export_session_lineage()`` -> ``self.import_sessions()``.
Import semantics apply unchanged: gateway routing, handoff, and live
activity fields are reset; already-present ids are skipped
(idempotent re-adoption after a partial run).
With ``retire_donor`` and a complete adoption, donor rows are ARCHIVED
(never deleted) with ``end_reason='adopted_by_profile'``. That
end_reason is deliberately NOT in the recoverable set
(agent_close/ws_orphan_reap): resurrection must not undo an adoption.
When ``retire_donor`` is True and at least one segment was imported
(or every segment already exists here), the donor rows are ARCHIVED —
never deleted — with ``end_reason='adopted_by_profile'`` so the
default profile's list stops advertising a conversation that now
lives elsewhere, while the bytes stay recoverable. The archive is
deliberately NOT in the recoverable set (agent_close/ws_orphan_reap):
canonical-lookup resurrection must not undo an adoption.
Returns the ``import_sessions`` result dict, plus ``adopted`` (bool)
and ``donor_retired`` (bool — True only when EVERY segment's
retirement actually applied).
Returns the ``import_sessions`` dict plus ``adopted`` and
``donor_retired`` (True only when EVERY segment's retirement applied).
"""
payload = donor_db.export_session_lineage(session_id)
if not payload:
return {
"ok": False,
"adopted": False,
"donor_retired": False,
"ok": False, "adopted": False, "donor_retired": False,
"error": f"session {session_id!r} not found in donor store",
}
segments = payload.get("segments") or [payload]
# Divergence guard: a segment we are about to SKIP (already present
# here) may have kept accumulating messages in the donor store after
# a partial earlier adoption. Retiring it would strand those newer
# messages behind a non-recoverable archive. Compare counts up front
# and refuse to retire (still adopt/import) when the donor is ahead.
# Divergence guard: a segment we will SKIP (already here) may have kept
# growing in the donor after a partial adoption; retiring it would strand
# those messages behind a non-recoverable archive. Still import, but
# refuse to retire when the donor is ahead.
donor_ahead = False
for seg in segments:
seg_id = seg.get("id")
@@ -394,15 +293,11 @@ class SessionPortabilityMixin:
if not seg_id:
continue
try:
# TOCTOU close-out: the guard above compared EXPORT-TIME
# counts, but another backend can append donor messages
# between export and this loop. Re-read both stores right
# before stamping; a donor-ahead signal here skips the
# stamp so growth never lands behind a non-recoverable
# archive. (Count comparison cannot see equal-count
# CONTENT divergence — e.g. a donor rewind+rewrite; that
# residual case is accepted: bytes stay in the donor
# store either way, only reachability differs.)
# TOCTOU close-out: the guard above used EXPORT-TIME counts;
# re-read both stores right before stamping so donor growth
# never lands behind a non-recoverable archive. (Equal-count
# CONTENT divergence is accepted: bytes stay in the donor
# either way, only reachability differs.)
donor_now = len(donor_db.get_messages(seg_id))
local_now = len(self.get_messages(seg_id))
if donor_now > local_now:
@@ -414,17 +309,15 @@ class SessionPortabilityMixin:
seg_id, donor_now, local_now,
)
continue
# First end_reason wins in end_session(); reopen first so
# the adoption boundary is stamped even on ended segments
# (e.g. 'compression' parents).
# First end_reason wins in end_session(); reopen so the
# adoption boundary is stamped even on ended segments.
donor_db.reopen_session(seg_id)
donor_db.end_session(seg_id, "adopted_by_profile")
donor_db.set_session_archived(seg_id, True)
except Exception:
# Best-effort by design: a retirement failure must not
# fail the adoption (the profile copy is already whole;
# a later resume retries retirement idempotently). But
# never claim success we didn't have.
# Best-effort: a retirement failure must not fail the adoption
# (a later resume retries idempotently) — but never claim
# success we didn't have.
retire_ok = False
logger.warning(
"failed to retire donor segment %s after adoption",
@@ -507,21 +400,16 @@ class SessionPortabilityMixin:
def import_sessions(self, sessions: List[Dict[str, Any]]) -> Dict[str, Any]:
"""Import sessions exported by :meth:`export_session` or ``export_all``.
Existing session IDs are skipped. Imported child sessions keep their
parent only when that parent already exists or is included in the same
import payload; otherwise the child is detached so partial imports don't
fail foreign-key validation. Gateway routing, handoff, rewind, and other
live runtime state are intentionally reset: this restores conversation
history, not ownership of a live channel or process.
Existing ids are skipped. A child keeps its parent only when the parent
exists or is in the same payload; otherwise it is detached so partial
imports pass FK validation. Gateway routing, handoff, rewind and other
live runtime state are reset: this restores history, not ownership of
a live channel or process.
Activity contract (#76354 review S4): export INCLUDES the live
activity fields (``last_activity_at`` / ``last_activity_description``
/ ``last_activity_provenance``) because they are part of the durable
row, but import deliberately RESETS them to NULL. Resurrecting a
stale "working ..." label on a machine where no agent is running
would fabricate activity the watchdog and session listings act on.
This asymmetry is intentional and covered by regression
(tests/gateway/test_watchdog_review_76354.py::test_s4_export_includes_activity_import_resets_it).
Activity contract: export INCLUDES ``last_activity_*`` (durable row
fields) but import RESETS them to NULL — resurrecting a stale
"working ..." label would fabricate activity the watchdog and listings
act on. Intentional asymmetry, pinned by regression test.
"""
if not isinstance(sessions, list):
raise ValueError("sessions must be a list")
@@ -536,32 +424,14 @@ class SessionPortabilityMixin:
total_messages = 0
total_bytes = 0
session_text_fields = (
"source",
"user_id",
"model",
"system_prompt",
"end_reason",
"cwd",
"git_branch",
"git_repo_root",
"billing_provider",
"billing_base_url",
"billing_mode",
"cost_status",
"cost_source",
"pricing_version",
"title",
"source", "user_id", "model", "system_prompt", "end_reason", "cwd",
"git_branch", "git_repo_root", "billing_provider", "billing_base_url",
"billing_mode", "cost_status", "cost_source", "pricing_version", "title",
)
# ``role`` is validated separately below (non-empty string).
message_text_fields = (
"role",
"tool_call_id",
"tool_name",
"effect_disposition",
"finish_reason",
"reasoning",
"reasoning_content",
"platform_message_id",
"message_id",
"tool_call_id", "tool_name", "effect_disposition", "finish_reason",
"reasoning", "reasoning_content", "platform_message_id", "message_id",
)
for index, raw in enumerate(sessions):
@@ -580,43 +450,23 @@ class SessionPortabilityMixin:
errors.append(self._import_error(index, session_id, "messages must be a list"))
continue
if len(messages) > self._IMPORT_MAX_MESSAGES_PER_SESSION:
errors.append(
self._import_error(
index,
session_id,
"messages exceeds the per-session import limit",
)
)
errors.append(self._import_error(index, session_id, "messages exceeds the per-session import limit"))
continue
if any(not isinstance(msg, dict) for msg in messages):
errors.append(
self._import_error(
index,
session_id,
"messages must contain only objects",
)
)
errors.append(self._import_error(index, session_id, "messages must contain only objects"))
continue
try:
session_bytes = len(
json.dumps(raw, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
)
session_bytes = len(json.dumps(raw, ensure_ascii=False, separators=(",", ":")).encode("utf-8"))
except (TypeError, ValueError):
errors.append(
self._import_error(index, session_id, "session must be JSON serializable")
)
errors.append(self._import_error(index, session_id, "session must be JSON serializable"))
continue
if session_bytes > self._IMPORT_MAX_SESSION_BYTES:
errors.append(
self._import_error(index, session_id, "session exceeds the import size limit")
)
errors.append(self._import_error(index, session_id, "session exceeds the import size limit"))
continue
total_bytes += session_bytes
if total_bytes > self._IMPORT_MAX_TOTAL_BYTES:
errors.append(
self._import_error(index, session_id, "import exceeds the total size limit")
)
errors.append(self._import_error(index, session_id, "import exceeds the total size limit"))
continue
try:
@@ -640,8 +490,6 @@ class SessionPortabilityMixin:
if not isinstance(role, str) or not role:
raise ValueError(f"messages[{message_index}].role must be a non-empty string")
for field in message_text_fields:
if field == "role":
continue
clean_message[field] = self._import_text_or_none(
clean_message.get(field), field
)
@@ -655,13 +503,7 @@ class SessionPortabilityMixin:
total_messages += len(clean_messages)
if total_messages > self._IMPORT_MAX_TOTAL_MESSAGES:
errors.append(
self._import_error(
index,
session_id,
"messages exceeds the total import limit",
)
)
errors.append(self._import_error(index, session_id, "messages exceeds the total import limit"))
continue
seen_ids.add(session_id)
normalized.append(
@@ -669,13 +511,7 @@ class SessionPortabilityMixin:
)
if errors:
return {
"ok": False,
"imported": 0,
"skipped": 0,
"detached": 0,
"errors": errors,
}
return {"ok": False, "imported": 0, "skipped": 0, "detached": 0, "errors": errors}
def _do(conn):
imported_ids: List[str] = []
@@ -738,27 +574,17 @@ class SessionPortabilityMixin:
"end_reason": raw.get("end_reason"),
"input_tokens": self._int_or_default(raw.get("input_tokens")),
"output_tokens": self._int_or_default(raw.get("output_tokens")),
"cache_read_tokens": self._int_or_default(
raw.get("cache_read_tokens")
),
"cache_write_tokens": self._int_or_default(
raw.get("cache_write_tokens")
),
"reasoning_tokens": self._int_or_default(
raw.get("reasoning_tokens")
),
"cache_read_tokens": self._int_or_default(raw.get("cache_read_tokens")),
"cache_write_tokens": self._int_or_default(raw.get("cache_write_tokens")),
"reasoning_tokens": self._int_or_default(raw.get("reasoning_tokens")),
"cwd": raw.get("cwd"),
"git_branch": raw.get("git_branch"),
"git_repo_root": raw.get("git_repo_root"),
"billing_provider": raw.get("billing_provider"),
"billing_base_url": raw.get("billing_base_url"),
"billing_mode": raw.get("billing_mode"),
"estimated_cost_usd": self._float_or_none(
raw.get("estimated_cost_usd")
),
"actual_cost_usd": self._float_or_none(
raw.get("actual_cost_usd")
),
"estimated_cost_usd": self._float_or_none(raw.get("estimated_cost_usd")),
"actual_cost_usd": self._float_or_none(raw.get("actual_cost_usd")),
"cost_status": raw.get("cost_status"),
"cost_source": raw.get("cost_source"),
"pricing_version": raw.get("pricing_version"),
@@ -771,18 +597,12 @@ class SessionPortabilityMixin:
sanitized_messages: List[Dict[str, Any]] = []
for msg in messages:
clean = dict(msg)
for key in (
"reasoning_details",
"codex_reasoning_items",
"codex_message_items",
):
for key in ("reasoning_details", "codex_reasoning_items", "codex_message_items"):
clean[key] = self._reasoning_json_value(clean.get(key))
sanitized_messages.append(clean)
total_messages, total_tool_calls = self._insert_message_rows(
conn,
session_id,
sanitized_messages,
conn, session_id, sanitized_messages
)
conn.execute(
"UPDATE sessions SET message_count = ?, tool_call_count = ? WHERE id = ?",
@@ -826,9 +646,8 @@ class SessionPortabilityMixin:
(parent_id, session_id),
)
else:
# Drop only the closing edge. Later entries can still attach
# to this now-root session, preserving the acyclic portion
# of a malformed imported lineage.
# Drop only the closing edge; later entries can still attach
# to this now-root session.
parent_by_child.pop(session_id, None)
detached += 1