Files
hermes-agent/tests/run_agent/test_empty_response_recovery_persistence.py
T
devsart95 06ae5b6faa perf(state): batch the turn flush into one SQLite transaction
Re-derivation of #23254 (@devsart95) on today's flush loop. The turn
flush in _flush_messages_to_session_db wrote one BEGIN IMMEDIATE
transaction per message row; a typical agent turn (user + assistant +
tool results) paid 3-8 transactions -- and, off WAL (the default on
macOS while the WAL-reset guard is active), 3-8 fsyncs -- per turn.

Adds SessionDB.append_messages_batch: same row shape as append_message
(shared _prepare_message_row serializer + _MESSAGE_INSERT_SQL column
list, so the two writers cannot drift), same compression-lock and
compression-closed guards, one aggregated session-counter UPDATE, one
transaction for the whole batch. Row serialization stays outside the
write lock.

The flush loop now collects the turn's new rows and writes them in one
call. All-or-nothing pairs exactly with the persisted-marker stamping:
on failure no rows landed and no markers were stamped, so the next
flush re-writes the whole tail (same recovery contract as before,
minus the partial-prefix case that could double-count).

Measured (same harness, 5-message turn, journal_mode=DELETE,
synchronous=FULL): 2.32ms -> 0.83ms median per turn flush (64% faster,
5 fsyncs -> 1). On WAL the win is smaller but the atomicity fix holds.
2026-08-03 20:43:38 +05:30

167 lines
6.2 KiB
Python

"""Regression tests for empty-response recovery transcript persistence."""
from run_agent import AIAgent
class _CapturingSessionDB:
"""Minimal SessionDB stand-in that records every appended message."""
def __init__(self):
self.rows = []
def append_message(self, session_id, role, content=None, **kwargs):
self.rows.append({"role": role, "content": content})
return len(self.rows)
def append_messages_batch(self, session_id, messages, **kwargs):
# Mirror the real batch writer: same rows, one call.
for m in messages:
self.rows.append({"role": m.get("role"), "content": m.get("content")})
return list(range(len(self.rows) - len(messages) + 1, len(self.rows) + 1))
def _agent_with_capturing_db():
agent = AIAgent.__new__(AIAgent)
agent._persist_user_message_idx = None
agent._persist_user_message_override = None
agent._session_db = _CapturingSessionDB()
agent._session_db_created = True
agent._last_flushed_db_idx = 0
agent.session_id = "sess-test"
return agent
def _agent_with_stubbed_persistence():
agent = AIAgent.__new__(AIAgent)
agent._persist_user_message_idx = None
agent._persist_user_message_override = None
agent._session_db = None
agent._session_messages = []
agent.flushed_session_db_messages = []
agent._flush_messages_to_session_db = lambda messages, conversation_history=None: (
agent.flushed_session_db_messages.append([m.copy() for m in messages])
)
return agent
def test_persist_session_strips_trailing_empty_recovery_scaffolding():
"""After stripping scaffolding, also rewind past orphan trailing tool-result
messages that the failed iteration left behind. Otherwise the next user
message lands after a bare ``tool`` and produces a protocol-invalid
sequence that most providers silently fail on, retriggering the empty-
retry loop indefinitely.
"""
agent = _agent_with_stubbed_persistence()
messages = [
{"role": "user", "content": "run the task"},
{
"role": "assistant",
"content": "",
"tool_calls": [{"id": "call_1", "type": "function",
"function": {"name": "x", "arguments": "{}"}}],
},
{"role": "tool", "content": "{}", "tool_call_id": "call_1"},
{
"role": "assistant",
"content": "(empty)",
"_empty_recovery_synthetic": True,
},
{
"role": "user",
"content": (
"You just executed tool calls but returned an empty response. "
"Please process the tool results above and continue with the task."
),
"_empty_recovery_synthetic": True,
},
]
AIAgent._persist_session(agent, messages, conversation_history=[])
# After strip + rewind, only the original user message remains. The
# assistant(tool_calls) + tool pair is dropped because its iteration
# never produced a real response.
assert messages == [
{"role": "user", "content": "run the task"},
]
assert agent.flushed_session_db_messages[-1] == messages
assert all(not msg.get("_empty_recovery_synthetic") for msg in messages)
def test_persist_session_keeps_unmarked_terminal_empty_response():
agent = _agent_with_stubbed_persistence()
messages = [
{"role": "user", "content": "run the task"},
{"role": "assistant", "content": "(empty)"},
]
AIAgent._persist_session(agent, messages, conversation_history=[])
assert messages == [
{"role": "user", "content": "run the task"},
{"role": "assistant", "content": "(empty)"},
]
assert agent.flushed_session_db_messages[-1] == messages
def test_flush_never_writes_buried_empty_recovery_scaffolding():
"""When an empty-after-tools nudge is followed by a tool-calling response,
the synthetic ``(empty)`` + nudge pair stays buried in the live message
list (only the trailing copies are ever dropped). The append-only flush
must skip it regardless of position, otherwise the synthetic turns land in
the session store and pollute every resumed transcript.
"""
agent = _agent_with_capturing_db()
messages = [
{"role": "user", "content": "run the task"},
{
"role": "assistant",
"content": "",
"tool_calls": [{"id": "call_1", "type": "function",
"function": {"name": "x", "arguments": "{}"}}],
},
{"role": "tool", "content": "{}", "tool_call_id": "call_1"},
# Synthetic recovery scaffolding, now buried because the model answered
# the nudge with another tool call rather than terminating.
{"role": "assistant", "content": "(empty)", "_empty_recovery_synthetic": True},
{
"role": "user",
"content": "You just executed tool calls but returned an empty response.",
"_empty_recovery_synthetic": True,
},
{
"role": "assistant",
"content": "",
"tool_calls": [{"id": "call_2", "type": "function",
"function": {"name": "x", "arguments": "{}"}}],
},
{"role": "tool", "content": "{}", "tool_call_id": "call_2"},
{"role": "assistant", "content": "All done."},
]
agent._flush_messages_to_session_db(messages, conversation_history=[])
persisted = agent._session_db.rows
assert all(row["content"] != "(empty)" for row in persisted)
assert all("empty response" not in (row["content"] or "") for row in persisted)
# Only the genuine turns reach the store, in order.
assert [r["role"] for r in persisted] == [
"user", "assistant", "tool", "assistant", "tool", "assistant",
]
assert persisted[-1]["content"] == "All done."
def test_flush_skips_thinking_prefill_scaffolding():
agent = _agent_with_capturing_db()
messages = [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "", "_thinking_prefill": True},
{"role": "assistant", "content": "Hello!"},
]
agent._flush_messages_to_session_db(messages, conversation_history=[])
assert [r["content"] for r in agent._session_db.rows] == ["hi", "Hello!"]