1 Commits

Author SHA1 Message Date
m4 e0acc6155e feat: improve WebUI run recovery
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
2026-07-10 17:35:44 +08:00
11 changed files with 988 additions and 5 deletions
+1
View File
@@ -48,3 +48,4 @@ conversation_history/
*meals/
botpy.log
large_tool_results/
runs/
+9 -2
View File
@@ -41,9 +41,15 @@ from ..stream.console import console
# Front-end npm package + spec. ``@latest`` → always the newest published UI.
_WEBUI_PACKAGE = "@evoscientist/webui@latest"
_WEBUI_PACKAGE_ENV = "EVOSCIENTIST_WEBUI_PACKAGE"
_DEFAULT_WEBUI_PORT = 4716
def _resolve_webui_package() -> str:
"""Return the npm package spec to launch for the WebUI front-end."""
return os.getenv(_WEBUI_PACKAGE_ENV) or _WEBUI_PACKAGE
def run_webui(config: Any, workspace_dir: str | None = None) -> None:
"""Start the deploy-style backend + the WebUI front-end, then block.
@@ -203,6 +209,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
"PORT": str(webui_port),
}
)
webui_package = _resolve_webui_package()
console.print(
Panel(
Text.from_markup(
@@ -211,7 +218,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
f"[bold]WebUI:[/bold] http://localhost:{webui_port} "
f"[dim](opens in your browser)[/dim]\n"
f"[bold]Logs:[/bold] {_shorten(str(RUNTIME.log_file))}\n\n"
f"[dim]Fetching {_WEBUI_PACKAGE} via npx (first run may take a "
f"[dim]Fetching {webui_package} via npx (first run may take a "
f"moment)… Press Ctrl+C to stop.[/dim]"
),
title="[bold green]✓ EvoScientist WebUI[/bold green]",
@@ -228,7 +235,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
popen_kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP
try:
webui_proc = subprocess.Popen(
[npx, "--yes", _WEBUI_PACKAGE, "--port", str(webui_port)],
[npx, "--yes", webui_package, "--port", str(webui_port)],
**popen_kwargs,
)
except Exception as exc:
+281
View File
@@ -23,7 +23,17 @@ memory.
from __future__ import annotations
import asyncio
import json
import logging
import os
import sqlite3
from pathlib import Path
from typing import Any
import aiosqlite
import httpx
from langgraph.checkpoint.serde.jsonplus import JsonPlusSerializer
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
from starlette.applications import Starlette
from starlette.requests import Request
from starlette.responses import JSONResponse
@@ -31,6 +41,15 @@ from starlette.routing import Route
from EvoScientist.config import get_effective_config
from EvoScientist.llm.models import list_model_picker_entries
from EvoScientist.sessions import (
MAIN_THREAD_FILTER_PARAMS,
MAIN_THREAD_FILTER_SQL,
_load_checkpoint_messages,
_table_exists,
_to_short_path,
)
_logger = logging.getLogger(__name__)
async def get_models(_request: Request) -> JSONResponse:
@@ -75,8 +94,270 @@ async def get_models(_request: Request) -> JSONResponse:
)
def _message_type(message: Any) -> str | None:
if isinstance(message, dict):
role = message.get("role")
if role == "assistant":
return "ai"
if role == "user":
return "human"
value = message.get("type")
return str(value) if value is not None else None
value = getattr(message, "type", None)
return str(value) if value is not None else None
def _message_content(message: Any) -> Any:
if isinstance(message, dict):
return message.get("content")
return getattr(message, "content", None)
def _extract_text_content(content: Any) -> str:
if isinstance(content, str):
return content
if not isinstance(content, list):
return ""
parts: list[str] = []
for block in content:
if isinstance(block, str):
parts.append(block)
continue
if not isinstance(block, dict):
continue
block_type = block.get("type")
if block_type not in {"text", "output_text"}:
continue
text = block.get("text") or block.get("content")
if isinstance(text, str):
parts.append(text)
return "\n\n".join(part for part in parts if part)
def _is_tool_selection_payload(raw: str) -> bool:
try:
parsed = json.loads(raw)
except json.JSONDecodeError:
return False
return (
isinstance(parsed, dict)
and set(parsed) == {"tools"}
and isinstance(parsed["tools"], list)
and all(isinstance(tool, str) for tool in parsed["tools"])
)
def _split_json_objects(raw: str) -> list[str] | None:
objects: list[str] = []
depth = 0
start = -1
in_string = False
escaping = False
for i, char in enumerate(raw):
if in_string:
if escaping:
escaping = False
elif char == "\\":
escaping = True
elif char == '"':
in_string = False
continue
if char == '"':
if depth == 0:
return None
in_string = True
continue
if char == "{":
if depth == 0:
start = i
depth += 1
continue
if char == "}":
depth -= 1
if depth < 0 or start < 0:
return None
if depth == 0:
objects.append(raw[start : i + 1])
start = -1
continue
if depth == 0 and not char.isspace():
return None
if depth != 0 or in_string or not objects:
return None
return objects
def _is_tool_selection_text(text: str) -> bool:
stripped = text.strip()
if not stripped or '"tools"' not in stripped:
return False
if _is_tool_selection_payload(stripped):
return True
objects = _split_json_objects(stripped)
return objects is not None and all(_is_tool_selection_payload(obj) for obj in objects)
def _extract_final_answer(messages: list[Any]) -> str:
"""Return displayable text from the latest AI message in *messages*."""
for message in reversed(messages):
if _message_type(message) != "ai":
continue
content = _extract_text_content(_message_content(message)).strip()
if content and not _is_tool_selection_text(content):
return content
return ""
def _sessions_db_path_for_http() -> Path:
data_dir = os.getenv("EVOSCIENTIST_DATA_DIR")
base = Path(data_dir).expanduser() if data_dir else Path.home() / ".evoscientist"
return Path(_to_short_path(str(base))) / "sessions.db"
async def _get_thread_metadata_for_http(thread_id: str) -> dict | None:
try:
async with aiosqlite.connect(
str(_sessions_db_path_for_http()), timeout=30.0
) as conn:
if not await _table_exists(conn, "checkpoints"):
return None
query = f"""
SELECT json_extract(metadata, '$.workspace_dir') as workspace_dir,
json_extract(metadata, '$.model') as model,
json_extract(metadata, '$.updated_at') as updated_at
FROM checkpoints
WHERE thread_id = ?
AND {MAIN_THREAD_FILTER_SQL}
ORDER BY checkpoint_id DESC
LIMIT 1
"""
async with conn.execute(
query, (thread_id, *MAIN_THREAD_FILTER_PARAMS)
) as cur:
row = await cur.fetchone()
except (OSError, sqlite3.Error):
return None
if not row:
return None
return {
"workspace_dir": row[0],
"model": row[1],
"updated_at": row[2],
}
async def _get_thread_messages_for_http(thread_id: str) -> list:
try:
async with aiosqlite.connect(
str(_sessions_db_path_for_http()), timeout=30.0
) as conn:
if not await _table_exists(conn, "checkpoints"):
return []
check = f"""
SELECT 1 FROM checkpoints
WHERE thread_id = ? AND {MAIN_THREAD_FILTER_SQL}
LIMIT 1
"""
async with conn.execute(
check, (thread_id, *MAIN_THREAD_FILTER_PARAMS)
) as cur:
if not await cur.fetchone():
return []
serde = JsonPlusSerializer()
saver = AsyncSqliteSaver(conn, serde=serde)
return await _load_checkpoint_messages(saver, thread_id)
except (OSError, sqlite3.Error):
return []
async def _read_thread_runtime_state(request: Request, thread_id: str) -> dict[str, Any]:
"""Read thread status from the co-hosted langgraph-api endpoints."""
base_url = f"{request.url.scheme}://{request.url.netloc}"
timeout = httpx.Timeout(2.0, connect=0.5)
async with httpx.AsyncClient(base_url=base_url, timeout=timeout) as client:
thread_resp = await client.get(f"/threads/{thread_id}")
if thread_resp.status_code == 404:
return {"found": False}
thread_resp.raise_for_status()
thread = thread_resp.json()
state: dict[str, Any] = {}
state_resp = await client.get(f"/threads/{thread_id}/state")
if state_resp.status_code == 404:
return {"found": False}
if state_resp.status_code < 400:
state = state_resp.json()
next_nodes = state.get("next")
is_terminal_checkpoint = isinstance(next_nodes, (list, tuple)) and not next_nodes
status = thread.get("status")
complete = status == "idle" or is_terminal_checkpoint
completed_at = None
if complete:
completed_at = (
thread.get("state_updated_at")
or thread.get("updated_at")
or (state.get("metadata") or {}).get("updated_at")
)
return {
"found": True,
"complete": complete,
"completed_at": completed_at,
}
async def get_final_answer(request: Request) -> JSONResponse:
"""Return the latest checkpointed assistant answer for a WebUI thread.
This is a recovery surface for the WebUI stream consumer: when browser-side
SSE is interrupted but the langgraph run continues server-side, the final
answer is already persisted in ``sessions.db``. The route centralizes the
non-trivial "latest AIMessage text only" extraction so the browser does not
render reasoning blocks, tool calls, or tool-selection JSON fragments.
"""
thread_id = request.path_params["thread_id"]
metadata = await _get_thread_metadata_for_http(thread_id)
if metadata is None:
return JSONResponse({"error": "thread not found"}, status_code=404)
messages = await _get_thread_messages_for_http(thread_id)
content = _extract_final_answer(messages)
runtime: dict[str, Any] = {"found": True, "complete": False, "completed_at": None}
try:
runtime = await _read_thread_runtime_state(request, thread_id)
except Exception as exc:
_logger.debug(
"Could not read langgraph runtime state for thread %s: %s",
thread_id,
exc,
)
if runtime.get("found") is False:
return JSONResponse({"error": "thread not found"}, status_code=404)
completed_at = runtime.get("completed_at")
if completed_at is None and runtime.get("complete"):
completed_at = metadata.get("updated_at")
return JSONResponse(
{
"content": content,
"completed_at": completed_at,
"complete": bool(runtime.get("complete")),
}
)
app = Starlette(
routes=[
Route("/api/models", get_models, methods=["GET"]),
Route(
"/api/threads/{thread_id}/final-answer",
get_final_answer,
methods=["GET"],
),
]
)
+9
View File
@@ -474,6 +474,15 @@ def get_chat_model(
# Even native thinking models like kimi-k2-thinking operate in non-thinking mode.
if provider == "moonshot":
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
# custom-openai: some Codex-orientated reseller gateways (e.g.
# hunnuapi.top) blacklist the openai SDK's default
# "OpenAI/Python" User-Agent at the app layer and return
# 403 "Your request was blocked." to any langchain-openai client.
# Spoof a Codex-CLI UA so requests go through.
if provider == "custom-openai":
kwargs.setdefault("default_headers", {}).setdefault(
"User-Agent", "codex_cli_rs/0.0.0"
)
provider = "openai"
# OpenRouter → native ChatOpenRouter via init_chat_model.
+31 -1
View File
@@ -10,7 +10,7 @@
<a href="https://pypi.org/project/EvoScientist/"><picture>
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg">
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-dark.svg">
<img alt="PyPI v0.2.1" src="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg" height="28">
<img alt="PyPI v0.2.2" src="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg" height="28">
</picture></a><a href="https://EvoScientist.github.io/"><picture>
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-light.svg">
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-dark.svg">
@@ -137,6 +137,34 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
> [!TIP]
> Looking for ready-to-use research skills? Check out [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) — powered by [**EvoScientist**](https://github.com/EvoScientist/EvoScientist)'s engine and installable skills, the entire end-to-end research lifecycle is covered out of the box. [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) are also compatible with other CLI coding agents.
## WebUI Streaming Fix Notes
The WebUI streaming fix is based on one rule: a browser-side SSE disconnect is
only a transport event, not proof that the LangGraph run has finished.
- Keep the composer action button as **Stop** while the backend thread still has
pending work. The UI should not switch back to **Send** merely because the
live SSE stream ended.
- Derive the visible busy state from both the live stream and backend thread
state. In the patched WebUI this is represented as `isRunActive`, backed by
`threads.getState(threadId)`.
- Restore **Send** only after the backend reports a terminal state: idle,
completed final answer, failed, or cancelled.
- Preserve human-in-the-loop controls during tool approvals. Approval buttons
remain actionable, while the bottom composer still shows **Stop** to avoid
duplicate submissions.
- Treat leaked `{"tools":[...]}` payloads as internal tool-selection control
messages, not assistant text, and filter them from the transcript.
- Recover dropped long-response tails through the checkpoint-backed
`/api/threads/{thread_id}/final-answer` endpoint instead of relying only on
browser stream state.
- Use a long enough final-answer polling window for real research runs; short
fallback windows can expire before the backend completes and make the browser
appear truncated until refresh.
- Validate with real browser flows: normal streaming, simulated SSE disconnect
and reload, pending tool approval, rejection/cancellation, and final transition
back to **Send**.
## 🔥 News
- **[03 Jun 2026]** 🥈 Ranked #2 overall — and 🥇 #1 among `GPT-5.4`-based agents — on [ResearchClawBench](https://github.com/InternScience/ResearchClawBench) (Agent Mode)! [**Leaderboard**](https://internscience.github.io/ResearchClawBench-Home/) 👈
- **[18 Apr 2026]** 🥇 Ranked #1 on [DeepResearch Bench](https://deepresearch-bench.github.io/) at submission time! [**Leaderboard**](https://huggingface.co/spaces/muset-ai/DeepResearch-Bench-Leaderboard) 👈
@@ -151,6 +179,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
<details>
<summary>📦 Release Highlights — version changelog</summary>
- **[07 Jul 2026]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — WebUI streaming resilience hotfix: SSE disconnects no longer imply run completion; the composer stays on **Stop** until backend thread state is terminal; final-answer checkpoint recovery fills dropped response tails; tool-selection JSON payloads are filtered from live transcripts; `EVOSCIENTIST_WEBUI_PACKAGE` lets the launcher run a local patched WebUI package for validation before npm publication.
- **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5.
- **[26 Jun 2026]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — Scheduled tasks: cron-style recurring runs via `/schedule` or natural language, run unattended with shell-access gating; self-linking memory that connects observations into a knowledge graph (complements / contradicts / supersedes); a read-only `GET /api/models` endpoint for the WebUI model picker.
- **[23 Jun 2026]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — Hotfix for fresh installs: the first message crashed with `The subagent `task` tool cannot be exposed via `ptc`` after deepagents 0.6.11 / langchain-quickjs 0.3 reserved `task` as the REPL global. Removed `task` from the code-interpreter PTC allowlist (`task()` stays available as the REPL global; async dispatch stays in PTC) and pinned `deepagents[quickjs]~=0.6.11`.
@@ -180,6 +209,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
- [📦 Installation](#-installation)
- [🔑 Configuration](#-configuration)
- [⚡ Quick Start](#-quick-start)
- [WebUI Streaming Fix Notes](#webui-streaming-fix-notes)
- [⏰ Scheduled Tasks](#-scheduled-tasks)
- [🍪 Examples & Recipes](#-examples--recipes)
- [🔌 MCP Integration](#-mcp-integration)
@@ -0,0 +1,303 @@
# TUI as Primary Interface + Run Resilience
**Date:** 2026-07-06
**Status:** Design — pending implementation plan
**Scope:** EvoScientist CLI/TUI launch path, run-interruption recovery, WebUI opt-in hardening
## 1. Background & Root Cause
A user running the WebUI backend (`ui_backend = "webui"`) saw a long research run
(127 s) render truncated output in the browser, ending mid-token at `O-8282d7e`.
Investigation established:
- **The server-side run completed cleanly.** The thread checkpoint in
`~/.evoscientist/sessions.db` holds the full 6818-char final answer
(`status: idle`, `error: None`). The text `O-8282d7e` does not appear at the
end of any stored content — it is a mid-stream artifact.
- **The disconnect is browser-side.** `langgraph dev` sends an SSE heartbeat
every 5 s (`langgraph_api/sse.py:102`), `send_timeout` is `None` (never kills
slow sends), and the server supports resume via the `last-event-id` header
(`langgraph_api/api/runs.py:623`). The `runs/stream` endpoint is **POST**, so
the browser uses `fetch()` to read it; native `EventSource` auto-reconnect does
not apply, and the `@evoscientist/webui` front-end does not re-POST with
`last-event-id` to resume.
- **`stream_resumable=True`** lets the server-side run finish after the browser
drops, which is why "server succeeded" ≠ "browser received".
- A separate symptom from the same session — `Port 6174 cannot be bound` — was a
leftover `langgraph dev` process (PID 15288) whose `atexit` cleanup never ran
(parent killed hard).
**Conclusion:** the SSE disconnect is structural to the WebUI path. The TUI path
does not use SSE at all — `tui_interactive.py:297` calls `create_runtime_gateways()`
with no `backend`, so it defaults to `backend="local"` (`gateway/runtime.py`),
which runs the compiled graph **in-process** and is immune.
## 2. Goals
1. **Make the in-process TUI the primary interface.** Long runs no longer touch
SSE, so the class of failure the user hit cannot occur.
2. **Guarantee the user always sees the complete answer.** Even if a TUI run is
interrupted (Ctrl-C, crash, terminal close), the next invocation can surface
the latest persisted assistant message and optionally resume.
3. **Keep the WebUI available as opt-in** (`--web`), with the launch path
hardened against orphaned processes and port conflicts.
4. **Eliminate the port-6174 / orphaned-`langgraph dev` failure mode** for the
default (TUI) path: it must not start `langgraph dev` at all.
## 3. Non-Goals
- Patching `langgraph_api` upstream.
- Replacing the WebUI front-end's fetch-SSE transport. (`@evoscientist/webui`
lives in a separate npm repo with its own release cycle; a client-side resume
fix is noted as future work, not in scope here.)
- Changing the graph, middleware, or checkpointer semantics.
- Touching the `EvoSci deploy` standalone-server command.
## 4. Architecture
Three launch modes, selected by a single source of truth:
| Mode | `ui_backend` | Runs graph via | Starts `langgraph dev`? |
|---|---|---|---|
| **TUI (default)** | `tui` | `LocalGraphGateway` (in-process) | No |
| CLI | `cli` | `LocalGraphGateway` (in-process) | No |
| WebUI (opt-in) | `webui` | `langgraph dev` over HTTP/SSE | Yes |
A new `--web` flag on the top-level `evoscientist` command forces `webui` for
that invocation regardless of config; `--tui` forces `tui`. This lets users keep
`ui_backend = "webui"` in config but drop into the TUI without editing config,
and vice-versa.
### 4.1 Launch-mode resolution (single source of truth)
Today the default is inconsistent: `config/settings.py:279` defaults `ui_backend`
to `"tui"`, but `cli/tui_runtime.py:12` has `DEFAULT_UI_BACKEND = "cli"`. This
design collapses to one rule, evaluated in order:
1. `--web` flag → `webui`
2. `--tui` flag → `tui`
3. `config.ui_backend` (resolved via existing `resolve_ui_backend`)
4. `"tui"` (hardcoded final fallback)
`resolve_ui_backend` (`cli/tui_runtime.py:41`) stays the single normalizer; the
new flags feed into it rather than bypassing it.
### 4.2 Why TUI is immune (and what can still interrupt it)
In-process execution via `stream_agent_events` has no network hop. The only
"loss" modes are process-level:
- Ctrl-C mid-run
- Terminal close / SIGHUP
- Python crash / OOM
Because the checkpointer is `AsyncSqliteSaver` writing to
`~/.evoscientist/sessions.db` (`sessions.py:108` `get_db_path`), every completed
super-step is on disk before the next begins. An interrupted run therefore
leaves a recoverable checkpoint whose `next` tuple is non-empty (pending nodes)
and whose latest assistant message — if the model node finished — is the same
content the WebUI failed to deliver.
## 5. Components
### C1 — Launch-mode flags + TUI default
**Files:**
- `EvoScientist/cli/__init__.py` (Typer app entry) — add `--web` / `--tui`
options on the root command; thread the resolved `ui_backend` into the existing
dispatch.
- `EvoScientist/cli/tui_runtime.py:12` — remove the divergent
`DEFAULT_UI_BACKEND = "cli"`; route through `resolve_ui_backend` so
`settings.py`'s `"tui"` default wins.
- `EvoScientist/cli/commands.py` — wherever `ui_backend` is currently read, accept
the flag override.
**Behavior:** Running `evoscientist` with no flags and default config launches
the TUI and starts **no** `langgraph dev` subprocess. No port is bound; the
orphan-process class of bug disappears for the default path.
**Acceptance:** `evoscientist` (no args, default config) does not open a
listening socket on 6174 or any other port.
### C2 — TUI run resilience (Approach B)
Two cooperating pieces, both backed by `sessions.db`.
#### C2a — Interrupt-fallback in the streaming loop
Wrap the TUI streaming consumer so that on any interruption
(`KeyboardInterrupt`, `asyncio.CancelledError`, unexpected `Exception`) it reads
the thread's latest checkpoint and renders the latest assistant message before
exiting.
**Files:**
- `EvoScientist/stream/display.py` — `_run_streaming` and
`_astream_to_console` (the two streaming entry points used by
`cli/tui_backends.py`). Add a `finally`/except guard that, after an
interruption, calls a new `render_latest_assistant(thread_id)` helper.
- `EvoScientist/sessions.py` — add `get_latest_assistant_message(thread_id) ->
str | None`: read the most recent checkpoint for the thread, walk `messages`
backwards to the last `AIMessage` with non-empty content, return its text
(joined content blocks). Reuses existing `AsyncSqliteSaver` access already in
this module.
**Behavior on Ctrl-C mid-run:**
1. The guard catches the interrupt.
2. Prints a one-line notice: `Run interrupted — recovering latest output from
checkpoint…`.
3. Renders the latest persisted assistant message via the same Markdown renderer
the normal completion path uses.
4. If no assistant message is persisted yet (model node hadn't completed),
prints `Run interrupted before any output was saved. Resume with: evoscientist
--thread <id>` and exits.
**Non-goal:** this does not *continue* the run; it only guarantees visibility of
whatever completed. Continuation is C2b.
#### C2b — Startup detection of interrupted runs
Runs only when the TUI is launched to **resume a specific thread** (via the
existing thread-resume flag, i.e. the command `resume_hint.py` already prints at
exit). For a brand-new session there is nothing to detect and the prompt is
skipped. When resuming, check the thread's latest checkpoint: `next` is
non-empty **and** no run is currently active for that thread. If so, prompt:
```
You have an unfinished run in thread <id>:
[s] show what completed
[r] resume the run from its last checkpoint
[n] start fresh
```
**Files:**
- `EvoScientist/sessions.py` — add `detect_interrupted_run(thread_id) ->
InterruptedRunInfo | None` returning the checkpoint id, the pending node names
(`next`), and the timestamp. Implemented as a thin read over the existing
checkpointer's `aget_tuple`.
- `EvoScientist/cli/tui_interactive.py` — near the existing thread-load path
(around the `create_runtime_gateways()` call at line 297), invoke the detector
for the active thread and render the prompt above.
**Resume semantics:** selecting `r` re-invokes the graph with the same
`thread_id`. LangGraph resumes from the last checkpoint automatically (this is
the same mechanism `evoscientist --thread <id>` already relies on for session
continuity). No new graph code needed.
**Acceptance:** After killing a TUI mid-run with Ctrl-C, relaunching with
`--thread <id>` offers `[s]`/`[r]`/`[n]`; `[s]` prints the persisted final
answer; `[r]` continues execution from where it stopped.
### C3 — WebUI mode hardening (`--web`)
Applies only when `ui_backend == "webui"`. The goal is to make the existing
`deploy/webui.py` launch path robust and to mitigate (not eliminate) the SSE
limitation.
**Files:**
- `EvoScientist/deploy/webui.py` — three changes:
1. **Process-group + signal cleanup.** Today `atexit.register(stop_langgraph_dev,
...)` only fires on clean exit. Wrap the spawned `langgraph dev` (and the
`npx` child) in a process group (`start_new_session=True` on POSIX; on
Windows use a job object via the existing `_winloop.py` helpers) and
install `SIGINT`/`SIGTERM` handlers that tear the group down. This closes
the orphan-PID-15288 path for normal terminations. `SIGKILL` cannot run
handlers on either platform — document that hard kills may still orphan
the child, and the port-conflict UX in (2) is the user-facing recovery.
2. **Port-conflict UX.** The current "Port X is occupied by another process"
message is already correct; extend it to detect *whether the occupant is a
`langgraph dev`* (via the existing `/ok` health probe) and, if so, offer to
reuse it rather than bail. If the occupant is foreign, print the `lsof`
hint already present.
3. **Resume window.** Set `RESUMABLE_STREAM_TTL_SECONDS=600` in the env passed
to `start_langgraph_dev` so a browser that does reconnect can replay events
for runs up to 10 minutes long (the observed run was 127 s vs. the 120 s
default).
- `EvoScientist/deploy/server.py` — `start_langgraph_dev` already accepts env;
pass `RESUMABLE_STREAM_TTL_SECONDS` through.
**What C3 does NOT do:** it cannot make the browser auto-resume, because that
logic lives in `@evoscientist/webui`. It only widens the server-side window and
stops the orphan process. Documented in the WebUI exit message: *for long runs,
prefer the TUI (`evoscientist --tui`); the browser client does not resume a
dropped stream.*
### C4 — Documentation
- `README` / CLI `--help`: `--web` (browser), `--tui` (terminal, default,
recommended for long runs).
- A short "Why did my WebUI output stop?" note pointing at this design doc, so
future users reading the symptom can find the explanation.
## 6. Data Flow
### Default (TUI) path
```
evoscientist ─▶ resolve_ui_backend ─▶ "tui"
│
(no langgraph dev, no port bound)
▼
create_runtime_gateways(backend="local")
│
LocalGraphGateway.stream_events
│
stream_agent_events (in-process) ── checkpoint per super-step ──▶ sessions.db
│
on interrupt ──▶ get_latest_assistant_message(thread_id) ──▶ render
```
### WebUI (`--web`) path
unchanged from today (langgraph dev + `npx @evoscientist/webui`), plus C3's
process-group cleanup and `RESUMABLE_STREAM_TTL_SECONDS=600`.
## 7. Error Handling
| Failure | TUI default path | WebUI `--web` path |
|---|---|---|
| Ctrl-C mid-run | C2a renders latest persisted assistant message | Browser truncates; user runs `evoscientist --tui --thread <id>` → `[s]` to recover |
| Crash / SIGHUP | Checkpoint on disk; next `--thread <id>` shows `[s]`/`[r]` prompt (C2b) | Same recovery via TUI |
| Port 6174 bound | N/A — no port bound | C3 reuses if `langgraph dev`, else prints `lsof` hint |
| Orphaned `langgraph dev` | Cannot originate from TUI path | C3 process-group + signal handlers tear it down |
## 8. Testing
- **Unit** (`tests/test_sessions_*.py`): `get_latest_assistant_message` returns
the full final answer for a thread whose last run completed; returns `None`
for a thread interrupted before the model node.
- **Unit**: `detect_interrupted_run` returns pending `next` nodes for an
interrupted thread; `None` for an idle one.
- **Integration** (new `tests/test_tui_interrupt_recovery.py`): start a TUI run
against a graph whose model node sleeps, send `SIGINT` mid-run, assert the
process exits 0 after printing the recovered message.
- **Integration**: relaunch with `--thread <id>`, assert the `[s]/[r]/[n]`
prompt appears and `[r]` produces a completed run.
- **Launch** (`tests/test_cli_launch.py`, extend): `evoscientist` with default
config binds no port; `--web` binds 6174 and tears it down on `SIGTERM`.
- **WebUI** (manual / smoke): `--web`, kill parent with `SIGKILL`, assert no
orphan `langgraph dev` remains after the process-group change (best-effort —
`SIGKILL` cannot run handlers, but the process group lets the OS reap
children when the session ends; document this limit).
## 9. Rollout / Migration
- User's config currently has `ui_backend = "webui"`. After this change, that
config still selects WebUI (back-compatible). The user can run `evoscientist
--tui` immediately to get the resilient path, or `EvoSci config set ui_backend
tui` to make it permanent.
- No migration of `sessions.db` — schema is unchanged.
- No breaking change to `EvoSci deploy`.
## 10. Open Questions / Future Work
- **`@evoscientist/webui` client-side resume.** The front-end could re-POST
`/runs/stream` with `last-event-id` (or use the thread's SSE endpoint with
`since`) after a disconnect. This is the only way to fully fix the WebUI
symptom; it lives in the npm repo and is out of scope here. File a
cross-repo issue referencing this design.
- Whether `[r]` resume should replay the *interrupted* model call or only
continue from the *next* node. LangGraph's default (resume from pending
writes) is correct for most cases; revisit if users hit re-execution of
expensive tool calls.
- `cli` backend vs `tui` backend distinction — both are in-process; consider
deprecating the `cli`/`tui` split in a follow-up once the resilience work
lands, to reduce the three-way default confusion that caused the original
`DEFAULT_UI_BACKEND` divergence.
@@ -0,0 +1,196 @@
# WebUI SSE Truncation Recovery via Checkpoint Fallback (α)
**Date:** 2026-07-06
**Status:** Design — pending implementation plan
**Scope:** `@evoscientist/webui` front-end (separate npm repo) + one optional convenience route in this Python repo
## 1. Background & Root Cause
A WebUI user saw a 127 s research run render truncated output, ending mid-token at
`O-8282d7e`. Investigation established:
- The server-side run **completed cleanly**; the thread checkpoint in
`~/.evoscientist/sessions.db` holds the full 6818-char final answer
(`status: idle`, `error: None`).
- The disconnect is **browser-side**: `langgraph dev` sends an SSE heartbeat every
5 s, `send_timeout` is `None`, and the server supports resume via
`last-event-id` (`langgraph_api/api/runs.py:623`). But `/runs/stream` is
**POST**, so the browser reads it via `fetch()`; native `EventSource`
auto-reconnect does not apply, and `@evoscientist/webui` does not re-POST to
resume. When the connection drops mid-run, the front-end is left showing the
last partial token and never fetches the completed answer.
**Core insight:** the server already has the complete answer in the checkpoint.
The fix is to make the front-end fall back to that checkpoint when the stream
ends abnormally. This is small, surgical, and directly addresses the symptom.
## 2. Goal
**The user always sees the complete final answer in the WebUI, even if the SSE
stream drops mid-run.**
## 3. Non-Goals
- Seamless stream resume (tracking `last-event-id` / `seq` and replaying
buffered events). That is option β — better UX, ~3–5× the code, deferred.
- The TUI pivot / launch-mode decoupling (option γ). Separable; only relevant if
the port-conflict / orphan-process issues need addressing.
- Patching `langgraph_api` upstream.
## 4. Detection Logic — What Counts as "Truncated"
The langgraph SSE protocol emits terminal events when a run finishes
(success → `end`, failure → `error`; exact names to be confirmed against the
`@langchain/langgraph-sdk` version during implementation). The front-end's
streaming reader loop currently processes events until the `fetch()` ReadableStream
closes.
**Truncation** = the stream closed (reader returned) **without** a terminal event
having been received. Causes: browser tab throttled, network blip, proxy idle
timeout. On truncation, the in-progress assistant bubble is left showing a prefix
of the real answer.
## 5. Components
### F1 — Truncation detector (front-end)
In the streaming reader loop, track a `terminalSeen` flag. Set it when a
terminal event arrives. When the reader returns, if `!terminalSeen`, mark the
run `truncated` and trigger F2.
**File:** `@evoscientist/webui` — the run-stream consumer (the module that wraps
the langgraph-ts SDK's `runs.stream` and renders into the message list).
### F2 — Checkpoint fallback fetch (front-end)
On `truncated`, fetch the complete final assistant message and **replace** the
in-progress bubble's content with it (not append — the partial tokens are a
prefix of the same message).
Two implementation choices, pick one:
- **F2a (zero server change):** call the standard langgraph endpoint
`GET /threads/{thread_id}/state`, walk `values.messages` backwards to the last
`AIMessage`, join its content blocks. No new route, but the front-end parses
raw state and reimplements "find latest assistant text" logic.
- **F2b (recommended, uses S1 below):** call `GET /api/threads/{id}/final-answer`
→ `{content, completed_at, complete}`. Server owns the parsing; front-end
stays dumb.
Render a subtle affordance (e.g., a dim "⚠ stream dropped — recovered from
checkpoint" line above the message) so the user knows a disconnect happened and
the displayed text is authoritative, not stale streaming.
### F3 — Retry until the run settles (front-end)
The browser may drop while the run is **still executing** server-side. A single
fallback fetch at drop-time could return a not-yet-final message. So F2 retries
with backoff until either:
- `complete == true` (run finished — render and stop), or
- a max wait elapses (default 240 s — comfortably exceeds the observed 127 s
run even if the drop happens at t=0), then render whatever is latest and
stop retrying.
Backoff: poll at 1 s, 2 s, 4 s, then every 5 s up to the max. Each poll replaces
the bubble with the latest content, so if the run finishes mid-retry the user
sees it update to the final form.
### S1 — Convenience route `GET /api/threads/{id}/final-answer` (this repo, optional but recommended)
Enables F2b. Mounted on the existing custom ASGI app already wired via
`langgraph.json` → `EvoScientist.langgraph_dev.http:app` (currently only serves
`/api/models`).
**Contract:**
```
GET /api/threads/{thread_id}/final-answer
200 → { "content": str, # last AIMessage text, blocks joined
"completed_at": iso8601, # run end timestamp, or null
"complete": bool } # true iff run reached terminal state
404 → thread not found
```
**`complete` derivation** (run reached a terminal state):
`thread.status == "idle"` **or** latest checkpoint's `next` tuple is empty.
**File:** `EvoScientist/langgraph_dev/http.py` — add `get_final_answer` route
beside the existing `get_models`. Implementation reads the thread state via the
shared checkpointer (`AsyncSqliteSaver` at `~/.evoscientist/sessions.db`,
already used by `sessions.py`). If direct checkpointer access is awkward from
inside the mounted app, fall back to an internal `langgraph_sdk.get_client()`
call to `GET /threads/{id}/state` on localhost — the contract stays identical.
**Why prefer F2b+S1 over F2a:** the "find latest AIMessage + join content blocks
+ decide complete?" logic is non-trivial (content is a list of typed blocks,
including `reasoning` blocks that should be excluded from the rendered answer —
see the checkpoint: block[0] was `reasoning`, block[1] was `text`). Centralizing
it server-side keeps the front-end thin and makes the same logic reusable by the
TUI's future recovery path if γ is ever pursued.
## 6. Data Flow
```
browser fetch(/runs/stream, POST) ──SSE──▶ langgraph dev
│ │
│ (mid-run, connection drops) │ run continues to completion
▼ │
reader returns, no terminal event ──▶ truncated │
│ │
│ GET /api/threads/{id}/final-answer │
▼ ▼
http.py::get_final_answer ──read checkpoint──▶ sessions.db
│
▼
{content, complete} ──▶ if !complete, retry (F3); else render (F2)
```
## 7. Error Handling
| Case | Behavior |
|---|---|
| Stream ends normally (`end`/`error` received) | F1 sets `terminalSeen`; no fallback; existing render path unchanged |
| Stream drops, run already finished server-side | First fallback fetch returns `complete=true`; render immediately |
| Stream drops, run still executing | F3 retries with backoff; each retry renders latest; stops when `complete` or 120 s max |
| Thread unknown / deleted | 404; front-end shows "stream interrupted and the session could not be recovered" |
| Convenience route unreachable (older server without S1) | Front-end falls back to F2a (raw state endpoint) — degrade gracefully |
## 8. Testing
**Front-end (`@evoscientist/webui`):**
- Unit: feed the reader a synthetic event stream that closes without a terminal
event → assert `truncated == true`; feed one with `end` → `false`.
- Integration: mock `fetch` to drop mid-stream; assert the fallback fetch fires
and the bubble's content is replaced with the checkpoint value.
**Server (this repo, if S1):** new `tests/test_http_final_answer.py`
- Completed thread → 200, `complete=true`, content matches the last AIMessage
text (not the reasoning block).
- Mid-run thread (forced `busy` / non-empty `next`) → 200, `complete=false`.
- Unknown thread → 404.
- Content with mixed `reasoning` + `text` blocks → only `text` in `content`.
## 9. Cross-Repo Coordination & Rollout
- **Order:** land S1 in this Python repo first (route + tests, behind no flag —
it's a pure addition). Then F1/F2/F3 in `@evoscientist/webui`.
- **Versioning:** the WebUI launches via `npx @evoscientist/webui@latest`, so
the front-end fix reaches users on their next launch automatically once
published. No coordinated upgrade required on the Python side.
- **Backward compatibility:** if a user runs a new front-end against an older
Python server without S1, the front-end must detect the 404 and fall back to
F2a (raw state endpoint). If an old front-end runs against a new server, S1
is simply unused — no harm.
## 10. Open Questions / Future Work
- **Exact terminal-event names** for the SDK version in use (`end`/`error` vs.
`done`/`result`). Confirm against `@langchain/langgraph-sdk` during
implementation; the detector is parameterized on this.
- **True resume (β):** track `last-event-id` (or the V2 thread-stream `seq`) and
replay buffered events on reconnect — seamless, no visible "recovered" flash.
Larger front-end effort; defer until α is validated.
- **TUI pivot (γ):** separable work to make the in-process TUI the default and
eliminate the port-conflict / orphan-process issues. Independent spec if
pursued.
+1 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "EvoScientist"
version = "0.2.1"
version = "0.2.2"
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
readme = "README.md"
requires-python = ">=3.11"
+141
View File
@@ -7,6 +7,7 @@ from __future__ import annotations
from unittest.mock import patch
from langchain_core.messages import AIMessage, HumanMessage
from starlette.testclient import TestClient
from EvoScientist.config import EvoScientistConfig
@@ -161,3 +162,143 @@ def test_ollama_discovery_skipped_when_base_url_absent():
{"name": n, "model_id": m, "provider": p}
for n, m, p in list_models_by_provider()
]
def test_final_answer_extracts_latest_ai_text_blocks():
async def fake_metadata(_thread_id):
return {"updated_at": "2026-07-06T14:14:53+00:00"}
async def fake_messages(_thread_id):
return [
HumanMessage(content="question"),
AIMessage(content="old answer"),
AIMessage(
content=[
{"type": "reasoning", "text": "internal"},
{"type": "text", "text": "Part A"},
{"type": "tool_use", "name": "search"},
{"type": "output_text", "text": "Part B"},
]
),
]
async def fake_runtime(_request, _thread_id):
return {
"found": True,
"complete": True,
"completed_at": "2026-07-06T14:15:00+00:00",
}
with (
patch(
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
new=fake_metadata,
),
patch(
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
new=fake_messages,
),
patch(
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
new=fake_runtime,
),
):
resp = client.get("/api/threads/thread-1/final-answer")
assert resp.status_code == 200
assert resp.json() == {
"content": "Part A\n\nPart B",
"completed_at": "2026-07-06T14:15:00+00:00",
"complete": True,
}
def test_final_answer_skips_tool_selection_json_text():
async def fake_metadata(_thread_id):
return {"updated_at": "2026-07-06T14:14:53+00:00"}
async def fake_messages(_thread_id):
return [
HumanMessage(content="question"),
AIMessage(content="stable answer"),
AIMessage(
content=(
'{"tools":["search_papers","get_abstract"]}'
'{"tools":["web_search_exa"]}'
)
),
]
async def fake_runtime(_request, _thread_id):
return {
"found": True,
"complete": True,
"completed_at": "2026-07-06T14:15:00+00:00",
}
with (
patch(
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
new=fake_metadata,
),
patch(
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
new=fake_messages,
),
patch(
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
new=fake_runtime,
),
):
resp = client.get("/api/threads/thread-1/final-answer")
assert resp.status_code == 200
assert resp.json()["content"] == "stable answer"
def test_final_answer_returns_404_for_unknown_thread():
async def fake_metadata(_thread_id):
return None
with patch(
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
new=fake_metadata,
):
resp = client.get("/api/threads/missing/final-answer")
assert resp.status_code == 404
assert resp.json() == {"error": "thread not found"}
def test_final_answer_does_not_mark_complete_when_runtime_state_fails():
async def fake_metadata(_thread_id):
return {"updated_at": "2026-07-06T14:14:53+00:00"}
async def fake_messages(_thread_id):
return [AIMessage(content="checkpoint answer")]
async def fake_runtime(_request, _thread_id):
raise RuntimeError("langgraph runtime unavailable")
with (
patch(
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
new=fake_metadata,
),
patch(
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
new=fake_messages,
),
patch(
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
new=fake_runtime,
),
):
resp = client.get("/api/threads/thread-1/final-answer")
assert resp.status_code == 200
assert resp.json() == {
"content": "checkpoint answer",
"completed_at": None,
"complete": False,
}
+15
View File
@@ -0,0 +1,15 @@
from __future__ import annotations
from EvoScientist.deploy.webui import _resolve_webui_package
def test_resolve_webui_package_defaults_to_published_package(monkeypatch):
monkeypatch.delenv("EVOSCIENTIST_WEBUI_PACKAGE", raising=False)
assert _resolve_webui_package() == "@evoscientist/webui@latest"
def test_resolve_webui_package_allows_local_override(monkeypatch):
monkeypatch.setenv("EVOSCIENTIST_WEBUI_PACKAGE", "/tmp/EvoScientist-WebUI")
assert _resolve_webui_package() == "/tmp/EvoScientist-WebUI"
Generated
+1 -1
View File
@@ -944,7 +944,7 @@ wheels = [
[[package]]
name = "evoscientist"
version = "0.2.0"
version = "0.2.2"
source = { editable = "." }
dependencies = [
{ name = "deepagents", extra = ["quickjs"] },