From e0acc6155edd83ed26d10c9d5f20fbec5efbd430 Mon Sep 17 00:00:00 2001 From: m4 Date: Fri, 10 Jul 2026 17:35:44 +0800 Subject: [PATCH] feat: improve WebUI run recovery --- .gitignore | 1 + EvoScientist/deploy/webui.py | 11 +- EvoScientist/langgraph_dev/http.py | 281 ++++++++++++++++ EvoScientist/llm/models.py | 9 + README.md | 32 +- ...07-06-tui-primary-run-resilience-design.md | 303 ++++++++++++++++++ ...06-webui-sse-checkpoint-fallback-design.md | 196 +++++++++++ pyproject.toml | 2 +- tests/test_langgraph_dev_http.py | 141 ++++++++ tests/test_webui_launcher.py | 15 + uv.lock | 2 +- 11 files changed, 988 insertions(+), 5 deletions(-) create mode 100644 docs/superpowers/specs/2026-07-06-tui-primary-run-resilience-design.md create mode 100644 docs/superpowers/specs/2026-07-06-webui-sse-checkpoint-fallback-design.md create mode 100644 tests/test_webui_launcher.py diff --git a/.gitignore b/.gitignore index 9186f4f..ab4bb6a 100644 --- a/.gitignore +++ b/.gitignore @@ -48,3 +48,4 @@ conversation_history/ *meals/ botpy.log large_tool_results/ +runs/ diff --git a/EvoScientist/deploy/webui.py b/EvoScientist/deploy/webui.py index cd44fd1..12d750b 100644 --- a/EvoScientist/deploy/webui.py +++ b/EvoScientist/deploy/webui.py @@ -41,9 +41,15 @@ from ..stream.console import console # Front-end npm package + spec. ``@latest`` → always the newest published UI. _WEBUI_PACKAGE = "@evoscientist/webui@latest" +_WEBUI_PACKAGE_ENV = "EVOSCIENTIST_WEBUI_PACKAGE" _DEFAULT_WEBUI_PORT = 4716 +def _resolve_webui_package() -> str: + """Return the npm package spec to launch for the WebUI front-end.""" + return os.getenv(_WEBUI_PACKAGE_ENV) or _WEBUI_PACKAGE + + def run_webui(config: Any, workspace_dir: str | None = None) -> None: """Start the deploy-style backend + the WebUI front-end, then block. @@ -203,6 +209,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None: "PORT": str(webui_port), } ) + webui_package = _resolve_webui_package() console.print( Panel( Text.from_markup( @@ -211,7 +218,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None: f"[bold]WebUI:[/bold] http://localhost:{webui_port} " f"[dim](opens in your browser)[/dim]\n" f"[bold]Logs:[/bold] {_shorten(str(RUNTIME.log_file))}\n\n" - f"[dim]Fetching {_WEBUI_PACKAGE} via npx (first run may take a " + f"[dim]Fetching {webui_package} via npx (first run may take a " f"moment)… Press Ctrl+C to stop.[/dim]" ), title="[bold green]✓ EvoScientist WebUI[/bold green]", @@ -228,7 +235,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None: popen_kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP try: webui_proc = subprocess.Popen( - [npx, "--yes", _WEBUI_PACKAGE, "--port", str(webui_port)], + [npx, "--yes", webui_package, "--port", str(webui_port)], **popen_kwargs, ) except Exception as exc: diff --git a/EvoScientist/langgraph_dev/http.py b/EvoScientist/langgraph_dev/http.py index 6b83bd7..619c2ec 100644 --- a/EvoScientist/langgraph_dev/http.py +++ b/EvoScientist/langgraph_dev/http.py @@ -23,7 +23,17 @@ memory. from __future__ import annotations import asyncio +import json +import logging +import os +import sqlite3 +from pathlib import Path +from typing import Any +import aiosqlite +import httpx +from langgraph.checkpoint.serde.jsonplus import JsonPlusSerializer +from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver from starlette.applications import Starlette from starlette.requests import Request from starlette.responses import JSONResponse @@ -31,6 +41,15 @@ from starlette.routing import Route from EvoScientist.config import get_effective_config from EvoScientist.llm.models import list_model_picker_entries +from EvoScientist.sessions import ( + MAIN_THREAD_FILTER_PARAMS, + MAIN_THREAD_FILTER_SQL, + _load_checkpoint_messages, + _table_exists, + _to_short_path, +) + +_logger = logging.getLogger(__name__) async def get_models(_request: Request) -> JSONResponse: @@ -75,8 +94,270 @@ async def get_models(_request: Request) -> JSONResponse: ) +def _message_type(message: Any) -> str | None: + if isinstance(message, dict): + role = message.get("role") + if role == "assistant": + return "ai" + if role == "user": + return "human" + value = message.get("type") + return str(value) if value is not None else None + value = getattr(message, "type", None) + return str(value) if value is not None else None + + +def _message_content(message: Any) -> Any: + if isinstance(message, dict): + return message.get("content") + return getattr(message, "content", None) + + +def _extract_text_content(content: Any) -> str: + if isinstance(content, str): + return content + if not isinstance(content, list): + return "" + + parts: list[str] = [] + for block in content: + if isinstance(block, str): + parts.append(block) + continue + if not isinstance(block, dict): + continue + block_type = block.get("type") + if block_type not in {"text", "output_text"}: + continue + text = block.get("text") or block.get("content") + if isinstance(text, str): + parts.append(text) + return "\n\n".join(part for part in parts if part) + + +def _is_tool_selection_payload(raw: str) -> bool: + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return False + return ( + isinstance(parsed, dict) + and set(parsed) == {"tools"} + and isinstance(parsed["tools"], list) + and all(isinstance(tool, str) for tool in parsed["tools"]) + ) + + +def _split_json_objects(raw: str) -> list[str] | None: + objects: list[str] = [] + depth = 0 + start = -1 + in_string = False + escaping = False + + for i, char in enumerate(raw): + if in_string: + if escaping: + escaping = False + elif char == "\\": + escaping = True + elif char == '"': + in_string = False + continue + + if char == '"': + if depth == 0: + return None + in_string = True + continue + if char == "{": + if depth == 0: + start = i + depth += 1 + continue + if char == "}": + depth -= 1 + if depth < 0 or start < 0: + return None + if depth == 0: + objects.append(raw[start : i + 1]) + start = -1 + continue + if depth == 0 and not char.isspace(): + return None + + if depth != 0 or in_string or not objects: + return None + return objects + + +def _is_tool_selection_text(text: str) -> bool: + stripped = text.strip() + if not stripped or '"tools"' not in stripped: + return False + if _is_tool_selection_payload(stripped): + return True + objects = _split_json_objects(stripped) + return objects is not None and all(_is_tool_selection_payload(obj) for obj in objects) + + +def _extract_final_answer(messages: list[Any]) -> str: + """Return displayable text from the latest AI message in *messages*.""" + for message in reversed(messages): + if _message_type(message) != "ai": + continue + content = _extract_text_content(_message_content(message)).strip() + if content and not _is_tool_selection_text(content): + return content + return "" + + +def _sessions_db_path_for_http() -> Path: + data_dir = os.getenv("EVOSCIENTIST_DATA_DIR") + base = Path(data_dir).expanduser() if data_dir else Path.home() / ".evoscientist" + return Path(_to_short_path(str(base))) / "sessions.db" + + +async def _get_thread_metadata_for_http(thread_id: str) -> dict | None: + try: + async with aiosqlite.connect( + str(_sessions_db_path_for_http()), timeout=30.0 + ) as conn: + if not await _table_exists(conn, "checkpoints"): + return None + query = f""" + SELECT json_extract(metadata, '$.workspace_dir') as workspace_dir, + json_extract(metadata, '$.model') as model, + json_extract(metadata, '$.updated_at') as updated_at + FROM checkpoints + WHERE thread_id = ? + AND {MAIN_THREAD_FILTER_SQL} + ORDER BY checkpoint_id DESC + LIMIT 1 + """ + async with conn.execute( + query, (thread_id, *MAIN_THREAD_FILTER_PARAMS) + ) as cur: + row = await cur.fetchone() + except (OSError, sqlite3.Error): + return None + if not row: + return None + return { + "workspace_dir": row[0], + "model": row[1], + "updated_at": row[2], + } + + +async def _get_thread_messages_for_http(thread_id: str) -> list: + try: + async with aiosqlite.connect( + str(_sessions_db_path_for_http()), timeout=30.0 + ) as conn: + if not await _table_exists(conn, "checkpoints"): + return [] + check = f""" + SELECT 1 FROM checkpoints + WHERE thread_id = ? AND {MAIN_THREAD_FILTER_SQL} + LIMIT 1 + """ + async with conn.execute( + check, (thread_id, *MAIN_THREAD_FILTER_PARAMS) + ) as cur: + if not await cur.fetchone(): + return [] + serde = JsonPlusSerializer() + saver = AsyncSqliteSaver(conn, serde=serde) + return await _load_checkpoint_messages(saver, thread_id) + except (OSError, sqlite3.Error): + return [] + + +async def _read_thread_runtime_state(request: Request, thread_id: str) -> dict[str, Any]: + """Read thread status from the co-hosted langgraph-api endpoints.""" + base_url = f"{request.url.scheme}://{request.url.netloc}" + timeout = httpx.Timeout(2.0, connect=0.5) + async with httpx.AsyncClient(base_url=base_url, timeout=timeout) as client: + thread_resp = await client.get(f"/threads/{thread_id}") + if thread_resp.status_code == 404: + return {"found": False} + thread_resp.raise_for_status() + thread = thread_resp.json() + + state: dict[str, Any] = {} + state_resp = await client.get(f"/threads/{thread_id}/state") + if state_resp.status_code == 404: + return {"found": False} + if state_resp.status_code < 400: + state = state_resp.json() + + next_nodes = state.get("next") + is_terminal_checkpoint = isinstance(next_nodes, (list, tuple)) and not next_nodes + status = thread.get("status") + complete = status == "idle" or is_terminal_checkpoint + completed_at = None + if complete: + completed_at = ( + thread.get("state_updated_at") + or thread.get("updated_at") + or (state.get("metadata") or {}).get("updated_at") + ) + return { + "found": True, + "complete": complete, + "completed_at": completed_at, + } + + +async def get_final_answer(request: Request) -> JSONResponse: + """Return the latest checkpointed assistant answer for a WebUI thread. + + This is a recovery surface for the WebUI stream consumer: when browser-side + SSE is interrupted but the langgraph run continues server-side, the final + answer is already persisted in ``sessions.db``. The route centralizes the + non-trivial "latest AIMessage text only" extraction so the browser does not + render reasoning blocks, tool calls, or tool-selection JSON fragments. + """ + thread_id = request.path_params["thread_id"] + metadata = await _get_thread_metadata_for_http(thread_id) + if metadata is None: + return JSONResponse({"error": "thread not found"}, status_code=404) + + messages = await _get_thread_messages_for_http(thread_id) + content = _extract_final_answer(messages) + + runtime: dict[str, Any] = {"found": True, "complete": False, "completed_at": None} + try: + runtime = await _read_thread_runtime_state(request, thread_id) + except Exception as exc: + _logger.debug( + "Could not read langgraph runtime state for thread %s: %s", + thread_id, + exc, + ) + if runtime.get("found") is False: + return JSONResponse({"error": "thread not found"}, status_code=404) + + completed_at = runtime.get("completed_at") + if completed_at is None and runtime.get("complete"): + completed_at = metadata.get("updated_at") + return JSONResponse( + { + "content": content, + "completed_at": completed_at, + "complete": bool(runtime.get("complete")), + } + ) + + app = Starlette( routes=[ Route("/api/models", get_models, methods=["GET"]), + Route( + "/api/threads/{thread_id}/final-answer", + get_final_answer, + methods=["GET"], + ), ] ) diff --git a/EvoScientist/llm/models.py b/EvoScientist/llm/models.py index 9eeb1af..c63735e 100644 --- a/EvoScientist/llm/models.py +++ b/EvoScientist/llm/models.py @@ -474,6 +474,15 @@ def get_chat_model( # Even native thinking models like kimi-k2-thinking operate in non-thinking mode. if provider == "moonshot": kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"} + # custom-openai: some Codex-orientated reseller gateways (e.g. + # hunnuapi.top) blacklist the openai SDK's default + # "OpenAI/Python" User-Agent at the app layer and return + # 403 "Your request was blocked." to any langchain-openai client. + # Spoof a Codex-CLI UA so requests go through. + if provider == "custom-openai": + kwargs.setdefault("default_headers", {}).setdefault( + "User-Agent", "codex_cli_rs/0.0.0" + ) provider = "openai" # OpenRouter → native ChatOpenRouter via init_chat_model. diff --git a/README.md b/README.md index 55e8093..d9f9910 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ - PyPI v0.2.1 + PyPI v0.2.2 @@ -137,6 +137,34 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human > [!TIP] > Looking for ready-to-use research skills? Check out [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) — powered by [**EvoScientist**](https://github.com/EvoScientist/EvoScientist)'s engine and installable skills, the entire end-to-end research lifecycle is covered out of the box. [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) are also compatible with other CLI coding agents. +## WebUI Streaming Fix Notes + +The WebUI streaming fix is based on one rule: a browser-side SSE disconnect is +only a transport event, not proof that the LangGraph run has finished. + +- Keep the composer action button as **Stop** while the backend thread still has + pending work. The UI should not switch back to **Send** merely because the + live SSE stream ended. +- Derive the visible busy state from both the live stream and backend thread + state. In the patched WebUI this is represented as `isRunActive`, backed by + `threads.getState(threadId)`. +- Restore **Send** only after the backend reports a terminal state: idle, + completed final answer, failed, or cancelled. +- Preserve human-in-the-loop controls during tool approvals. Approval buttons + remain actionable, while the bottom composer still shows **Stop** to avoid + duplicate submissions. +- Treat leaked `{"tools":[...]}` payloads as internal tool-selection control + messages, not assistant text, and filter them from the transcript. +- Recover dropped long-response tails through the checkpoint-backed + `/api/threads/{thread_id}/final-answer` endpoint instead of relying only on + browser stream state. +- Use a long enough final-answer polling window for real research runs; short + fallback windows can expire before the backend completes and make the browser + appear truncated until refresh. +- Validate with real browser flows: normal streaming, simulated SSE disconnect + and reload, pending tool approval, rejection/cancellation, and final transition + back to **Send**. + ## 🔥 News - **[03 Jun 2026]** 🥈 Ranked #2 overall — and 🥇 #1 among `GPT-5.4`-based agents — on [ResearchClawBench](https://github.com/InternScience/ResearchClawBench) (Agent Mode)! [**Leaderboard**](https://internscience.github.io/ResearchClawBench-Home/) 👈 - **[18 Apr 2026]** 🥇 Ranked #1 on [DeepResearch Bench](https://deepresearch-bench.github.io/) at submission time! [**Leaderboard**](https://huggingface.co/spaces/muset-ai/DeepResearch-Bench-Leaderboard) 👈 @@ -151,6 +179,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
📦 Release Highlights — version changelog +- **[07 Jul 2026]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — WebUI streaming resilience hotfix: SSE disconnects no longer imply run completion; the composer stays on **Stop** until backend thread state is terminal; final-answer checkpoint recovery fills dropped response tails; tool-selection JSON payloads are filtered from live transcripts; `EVOSCIENTIST_WEBUI_PACKAGE` lets the launcher run a local patched WebUI package for validation before npm publication. - **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5. - **[26 Jun 2026]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — Scheduled tasks: cron-style recurring runs via `/schedule` or natural language, run unattended with shell-access gating; self-linking memory that connects observations into a knowledge graph (complements / contradicts / supersedes); a read-only `GET /api/models` endpoint for the WebUI model picker. - **[23 Jun 2026]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — Hotfix for fresh installs: the first message crashed with `The subagent `task` tool cannot be exposed via `ptc`` after deepagents 0.6.11 / langchain-quickjs 0.3 reserved `task` as the REPL global. Removed `task` from the code-interpreter PTC allowlist (`task()` stays available as the REPL global; async dispatch stays in PTC) and pinned `deepagents[quickjs]~=0.6.11`. @@ -180,6 +209,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human - [📦 Installation](#-installation) - [🔑 Configuration](#-configuration) - [⚡ Quick Start](#-quick-start) +- [WebUI Streaming Fix Notes](#webui-streaming-fix-notes) - [⏰ Scheduled Tasks](#-scheduled-tasks) - [🍪 Examples & Recipes](#-examples--recipes) - [🔌 MCP Integration](#-mcp-integration) diff --git a/docs/superpowers/specs/2026-07-06-tui-primary-run-resilience-design.md b/docs/superpowers/specs/2026-07-06-tui-primary-run-resilience-design.md new file mode 100644 index 0000000..e412b3b --- /dev/null +++ b/docs/superpowers/specs/2026-07-06-tui-primary-run-resilience-design.md @@ -0,0 +1,303 @@ +# TUI as Primary Interface + Run Resilience + +**Date:** 2026-07-06 +**Status:** Design — pending implementation plan +**Scope:** EvoScientist CLI/TUI launch path, run-interruption recovery, WebUI opt-in hardening + +## 1. Background & Root Cause + +A user running the WebUI backend (`ui_backend = "webui"`) saw a long research run +(127 s) render truncated output in the browser, ending mid-token at `O-8282d7e`. +Investigation established: + +- **The server-side run completed cleanly.** The thread checkpoint in + `~/.evoscientist/sessions.db` holds the full 6818-char final answer + (`status: idle`, `error: None`). The text `O-8282d7e` does not appear at the + end of any stored content — it is a mid-stream artifact. +- **The disconnect is browser-side.** `langgraph dev` sends an SSE heartbeat + every 5 s (`langgraph_api/sse.py:102`), `send_timeout` is `None` (never kills + slow sends), and the server supports resume via the `last-event-id` header + (`langgraph_api/api/runs.py:623`). The `runs/stream` endpoint is **POST**, so + the browser uses `fetch()` to read it; native `EventSource` auto-reconnect does + not apply, and the `@evoscientist/webui` front-end does not re-POST with + `last-event-id` to resume. +- **`stream_resumable=True`** lets the server-side run finish after the browser + drops, which is why "server succeeded" ≠ "browser received". +- A separate symptom from the same session — `Port 6174 cannot be bound` — was a + leftover `langgraph dev` process (PID 15288) whose `atexit` cleanup never ran + (parent killed hard). + +**Conclusion:** the SSE disconnect is structural to the WebUI path. The TUI path +does not use SSE at all — `tui_interactive.py:297` calls `create_runtime_gateways()` +with no `backend`, so it defaults to `backend="local"` (`gateway/runtime.py`), +which runs the compiled graph **in-process** and is immune. + +## 2. Goals + +1. **Make the in-process TUI the primary interface.** Long runs no longer touch + SSE, so the class of failure the user hit cannot occur. +2. **Guarantee the user always sees the complete answer.** Even if a TUI run is + interrupted (Ctrl-C, crash, terminal close), the next invocation can surface + the latest persisted assistant message and optionally resume. +3. **Keep the WebUI available as opt-in** (`--web`), with the launch path + hardened against orphaned processes and port conflicts. +4. **Eliminate the port-6174 / orphaned-`langgraph dev` failure mode** for the + default (TUI) path: it must not start `langgraph dev` at all. + +## 3. Non-Goals + +- Patching `langgraph_api` upstream. +- Replacing the WebUI front-end's fetch-SSE transport. (`@evoscientist/webui` + lives in a separate npm repo with its own release cycle; a client-side resume + fix is noted as future work, not in scope here.) +- Changing the graph, middleware, or checkpointer semantics. +- Touching the `EvoSci deploy` standalone-server command. + +## 4. Architecture + +Three launch modes, selected by a single source of truth: + +| Mode | `ui_backend` | Runs graph via | Starts `langgraph dev`? | +|---|---|---|---| +| **TUI (default)** | `tui` | `LocalGraphGateway` (in-process) | No | +| CLI | `cli` | `LocalGraphGateway` (in-process) | No | +| WebUI (opt-in) | `webui` | `langgraph dev` over HTTP/SSE | Yes | + +A new `--web` flag on the top-level `evoscientist` command forces `webui` for +that invocation regardless of config; `--tui` forces `tui`. This lets users keep +`ui_backend = "webui"` in config but drop into the TUI without editing config, +and vice-versa. + +### 4.1 Launch-mode resolution (single source of truth) + +Today the default is inconsistent: `config/settings.py:279` defaults `ui_backend` +to `"tui"`, but `cli/tui_runtime.py:12` has `DEFAULT_UI_BACKEND = "cli"`. This +design collapses to one rule, evaluated in order: + +1. `--web` flag → `webui` +2. `--tui` flag → `tui` +3. `config.ui_backend` (resolved via existing `resolve_ui_backend`) +4. `"tui"` (hardcoded final fallback) + +`resolve_ui_backend` (`cli/tui_runtime.py:41`) stays the single normalizer; the +new flags feed into it rather than bypassing it. + +### 4.2 Why TUI is immune (and what can still interrupt it) + +In-process execution via `stream_agent_events` has no network hop. The only +"loss" modes are process-level: + +- Ctrl-C mid-run +- Terminal close / SIGHUP +- Python crash / OOM + +Because the checkpointer is `AsyncSqliteSaver` writing to +`~/.evoscientist/sessions.db` (`sessions.py:108` `get_db_path`), every completed +super-step is on disk before the next begins. An interrupted run therefore +leaves a recoverable checkpoint whose `next` tuple is non-empty (pending nodes) +and whose latest assistant message — if the model node finished — is the same +content the WebUI failed to deliver. + +## 5. Components + +### C1 — Launch-mode flags + TUI default + +**Files:** +- `EvoScientist/cli/__init__.py` (Typer app entry) — add `--web` / `--tui` + options on the root command; thread the resolved `ui_backend` into the existing + dispatch. +- `EvoScientist/cli/tui_runtime.py:12` — remove the divergent + `DEFAULT_UI_BACKEND = "cli"`; route through `resolve_ui_backend` so + `settings.py`'s `"tui"` default wins. +- `EvoScientist/cli/commands.py` — wherever `ui_backend` is currently read, accept + the flag override. + +**Behavior:** Running `evoscientist` with no flags and default config launches +the TUI and starts **no** `langgraph dev` subprocess. No port is bound; the +orphan-process class of bug disappears for the default path. + +**Acceptance:** `evoscientist` (no args, default config) does not open a +listening socket on 6174 or any other port. + +### C2 — TUI run resilience (Approach B) + +Two cooperating pieces, both backed by `sessions.db`. + +#### C2a — Interrupt-fallback in the streaming loop + +Wrap the TUI streaming consumer so that on any interruption +(`KeyboardInterrupt`, `asyncio.CancelledError`, unexpected `Exception`) it reads +the thread's latest checkpoint and renders the latest assistant message before +exiting. + +**Files:** +- `EvoScientist/stream/display.py` — `_run_streaming` and + `_astream_to_console` (the two streaming entry points used by + `cli/tui_backends.py`). Add a `finally`/except guard that, after an + interruption, calls a new `render_latest_assistant(thread_id)` helper. +- `EvoScientist/sessions.py` — add `get_latest_assistant_message(thread_id) -> + str | None`: read the most recent checkpoint for the thread, walk `messages` + backwards to the last `AIMessage` with non-empty content, return its text + (joined content blocks). Reuses existing `AsyncSqliteSaver` access already in + this module. + +**Behavior on Ctrl-C mid-run:** +1. The guard catches the interrupt. +2. Prints a one-line notice: `Run interrupted — recovering latest output from + checkpoint…`. +3. Renders the latest persisted assistant message via the same Markdown renderer + the normal completion path uses. +4. If no assistant message is persisted yet (model node hadn't completed), + prints `Run interrupted before any output was saved. Resume with: evoscientist + --thread ` and exits. + +**Non-goal:** this does not *continue* the run; it only guarantees visibility of +whatever completed. Continuation is C2b. + +#### C2b — Startup detection of interrupted runs + +Runs only when the TUI is launched to **resume a specific thread** (via the +existing thread-resume flag, i.e. the command `resume_hint.py` already prints at +exit). For a brand-new session there is nothing to detect and the prompt is +skipped. When resuming, check the thread's latest checkpoint: `next` is +non-empty **and** no run is currently active for that thread. If so, prompt: + +``` +You have an unfinished run in thread : + [s] show what completed + [r] resume the run from its last checkpoint + [n] start fresh +``` + +**Files:** +- `EvoScientist/sessions.py` — add `detect_interrupted_run(thread_id) -> + InterruptedRunInfo | None` returning the checkpoint id, the pending node names + (`next`), and the timestamp. Implemented as a thin read over the existing + checkpointer's `aget_tuple`. +- `EvoScientist/cli/tui_interactive.py` — near the existing thread-load path + (around the `create_runtime_gateways()` call at line 297), invoke the detector + for the active thread and render the prompt above. + +**Resume semantics:** selecting `r` re-invokes the graph with the same +`thread_id`. LangGraph resumes from the last checkpoint automatically (this is +the same mechanism `evoscientist --thread ` already relies on for session +continuity). No new graph code needed. + +**Acceptance:** After killing a TUI mid-run with Ctrl-C, relaunching with +`--thread ` offers `[s]`/`[r]`/`[n]`; `[s]` prints the persisted final +answer; `[r]` continues execution from where it stopped. + +### C3 — WebUI mode hardening (`--web`) + +Applies only when `ui_backend == "webui"`. The goal is to make the existing +`deploy/webui.py` launch path robust and to mitigate (not eliminate) the SSE +limitation. + +**Files:** +- `EvoScientist/deploy/webui.py` — three changes: + 1. **Process-group + signal cleanup.** Today `atexit.register(stop_langgraph_dev, + ...)` only fires on clean exit. Wrap the spawned `langgraph dev` (and the + `npx` child) in a process group (`start_new_session=True` on POSIX; on + Windows use a job object via the existing `_winloop.py` helpers) and + install `SIGINT`/`SIGTERM` handlers that tear the group down. This closes + the orphan-PID-15288 path for normal terminations. `SIGKILL` cannot run + handlers on either platform — document that hard kills may still orphan + the child, and the port-conflict UX in (2) is the user-facing recovery. + 2. **Port-conflict UX.** The current "Port X is occupied by another process" + message is already correct; extend it to detect *whether the occupant is a + `langgraph dev`* (via the existing `/ok` health probe) and, if so, offer to + reuse it rather than bail. If the occupant is foreign, print the `lsof` + hint already present. + 3. **Resume window.** Set `RESUMABLE_STREAM_TTL_SECONDS=600` in the env passed + to `start_langgraph_dev` so a browser that does reconnect can replay events + for runs up to 10 minutes long (the observed run was 127 s vs. the 120 s + default). +- `EvoScientist/deploy/server.py` — `start_langgraph_dev` already accepts env; + pass `RESUMABLE_STREAM_TTL_SECONDS` through. + +**What C3 does NOT do:** it cannot make the browser auto-resume, because that +logic lives in `@evoscientist/webui`. It only widens the server-side window and +stops the orphan process. Documented in the WebUI exit message: *for long runs, +prefer the TUI (`evoscientist --tui`); the browser client does not resume a +dropped stream.* + +### C4 — Documentation + +- `README` / CLI `--help`: `--web` (browser), `--tui` (terminal, default, + recommended for long runs). +- A short "Why did my WebUI output stop?" note pointing at this design doc, so + future users reading the symptom can find the explanation. + +## 6. Data Flow + +### Default (TUI) path +``` +evoscientist ─▶ resolve_ui_backend ─▶ "tui" + │ + (no langgraph dev, no port bound) + ▼ + create_runtime_gateways(backend="local") + │ + LocalGraphGateway.stream_events + │ + stream_agent_events (in-process) ── checkpoint per super-step ──▶ sessions.db + │ + on interrupt ──▶ get_latest_assistant_message(thread_id) ──▶ render +``` + +### WebUI (`--web`) path +unchanged from today (langgraph dev + `npx @evoscientist/webui`), plus C3's +process-group cleanup and `RESUMABLE_STREAM_TTL_SECONDS=600`. + +## 7. Error Handling + +| Failure | TUI default path | WebUI `--web` path | +|---|---|---| +| Ctrl-C mid-run | C2a renders latest persisted assistant message | Browser truncates; user runs `evoscientist --tui --thread ` → `[s]` to recover | +| Crash / SIGHUP | Checkpoint on disk; next `--thread ` shows `[s]`/`[r]` prompt (C2b) | Same recovery via TUI | +| Port 6174 bound | N/A — no port bound | C3 reuses if `langgraph dev`, else prints `lsof` hint | +| Orphaned `langgraph dev` | Cannot originate from TUI path | C3 process-group + signal handlers tear it down | + +## 8. Testing + +- **Unit** (`tests/test_sessions_*.py`): `get_latest_assistant_message` returns + the full final answer for a thread whose last run completed; returns `None` + for a thread interrupted before the model node. +- **Unit**: `detect_interrupted_run` returns pending `next` nodes for an + interrupted thread; `None` for an idle one. +- **Integration** (new `tests/test_tui_interrupt_recovery.py`): start a TUI run + against a graph whose model node sleeps, send `SIGINT` mid-run, assert the + process exits 0 after printing the recovered message. +- **Integration**: relaunch with `--thread `, assert the `[s]/[r]/[n]` + prompt appears and `[r]` produces a completed run. +- **Launch** (`tests/test_cli_launch.py`, extend): `evoscientist` with default + config binds no port; `--web` binds 6174 and tears it down on `SIGTERM`. +- **WebUI** (manual / smoke): `--web`, kill parent with `SIGKILL`, assert no + orphan `langgraph dev` remains after the process-group change (best-effort — + `SIGKILL` cannot run handlers, but the process group lets the OS reap + children when the session ends; document this limit). + +## 9. Rollout / Migration + +- User's config currently has `ui_backend = "webui"`. After this change, that + config still selects WebUI (back-compatible). The user can run `evoscientist + --tui` immediately to get the resilient path, or `EvoSci config set ui_backend + tui` to make it permanent. +- No migration of `sessions.db` — schema is unchanged. +- No breaking change to `EvoSci deploy`. + +## 10. Open Questions / Future Work + +- **`@evoscientist/webui` client-side resume.** The front-end could re-POST + `/runs/stream` with `last-event-id` (or use the thread's SSE endpoint with + `since`) after a disconnect. This is the only way to fully fix the WebUI + symptom; it lives in the npm repo and is out of scope here. File a + cross-repo issue referencing this design. +- Whether `[r]` resume should replay the *interrupted* model call or only + continue from the *next* node. LangGraph's default (resume from pending + writes) is correct for most cases; revisit if users hit re-execution of + expensive tool calls. +- `cli` backend vs `tui` backend distinction — both are in-process; consider + deprecating the `cli`/`tui` split in a follow-up once the resilience work + lands, to reduce the three-way default confusion that caused the original + `DEFAULT_UI_BACKEND` divergence. diff --git a/docs/superpowers/specs/2026-07-06-webui-sse-checkpoint-fallback-design.md b/docs/superpowers/specs/2026-07-06-webui-sse-checkpoint-fallback-design.md new file mode 100644 index 0000000..81f287e --- /dev/null +++ b/docs/superpowers/specs/2026-07-06-webui-sse-checkpoint-fallback-design.md @@ -0,0 +1,196 @@ +# WebUI SSE Truncation Recovery via Checkpoint Fallback (α) + +**Date:** 2026-07-06 +**Status:** Design — pending implementation plan +**Scope:** `@evoscientist/webui` front-end (separate npm repo) + one optional convenience route in this Python repo + +## 1. Background & Root Cause + +A WebUI user saw a 127 s research run render truncated output, ending mid-token at +`O-8282d7e`. Investigation established: + +- The server-side run **completed cleanly**; the thread checkpoint in + `~/.evoscientist/sessions.db` holds the full 6818-char final answer + (`status: idle`, `error: None`). +- The disconnect is **browser-side**: `langgraph dev` sends an SSE heartbeat every + 5 s, `send_timeout` is `None`, and the server supports resume via + `last-event-id` (`langgraph_api/api/runs.py:623`). But `/runs/stream` is + **POST**, so the browser reads it via `fetch()`; native `EventSource` + auto-reconnect does not apply, and `@evoscientist/webui` does not re-POST to + resume. When the connection drops mid-run, the front-end is left showing the + last partial token and never fetches the completed answer. + +**Core insight:** the server already has the complete answer in the checkpoint. +The fix is to make the front-end fall back to that checkpoint when the stream +ends abnormally. This is small, surgical, and directly addresses the symptom. + +## 2. Goal + +**The user always sees the complete final answer in the WebUI, even if the SSE +stream drops mid-run.** + +## 3. Non-Goals + +- Seamless stream resume (tracking `last-event-id` / `seq` and replaying + buffered events). That is option β — better UX, ~3–5× the code, deferred. +- The TUI pivot / launch-mode decoupling (option γ). Separable; only relevant if + the port-conflict / orphan-process issues need addressing. +- Patching `langgraph_api` upstream. + +## 4. Detection Logic — What Counts as "Truncated" + +The langgraph SSE protocol emits terminal events when a run finishes +(success → `end`, failure → `error`; exact names to be confirmed against the +`@langchain/langgraph-sdk` version during implementation). The front-end's +streaming reader loop currently processes events until the `fetch()` ReadableStream +closes. + +**Truncation** = the stream closed (reader returned) **without** a terminal event +having been received. Causes: browser tab throttled, network blip, proxy idle +timeout. On truncation, the in-progress assistant bubble is left showing a prefix +of the real answer. + +## 5. Components + +### F1 — Truncation detector (front-end) + +In the streaming reader loop, track a `terminalSeen` flag. Set it when a +terminal event arrives. When the reader returns, if `!terminalSeen`, mark the +run `truncated` and trigger F2. + +**File:** `@evoscientist/webui` — the run-stream consumer (the module that wraps +the langgraph-ts SDK's `runs.stream` and renders into the message list). + +### F2 — Checkpoint fallback fetch (front-end) + +On `truncated`, fetch the complete final assistant message and **replace** the +in-progress bubble's content with it (not append — the partial tokens are a +prefix of the same message). + +Two implementation choices, pick one: + +- **F2a (zero server change):** call the standard langgraph endpoint + `GET /threads/{thread_id}/state`, walk `values.messages` backwards to the last + `AIMessage`, join its content blocks. No new route, but the front-end parses + raw state and reimplements "find latest assistant text" logic. +- **F2b (recommended, uses S1 below):** call `GET /api/threads/{id}/final-answer` + → `{content, completed_at, complete}`. Server owns the parsing; front-end + stays dumb. + +Render a subtle affordance (e.g., a dim "⚠ stream dropped — recovered from +checkpoint" line above the message) so the user knows a disconnect happened and +the displayed text is authoritative, not stale streaming. + +### F3 — Retry until the run settles (front-end) + +The browser may drop while the run is **still executing** server-side. A single +fallback fetch at drop-time could return a not-yet-final message. So F2 retries +with backoff until either: + +- `complete == true` (run finished — render and stop), or +- a max wait elapses (default 240 s — comfortably exceeds the observed 127 s + run even if the drop happens at t=0), then render whatever is latest and + stop retrying. + +Backoff: poll at 1 s, 2 s, 4 s, then every 5 s up to the max. Each poll replaces +the bubble with the latest content, so if the run finishes mid-retry the user +sees it update to the final form. + +### S1 — Convenience route `GET /api/threads/{id}/final-answer` (this repo, optional but recommended) + +Enables F2b. Mounted on the existing custom ASGI app already wired via +`langgraph.json` → `EvoScientist.langgraph_dev.http:app` (currently only serves +`/api/models`). + +**Contract:** + +``` +GET /api/threads/{thread_id}/final-answer +200 → { "content": str, # last AIMessage text, blocks joined + "completed_at": iso8601, # run end timestamp, or null + "complete": bool } # true iff run reached terminal state +404 → thread not found +``` + +**`complete` derivation** (run reached a terminal state): +`thread.status == "idle"` **or** latest checkpoint's `next` tuple is empty. + +**File:** `EvoScientist/langgraph_dev/http.py` — add `get_final_answer` route +beside the existing `get_models`. Implementation reads the thread state via the +shared checkpointer (`AsyncSqliteSaver` at `~/.evoscientist/sessions.db`, +already used by `sessions.py`). If direct checkpointer access is awkward from +inside the mounted app, fall back to an internal `langgraph_sdk.get_client()` +call to `GET /threads/{id}/state` on localhost — the contract stays identical. + +**Why prefer F2b+S1 over F2a:** the "find latest AIMessage + join content blocks ++ decide complete?" logic is non-trivial (content is a list of typed blocks, +including `reasoning` blocks that should be excluded from the rendered answer — +see the checkpoint: block[0] was `reasoning`, block[1] was `text`). Centralizing +it server-side keeps the front-end thin and makes the same logic reusable by the +TUI's future recovery path if γ is ever pursued. + +## 6. Data Flow + +``` +browser fetch(/runs/stream, POST) ──SSE──▶ langgraph dev + │ │ + │ (mid-run, connection drops) │ run continues to completion + ▼ │ + reader returns, no terminal event ──▶ truncated │ + │ │ + │ GET /api/threads/{id}/final-answer │ + ▼ ▼ + http.py::get_final_answer ──read checkpoint──▶ sessions.db + │ + ▼ + {content, complete} ──▶ if !complete, retry (F3); else render (F2) +``` + +## 7. Error Handling + +| Case | Behavior | +|---|---| +| Stream ends normally (`end`/`error` received) | F1 sets `terminalSeen`; no fallback; existing render path unchanged | +| Stream drops, run already finished server-side | First fallback fetch returns `complete=true`; render immediately | +| Stream drops, run still executing | F3 retries with backoff; each retry renders latest; stops when `complete` or 120 s max | +| Thread unknown / deleted | 404; front-end shows "stream interrupted and the session could not be recovered" | +| Convenience route unreachable (older server without S1) | Front-end falls back to F2a (raw state endpoint) — degrade gracefully | + +## 8. Testing + +**Front-end (`@evoscientist/webui`):** +- Unit: feed the reader a synthetic event stream that closes without a terminal + event → assert `truncated == true`; feed one with `end` → `false`. +- Integration: mock `fetch` to drop mid-stream; assert the fallback fetch fires + and the bubble's content is replaced with the checkpoint value. + +**Server (this repo, if S1):** new `tests/test_http_final_answer.py` +- Completed thread → 200, `complete=true`, content matches the last AIMessage + text (not the reasoning block). +- Mid-run thread (forced `busy` / non-empty `next`) → 200, `complete=false`. +- Unknown thread → 404. +- Content with mixed `reasoning` + `text` blocks → only `text` in `content`. + +## 9. Cross-Repo Coordination & Rollout + +- **Order:** land S1 in this Python repo first (route + tests, behind no flag — + it's a pure addition). Then F1/F2/F3 in `@evoscientist/webui`. +- **Versioning:** the WebUI launches via `npx @evoscientist/webui@latest`, so + the front-end fix reaches users on their next launch automatically once + published. No coordinated upgrade required on the Python side. +- **Backward compatibility:** if a user runs a new front-end against an older + Python server without S1, the front-end must detect the 404 and fall back to + F2a (raw state endpoint). If an old front-end runs against a new server, S1 + is simply unused — no harm. + +## 10. Open Questions / Future Work + +- **Exact terminal-event names** for the SDK version in use (`end`/`error` vs. + `done`/`result`). Confirm against `@langchain/langgraph-sdk` during + implementation; the detector is parameterized on this. +- **True resume (β):** track `last-event-id` (or the V2 thread-stream `seq`) and + replay buffered events on reconnect — seamless, no visible "recovered" flash. + Larger front-end effort; defer until α is validated. +- **TUI pivot (γ):** separable work to make the in-process TUI the default and + eliminate the port-conflict / orphan-process issues. Independent spec if + pursued. diff --git a/pyproject.toml b/pyproject.toml index 808222d..e73a31b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "EvoScientist" -version = "0.2.1" +version = "0.2.2" description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery" readme = "README.md" requires-python = ">=3.11" diff --git a/tests/test_langgraph_dev_http.py b/tests/test_langgraph_dev_http.py index 88cd3ca..a33d280 100644 --- a/tests/test_langgraph_dev_http.py +++ b/tests/test_langgraph_dev_http.py @@ -7,6 +7,7 @@ from __future__ import annotations from unittest.mock import patch +from langchain_core.messages import AIMessage, HumanMessage from starlette.testclient import TestClient from EvoScientist.config import EvoScientistConfig @@ -161,3 +162,143 @@ def test_ollama_discovery_skipped_when_base_url_absent(): {"name": n, "model_id": m, "provider": p} for n, m, p in list_models_by_provider() ] + + +def test_final_answer_extracts_latest_ai_text_blocks(): + async def fake_metadata(_thread_id): + return {"updated_at": "2026-07-06T14:14:53+00:00"} + + async def fake_messages(_thread_id): + return [ + HumanMessage(content="question"), + AIMessage(content="old answer"), + AIMessage( + content=[ + {"type": "reasoning", "text": "internal"}, + {"type": "text", "text": "Part A"}, + {"type": "tool_use", "name": "search"}, + {"type": "output_text", "text": "Part B"}, + ] + ), + ] + + async def fake_runtime(_request, _thread_id): + return { + "found": True, + "complete": True, + "completed_at": "2026-07-06T14:15:00+00:00", + } + + with ( + patch( + "EvoScientist.langgraph_dev.http._get_thread_metadata_for_http", + new=fake_metadata, + ), + patch( + "EvoScientist.langgraph_dev.http._get_thread_messages_for_http", + new=fake_messages, + ), + patch( + "EvoScientist.langgraph_dev.http._read_thread_runtime_state", + new=fake_runtime, + ), + ): + resp = client.get("/api/threads/thread-1/final-answer") + + assert resp.status_code == 200 + assert resp.json() == { + "content": "Part A\n\nPart B", + "completed_at": "2026-07-06T14:15:00+00:00", + "complete": True, + } + + +def test_final_answer_skips_tool_selection_json_text(): + async def fake_metadata(_thread_id): + return {"updated_at": "2026-07-06T14:14:53+00:00"} + + async def fake_messages(_thread_id): + return [ + HumanMessage(content="question"), + AIMessage(content="stable answer"), + AIMessage( + content=( + '{"tools":["search_papers","get_abstract"]}' + '{"tools":["web_search_exa"]}' + ) + ), + ] + + async def fake_runtime(_request, _thread_id): + return { + "found": True, + "complete": True, + "completed_at": "2026-07-06T14:15:00+00:00", + } + + with ( + patch( + "EvoScientist.langgraph_dev.http._get_thread_metadata_for_http", + new=fake_metadata, + ), + patch( + "EvoScientist.langgraph_dev.http._get_thread_messages_for_http", + new=fake_messages, + ), + patch( + "EvoScientist.langgraph_dev.http._read_thread_runtime_state", + new=fake_runtime, + ), + ): + resp = client.get("/api/threads/thread-1/final-answer") + + assert resp.status_code == 200 + assert resp.json()["content"] == "stable answer" + + +def test_final_answer_returns_404_for_unknown_thread(): + async def fake_metadata(_thread_id): + return None + + with patch( + "EvoScientist.langgraph_dev.http._get_thread_metadata_for_http", + new=fake_metadata, + ): + resp = client.get("/api/threads/missing/final-answer") + + assert resp.status_code == 404 + assert resp.json() == {"error": "thread not found"} + + +def test_final_answer_does_not_mark_complete_when_runtime_state_fails(): + async def fake_metadata(_thread_id): + return {"updated_at": "2026-07-06T14:14:53+00:00"} + + async def fake_messages(_thread_id): + return [AIMessage(content="checkpoint answer")] + + async def fake_runtime(_request, _thread_id): + raise RuntimeError("langgraph runtime unavailable") + + with ( + patch( + "EvoScientist.langgraph_dev.http._get_thread_metadata_for_http", + new=fake_metadata, + ), + patch( + "EvoScientist.langgraph_dev.http._get_thread_messages_for_http", + new=fake_messages, + ), + patch( + "EvoScientist.langgraph_dev.http._read_thread_runtime_state", + new=fake_runtime, + ), + ): + resp = client.get("/api/threads/thread-1/final-answer") + + assert resp.status_code == 200 + assert resp.json() == { + "content": "checkpoint answer", + "completed_at": None, + "complete": False, + } diff --git a/tests/test_webui_launcher.py b/tests/test_webui_launcher.py new file mode 100644 index 0000000..aa5957c --- /dev/null +++ b/tests/test_webui_launcher.py @@ -0,0 +1,15 @@ +from __future__ import annotations + +from EvoScientist.deploy.webui import _resolve_webui_package + + +def test_resolve_webui_package_defaults_to_published_package(monkeypatch): + monkeypatch.delenv("EVOSCIENTIST_WEBUI_PACKAGE", raising=False) + + assert _resolve_webui_package() == "@evoscientist/webui@latest" + + +def test_resolve_webui_package_allows_local_override(monkeypatch): + monkeypatch.setenv("EVOSCIENTIST_WEBUI_PACKAGE", "/tmp/EvoScientist-WebUI") + + assert _resolve_webui_package() == "/tmp/EvoScientist-WebUI" diff --git a/uv.lock b/uv.lock index 7616673..5ad4352 100644 --- a/uv.lock +++ b/uv.lock @@ -944,7 +944,7 @@ wheels = [ [[package]] name = "evoscientist" -version = "0.2.0" +version = "0.2.2" source = { editable = "." } dependencies = [ { name = "deepagents", extra = ["quickjs"] },