Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e0acc6155e |
@@ -48,3 +48,4 @@ conversation_history/
|
||||
*meals/
|
||||
botpy.log
|
||||
large_tool_results/
|
||||
runs/
|
||||
|
||||
@@ -41,9 +41,15 @@ from ..stream.console import console
|
||||
|
||||
# Front-end npm package + spec. ``@latest`` → always the newest published UI.
|
||||
_WEBUI_PACKAGE = "@evoscientist/webui@latest"
|
||||
_WEBUI_PACKAGE_ENV = "EVOSCIENTIST_WEBUI_PACKAGE"
|
||||
_DEFAULT_WEBUI_PORT = 4716
|
||||
|
||||
|
||||
def _resolve_webui_package() -> str:
|
||||
"""Return the npm package spec to launch for the WebUI front-end."""
|
||||
return os.getenv(_WEBUI_PACKAGE_ENV) or _WEBUI_PACKAGE
|
||||
|
||||
|
||||
def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
"""Start the deploy-style backend + the WebUI front-end, then block.
|
||||
|
||||
@@ -203,6 +209,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
"PORT": str(webui_port),
|
||||
}
|
||||
)
|
||||
webui_package = _resolve_webui_package()
|
||||
console.print(
|
||||
Panel(
|
||||
Text.from_markup(
|
||||
@@ -211,7 +218,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
f"[bold]WebUI:[/bold] http://localhost:{webui_port} "
|
||||
f"[dim](opens in your browser)[/dim]\n"
|
||||
f"[bold]Logs:[/bold] {_shorten(str(RUNTIME.log_file))}\n\n"
|
||||
f"[dim]Fetching {_WEBUI_PACKAGE} via npx (first run may take a "
|
||||
f"[dim]Fetching {webui_package} via npx (first run may take a "
|
||||
f"moment)… Press Ctrl+C to stop.[/dim]"
|
||||
),
|
||||
title="[bold green]✓ EvoScientist WebUI[/bold green]",
|
||||
@@ -228,7 +235,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||
popen_kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP
|
||||
try:
|
||||
webui_proc = subprocess.Popen(
|
||||
[npx, "--yes", _WEBUI_PACKAGE, "--port", str(webui_port)],
|
||||
[npx, "--yes", webui_package, "--port", str(webui_port)],
|
||||
**popen_kwargs,
|
||||
)
|
||||
except Exception as exc:
|
||||
|
||||
@@ -23,7 +23,17 @@ memory.
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sqlite3
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import aiosqlite
|
||||
import httpx
|
||||
from langgraph.checkpoint.serde.jsonplus import JsonPlusSerializer
|
||||
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
|
||||
from starlette.applications import Starlette
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import JSONResponse
|
||||
@@ -31,6 +41,15 @@ from starlette.routing import Route
|
||||
|
||||
from EvoScientist.config import get_effective_config
|
||||
from EvoScientist.llm.models import list_model_picker_entries
|
||||
from EvoScientist.sessions import (
|
||||
MAIN_THREAD_FILTER_PARAMS,
|
||||
MAIN_THREAD_FILTER_SQL,
|
||||
_load_checkpoint_messages,
|
||||
_table_exists,
|
||||
_to_short_path,
|
||||
)
|
||||
|
||||
_logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
async def get_models(_request: Request) -> JSONResponse:
|
||||
@@ -75,8 +94,270 @@ async def get_models(_request: Request) -> JSONResponse:
|
||||
)
|
||||
|
||||
|
||||
def _message_type(message: Any) -> str | None:
|
||||
if isinstance(message, dict):
|
||||
role = message.get("role")
|
||||
if role == "assistant":
|
||||
return "ai"
|
||||
if role == "user":
|
||||
return "human"
|
||||
value = message.get("type")
|
||||
return str(value) if value is not None else None
|
||||
value = getattr(message, "type", None)
|
||||
return str(value) if value is not None else None
|
||||
|
||||
|
||||
def _message_content(message: Any) -> Any:
|
||||
if isinstance(message, dict):
|
||||
return message.get("content")
|
||||
return getattr(message, "content", None)
|
||||
|
||||
|
||||
def _extract_text_content(content: Any) -> str:
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
if not isinstance(content, list):
|
||||
return ""
|
||||
|
||||
parts: list[str] = []
|
||||
for block in content:
|
||||
if isinstance(block, str):
|
||||
parts.append(block)
|
||||
continue
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
block_type = block.get("type")
|
||||
if block_type not in {"text", "output_text"}:
|
||||
continue
|
||||
text = block.get("text") or block.get("content")
|
||||
if isinstance(text, str):
|
||||
parts.append(text)
|
||||
return "\n\n".join(part for part in parts if part)
|
||||
|
||||
|
||||
def _is_tool_selection_payload(raw: str) -> bool:
|
||||
try:
|
||||
parsed = json.loads(raw)
|
||||
except json.JSONDecodeError:
|
||||
return False
|
||||
return (
|
||||
isinstance(parsed, dict)
|
||||
and set(parsed) == {"tools"}
|
||||
and isinstance(parsed["tools"], list)
|
||||
and all(isinstance(tool, str) for tool in parsed["tools"])
|
||||
)
|
||||
|
||||
|
||||
def _split_json_objects(raw: str) -> list[str] | None:
|
||||
objects: list[str] = []
|
||||
depth = 0
|
||||
start = -1
|
||||
in_string = False
|
||||
escaping = False
|
||||
|
||||
for i, char in enumerate(raw):
|
||||
if in_string:
|
||||
if escaping:
|
||||
escaping = False
|
||||
elif char == "\\":
|
||||
escaping = True
|
||||
elif char == '"':
|
||||
in_string = False
|
||||
continue
|
||||
|
||||
if char == '"':
|
||||
if depth == 0:
|
||||
return None
|
||||
in_string = True
|
||||
continue
|
||||
if char == "{":
|
||||
if depth == 0:
|
||||
start = i
|
||||
depth += 1
|
||||
continue
|
||||
if char == "}":
|
||||
depth -= 1
|
||||
if depth < 0 or start < 0:
|
||||
return None
|
||||
if depth == 0:
|
||||
objects.append(raw[start : i + 1])
|
||||
start = -1
|
||||
continue
|
||||
if depth == 0 and not char.isspace():
|
||||
return None
|
||||
|
||||
if depth != 0 or in_string or not objects:
|
||||
return None
|
||||
return objects
|
||||
|
||||
|
||||
def _is_tool_selection_text(text: str) -> bool:
|
||||
stripped = text.strip()
|
||||
if not stripped or '"tools"' not in stripped:
|
||||
return False
|
||||
if _is_tool_selection_payload(stripped):
|
||||
return True
|
||||
objects = _split_json_objects(stripped)
|
||||
return objects is not None and all(_is_tool_selection_payload(obj) for obj in objects)
|
||||
|
||||
|
||||
def _extract_final_answer(messages: list[Any]) -> str:
|
||||
"""Return displayable text from the latest AI message in *messages*."""
|
||||
for message in reversed(messages):
|
||||
if _message_type(message) != "ai":
|
||||
continue
|
||||
content = _extract_text_content(_message_content(message)).strip()
|
||||
if content and not _is_tool_selection_text(content):
|
||||
return content
|
||||
return ""
|
||||
|
||||
|
||||
def _sessions_db_path_for_http() -> Path:
|
||||
data_dir = os.getenv("EVOSCIENTIST_DATA_DIR")
|
||||
base = Path(data_dir).expanduser() if data_dir else Path.home() / ".evoscientist"
|
||||
return Path(_to_short_path(str(base))) / "sessions.db"
|
||||
|
||||
|
||||
async def _get_thread_metadata_for_http(thread_id: str) -> dict | None:
|
||||
try:
|
||||
async with aiosqlite.connect(
|
||||
str(_sessions_db_path_for_http()), timeout=30.0
|
||||
) as conn:
|
||||
if not await _table_exists(conn, "checkpoints"):
|
||||
return None
|
||||
query = f"""
|
||||
SELECT json_extract(metadata, '$.workspace_dir') as workspace_dir,
|
||||
json_extract(metadata, '$.model') as model,
|
||||
json_extract(metadata, '$.updated_at') as updated_at
|
||||
FROM checkpoints
|
||||
WHERE thread_id = ?
|
||||
AND {MAIN_THREAD_FILTER_SQL}
|
||||
ORDER BY checkpoint_id DESC
|
||||
LIMIT 1
|
||||
"""
|
||||
async with conn.execute(
|
||||
query, (thread_id, *MAIN_THREAD_FILTER_PARAMS)
|
||||
) as cur:
|
||||
row = await cur.fetchone()
|
||||
except (OSError, sqlite3.Error):
|
||||
return None
|
||||
if not row:
|
||||
return None
|
||||
return {
|
||||
"workspace_dir": row[0],
|
||||
"model": row[1],
|
||||
"updated_at": row[2],
|
||||
}
|
||||
|
||||
|
||||
async def _get_thread_messages_for_http(thread_id: str) -> list:
|
||||
try:
|
||||
async with aiosqlite.connect(
|
||||
str(_sessions_db_path_for_http()), timeout=30.0
|
||||
) as conn:
|
||||
if not await _table_exists(conn, "checkpoints"):
|
||||
return []
|
||||
check = f"""
|
||||
SELECT 1 FROM checkpoints
|
||||
WHERE thread_id = ? AND {MAIN_THREAD_FILTER_SQL}
|
||||
LIMIT 1
|
||||
"""
|
||||
async with conn.execute(
|
||||
check, (thread_id, *MAIN_THREAD_FILTER_PARAMS)
|
||||
) as cur:
|
||||
if not await cur.fetchone():
|
||||
return []
|
||||
serde = JsonPlusSerializer()
|
||||
saver = AsyncSqliteSaver(conn, serde=serde)
|
||||
return await _load_checkpoint_messages(saver, thread_id)
|
||||
except (OSError, sqlite3.Error):
|
||||
return []
|
||||
|
||||
|
||||
async def _read_thread_runtime_state(request: Request, thread_id: str) -> dict[str, Any]:
|
||||
"""Read thread status from the co-hosted langgraph-api endpoints."""
|
||||
base_url = f"{request.url.scheme}://{request.url.netloc}"
|
||||
timeout = httpx.Timeout(2.0, connect=0.5)
|
||||
async with httpx.AsyncClient(base_url=base_url, timeout=timeout) as client:
|
||||
thread_resp = await client.get(f"/threads/{thread_id}")
|
||||
if thread_resp.status_code == 404:
|
||||
return {"found": False}
|
||||
thread_resp.raise_for_status()
|
||||
thread = thread_resp.json()
|
||||
|
||||
state: dict[str, Any] = {}
|
||||
state_resp = await client.get(f"/threads/{thread_id}/state")
|
||||
if state_resp.status_code == 404:
|
||||
return {"found": False}
|
||||
if state_resp.status_code < 400:
|
||||
state = state_resp.json()
|
||||
|
||||
next_nodes = state.get("next")
|
||||
is_terminal_checkpoint = isinstance(next_nodes, (list, tuple)) and not next_nodes
|
||||
status = thread.get("status")
|
||||
complete = status == "idle" or is_terminal_checkpoint
|
||||
completed_at = None
|
||||
if complete:
|
||||
completed_at = (
|
||||
thread.get("state_updated_at")
|
||||
or thread.get("updated_at")
|
||||
or (state.get("metadata") or {}).get("updated_at")
|
||||
)
|
||||
return {
|
||||
"found": True,
|
||||
"complete": complete,
|
||||
"completed_at": completed_at,
|
||||
}
|
||||
|
||||
|
||||
async def get_final_answer(request: Request) -> JSONResponse:
|
||||
"""Return the latest checkpointed assistant answer for a WebUI thread.
|
||||
|
||||
This is a recovery surface for the WebUI stream consumer: when browser-side
|
||||
SSE is interrupted but the langgraph run continues server-side, the final
|
||||
answer is already persisted in ``sessions.db``. The route centralizes the
|
||||
non-trivial "latest AIMessage text only" extraction so the browser does not
|
||||
render reasoning blocks, tool calls, or tool-selection JSON fragments.
|
||||
"""
|
||||
thread_id = request.path_params["thread_id"]
|
||||
metadata = await _get_thread_metadata_for_http(thread_id)
|
||||
if metadata is None:
|
||||
return JSONResponse({"error": "thread not found"}, status_code=404)
|
||||
|
||||
messages = await _get_thread_messages_for_http(thread_id)
|
||||
content = _extract_final_answer(messages)
|
||||
|
||||
runtime: dict[str, Any] = {"found": True, "complete": False, "completed_at": None}
|
||||
try:
|
||||
runtime = await _read_thread_runtime_state(request, thread_id)
|
||||
except Exception as exc:
|
||||
_logger.debug(
|
||||
"Could not read langgraph runtime state for thread %s: %s",
|
||||
thread_id,
|
||||
exc,
|
||||
)
|
||||
if runtime.get("found") is False:
|
||||
return JSONResponse({"error": "thread not found"}, status_code=404)
|
||||
|
||||
completed_at = runtime.get("completed_at")
|
||||
if completed_at is None and runtime.get("complete"):
|
||||
completed_at = metadata.get("updated_at")
|
||||
return JSONResponse(
|
||||
{
|
||||
"content": content,
|
||||
"completed_at": completed_at,
|
||||
"complete": bool(runtime.get("complete")),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
app = Starlette(
|
||||
routes=[
|
||||
Route("/api/models", get_models, methods=["GET"]),
|
||||
Route(
|
||||
"/api/threads/{thread_id}/final-answer",
|
||||
get_final_answer,
|
||||
methods=["GET"],
|
||||
),
|
||||
]
|
||||
)
|
||||
|
||||
@@ -474,6 +474,15 @@ def get_chat_model(
|
||||
# Even native thinking models like kimi-k2-thinking operate in non-thinking mode.
|
||||
if provider == "moonshot":
|
||||
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
|
||||
# custom-openai: some Codex-orientated reseller gateways (e.g.
|
||||
# hunnuapi.top) blacklist the openai SDK's default
|
||||
# "OpenAI/Python" User-Agent at the app layer and return
|
||||
# 403 "Your request was blocked." to any langchain-openai client.
|
||||
# Spoof a Codex-CLI UA so requests go through.
|
||||
if provider == "custom-openai":
|
||||
kwargs.setdefault("default_headers", {}).setdefault(
|
||||
"User-Agent", "codex_cli_rs/0.0.0"
|
||||
)
|
||||
provider = "openai"
|
||||
|
||||
# OpenRouter → native ChatOpenRouter via init_chat_model.
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
<a href="https://pypi.org/project/EvoScientist/"><picture>
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg">
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-dark.svg">
|
||||
<img alt="PyPI v0.2.1" src="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg" height="28">
|
||||
<img alt="PyPI v0.2.2" src="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg" height="28">
|
||||
</picture></a><a href="https://EvoScientist.github.io/"><picture>
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-light.svg">
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-dark.svg">
|
||||
@@ -137,6 +137,34 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
|
||||
> [!TIP]
|
||||
> Looking for ready-to-use research skills? Check out [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) — powered by [**EvoScientist**](https://github.com/EvoScientist/EvoScientist)'s engine and installable skills, the entire end-to-end research lifecycle is covered out of the box. [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) are also compatible with other CLI coding agents.
|
||||
|
||||
## WebUI Streaming Fix Notes
|
||||
|
||||
The WebUI streaming fix is based on one rule: a browser-side SSE disconnect is
|
||||
only a transport event, not proof that the LangGraph run has finished.
|
||||
|
||||
- Keep the composer action button as **Stop** while the backend thread still has
|
||||
pending work. The UI should not switch back to **Send** merely because the
|
||||
live SSE stream ended.
|
||||
- Derive the visible busy state from both the live stream and backend thread
|
||||
state. In the patched WebUI this is represented as `isRunActive`, backed by
|
||||
`threads.getState(threadId)`.
|
||||
- Restore **Send** only after the backend reports a terminal state: idle,
|
||||
completed final answer, failed, or cancelled.
|
||||
- Preserve human-in-the-loop controls during tool approvals. Approval buttons
|
||||
remain actionable, while the bottom composer still shows **Stop** to avoid
|
||||
duplicate submissions.
|
||||
- Treat leaked `{"tools":[...]}` payloads as internal tool-selection control
|
||||
messages, not assistant text, and filter them from the transcript.
|
||||
- Recover dropped long-response tails through the checkpoint-backed
|
||||
`/api/threads/{thread_id}/final-answer` endpoint instead of relying only on
|
||||
browser stream state.
|
||||
- Use a long enough final-answer polling window for real research runs; short
|
||||
fallback windows can expire before the backend completes and make the browser
|
||||
appear truncated until refresh.
|
||||
- Validate with real browser flows: normal streaming, simulated SSE disconnect
|
||||
and reload, pending tool approval, rejection/cancellation, and final transition
|
||||
back to **Send**.
|
||||
|
||||
## 🔥 News
|
||||
- **[03 Jun 2026]** 🥈 Ranked #2 overall — and 🥇 #1 among `GPT-5.4`-based agents — on [ResearchClawBench](https://github.com/InternScience/ResearchClawBench) (Agent Mode)! [**Leaderboard**](https://internscience.github.io/ResearchClawBench-Home/) 👈
|
||||
- **[18 Apr 2026]** 🥇 Ranked #1 on [DeepResearch Bench](https://deepresearch-bench.github.io/) at submission time! [**Leaderboard**](https://huggingface.co/spaces/muset-ai/DeepResearch-Bench-Leaderboard) 👈
|
||||
@@ -151,6 +179,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
|
||||
<details>
|
||||
<summary>📦 Release Highlights — version changelog</summary>
|
||||
|
||||
- **[07 Jul 2026]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — WebUI streaming resilience hotfix: SSE disconnects no longer imply run completion; the composer stays on **Stop** until backend thread state is terminal; final-answer checkpoint recovery fills dropped response tails; tool-selection JSON payloads are filtered from live transcripts; `EVOSCIENTIST_WEBUI_PACKAGE` lets the launcher run a local patched WebUI package for validation before npm publication.
|
||||
- **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5.
|
||||
- **[26 Jun 2026]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — Scheduled tasks: cron-style recurring runs via `/schedule` or natural language, run unattended with shell-access gating; self-linking memory that connects observations into a knowledge graph (complements / contradicts / supersedes); a read-only `GET /api/models` endpoint for the WebUI model picker.
|
||||
- **[23 Jun 2026]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — Hotfix for fresh installs: the first message crashed with `The subagent `task` tool cannot be exposed via `ptc`` after deepagents 0.6.11 / langchain-quickjs 0.3 reserved `task` as the REPL global. Removed `task` from the code-interpreter PTC allowlist (`task()` stays available as the REPL global; async dispatch stays in PTC) and pinned `deepagents[quickjs]~=0.6.11`.
|
||||
@@ -180,6 +209,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
|
||||
- [📦 Installation](#-installation)
|
||||
- [🔑 Configuration](#-configuration)
|
||||
- [⚡ Quick Start](#-quick-start)
|
||||
- [WebUI Streaming Fix Notes](#webui-streaming-fix-notes)
|
||||
- [⏰ Scheduled Tasks](#-scheduled-tasks)
|
||||
- [🍪 Examples & Recipes](#-examples--recipes)
|
||||
- [🔌 MCP Integration](#-mcp-integration)
|
||||
|
||||
@@ -0,0 +1,303 @@
|
||||
# TUI as Primary Interface + Run Resilience
|
||||
|
||||
**Date:** 2026-07-06
|
||||
**Status:** Design — pending implementation plan
|
||||
**Scope:** EvoScientist CLI/TUI launch path, run-interruption recovery, WebUI opt-in hardening
|
||||
|
||||
## 1. Background & Root Cause
|
||||
|
||||
A user running the WebUI backend (`ui_backend = "webui"`) saw a long research run
|
||||
(127 s) render truncated output in the browser, ending mid-token at `O-8282d7e`.
|
||||
Investigation established:
|
||||
|
||||
- **The server-side run completed cleanly.** The thread checkpoint in
|
||||
`~/.evoscientist/sessions.db` holds the full 6818-char final answer
|
||||
(`status: idle`, `error: None`). The text `O-8282d7e` does not appear at the
|
||||
end of any stored content — it is a mid-stream artifact.
|
||||
- **The disconnect is browser-side.** `langgraph dev` sends an SSE heartbeat
|
||||
every 5 s (`langgraph_api/sse.py:102`), `send_timeout` is `None` (never kills
|
||||
slow sends), and the server supports resume via the `last-event-id` header
|
||||
(`langgraph_api/api/runs.py:623`). The `runs/stream` endpoint is **POST**, so
|
||||
the browser uses `fetch()` to read it; native `EventSource` auto-reconnect does
|
||||
not apply, and the `@evoscientist/webui` front-end does not re-POST with
|
||||
`last-event-id` to resume.
|
||||
- **`stream_resumable=True`** lets the server-side run finish after the browser
|
||||
drops, which is why "server succeeded" ≠ "browser received".
|
||||
- A separate symptom from the same session — `Port 6174 cannot be bound` — was a
|
||||
leftover `langgraph dev` process (PID 15288) whose `atexit` cleanup never ran
|
||||
(parent killed hard).
|
||||
|
||||
**Conclusion:** the SSE disconnect is structural to the WebUI path. The TUI path
|
||||
does not use SSE at all — `tui_interactive.py:297` calls `create_runtime_gateways()`
|
||||
with no `backend`, so it defaults to `backend="local"` (`gateway/runtime.py`),
|
||||
which runs the compiled graph **in-process** and is immune.
|
||||
|
||||
## 2. Goals
|
||||
|
||||
1. **Make the in-process TUI the primary interface.** Long runs no longer touch
|
||||
SSE, so the class of failure the user hit cannot occur.
|
||||
2. **Guarantee the user always sees the complete answer.** Even if a TUI run is
|
||||
interrupted (Ctrl-C, crash, terminal close), the next invocation can surface
|
||||
the latest persisted assistant message and optionally resume.
|
||||
3. **Keep the WebUI available as opt-in** (`--web`), with the launch path
|
||||
hardened against orphaned processes and port conflicts.
|
||||
4. **Eliminate the port-6174 / orphaned-`langgraph dev` failure mode** for the
|
||||
default (TUI) path: it must not start `langgraph dev` at all.
|
||||
|
||||
## 3. Non-Goals
|
||||
|
||||
- Patching `langgraph_api` upstream.
|
||||
- Replacing the WebUI front-end's fetch-SSE transport. (`@evoscientist/webui`
|
||||
lives in a separate npm repo with its own release cycle; a client-side resume
|
||||
fix is noted as future work, not in scope here.)
|
||||
- Changing the graph, middleware, or checkpointer semantics.
|
||||
- Touching the `EvoSci deploy` standalone-server command.
|
||||
|
||||
## 4. Architecture
|
||||
|
||||
Three launch modes, selected by a single source of truth:
|
||||
|
||||
| Mode | `ui_backend` | Runs graph via | Starts `langgraph dev`? |
|
||||
|---|---|---|---|
|
||||
| **TUI (default)** | `tui` | `LocalGraphGateway` (in-process) | No |
|
||||
| CLI | `cli` | `LocalGraphGateway` (in-process) | No |
|
||||
| WebUI (opt-in) | `webui` | `langgraph dev` over HTTP/SSE | Yes |
|
||||
|
||||
A new `--web` flag on the top-level `evoscientist` command forces `webui` for
|
||||
that invocation regardless of config; `--tui` forces `tui`. This lets users keep
|
||||
`ui_backend = "webui"` in config but drop into the TUI without editing config,
|
||||
and vice-versa.
|
||||
|
||||
### 4.1 Launch-mode resolution (single source of truth)
|
||||
|
||||
Today the default is inconsistent: `config/settings.py:279` defaults `ui_backend`
|
||||
to `"tui"`, but `cli/tui_runtime.py:12` has `DEFAULT_UI_BACKEND = "cli"`. This
|
||||
design collapses to one rule, evaluated in order:
|
||||
|
||||
1. `--web` flag → `webui`
|
||||
2. `--tui` flag → `tui`
|
||||
3. `config.ui_backend` (resolved via existing `resolve_ui_backend`)
|
||||
4. `"tui"` (hardcoded final fallback)
|
||||
|
||||
`resolve_ui_backend` (`cli/tui_runtime.py:41`) stays the single normalizer; the
|
||||
new flags feed into it rather than bypassing it.
|
||||
|
||||
### 4.2 Why TUI is immune (and what can still interrupt it)
|
||||
|
||||
In-process execution via `stream_agent_events` has no network hop. The only
|
||||
"loss" modes are process-level:
|
||||
|
||||
- Ctrl-C mid-run
|
||||
- Terminal close / SIGHUP
|
||||
- Python crash / OOM
|
||||
|
||||
Because the checkpointer is `AsyncSqliteSaver` writing to
|
||||
`~/.evoscientist/sessions.db` (`sessions.py:108` `get_db_path`), every completed
|
||||
super-step is on disk before the next begins. An interrupted run therefore
|
||||
leaves a recoverable checkpoint whose `next` tuple is non-empty (pending nodes)
|
||||
and whose latest assistant message — if the model node finished — is the same
|
||||
content the WebUI failed to deliver.
|
||||
|
||||
## 5. Components
|
||||
|
||||
### C1 — Launch-mode flags + TUI default
|
||||
|
||||
**Files:**
|
||||
- `EvoScientist/cli/__init__.py` (Typer app entry) — add `--web` / `--tui`
|
||||
options on the root command; thread the resolved `ui_backend` into the existing
|
||||
dispatch.
|
||||
- `EvoScientist/cli/tui_runtime.py:12` — remove the divergent
|
||||
`DEFAULT_UI_BACKEND = "cli"`; route through `resolve_ui_backend` so
|
||||
`settings.py`'s `"tui"` default wins.
|
||||
- `EvoScientist/cli/commands.py` — wherever `ui_backend` is currently read, accept
|
||||
the flag override.
|
||||
|
||||
**Behavior:** Running `evoscientist` with no flags and default config launches
|
||||
the TUI and starts **no** `langgraph dev` subprocess. No port is bound; the
|
||||
orphan-process class of bug disappears for the default path.
|
||||
|
||||
**Acceptance:** `evoscientist` (no args, default config) does not open a
|
||||
listening socket on 6174 or any other port.
|
||||
|
||||
### C2 — TUI run resilience (Approach B)
|
||||
|
||||
Two cooperating pieces, both backed by `sessions.db`.
|
||||
|
||||
#### C2a — Interrupt-fallback in the streaming loop
|
||||
|
||||
Wrap the TUI streaming consumer so that on any interruption
|
||||
(`KeyboardInterrupt`, `asyncio.CancelledError`, unexpected `Exception`) it reads
|
||||
the thread's latest checkpoint and renders the latest assistant message before
|
||||
exiting.
|
||||
|
||||
**Files:**
|
||||
- `EvoScientist/stream/display.py` — `_run_streaming` and
|
||||
`_astream_to_console` (the two streaming entry points used by
|
||||
`cli/tui_backends.py`). Add a `finally`/except guard that, after an
|
||||
interruption, calls a new `render_latest_assistant(thread_id)` helper.
|
||||
- `EvoScientist/sessions.py` — add `get_latest_assistant_message(thread_id) ->
|
||||
str | None`: read the most recent checkpoint for the thread, walk `messages`
|
||||
backwards to the last `AIMessage` with non-empty content, return its text
|
||||
(joined content blocks). Reuses existing `AsyncSqliteSaver` access already in
|
||||
this module.
|
||||
|
||||
**Behavior on Ctrl-C mid-run:**
|
||||
1. The guard catches the interrupt.
|
||||
2. Prints a one-line notice: `Run interrupted — recovering latest output from
|
||||
checkpoint…`.
|
||||
3. Renders the latest persisted assistant message via the same Markdown renderer
|
||||
the normal completion path uses.
|
||||
4. If no assistant message is persisted yet (model node hadn't completed),
|
||||
prints `Run interrupted before any output was saved. Resume with: evoscientist
|
||||
--thread <id>` and exits.
|
||||
|
||||
**Non-goal:** this does not *continue* the run; it only guarantees visibility of
|
||||
whatever completed. Continuation is C2b.
|
||||
|
||||
#### C2b — Startup detection of interrupted runs
|
||||
|
||||
Runs only when the TUI is launched to **resume a specific thread** (via the
|
||||
existing thread-resume flag, i.e. the command `resume_hint.py` already prints at
|
||||
exit). For a brand-new session there is nothing to detect and the prompt is
|
||||
skipped. When resuming, check the thread's latest checkpoint: `next` is
|
||||
non-empty **and** no run is currently active for that thread. If so, prompt:
|
||||
|
||||
```
|
||||
You have an unfinished run in thread <id>:
|
||||
[s] show what completed
|
||||
[r] resume the run from its last checkpoint
|
||||
[n] start fresh
|
||||
```
|
||||
|
||||
**Files:**
|
||||
- `EvoScientist/sessions.py` — add `detect_interrupted_run(thread_id) ->
|
||||
InterruptedRunInfo | None` returning the checkpoint id, the pending node names
|
||||
(`next`), and the timestamp. Implemented as a thin read over the existing
|
||||
checkpointer's `aget_tuple`.
|
||||
- `EvoScientist/cli/tui_interactive.py` — near the existing thread-load path
|
||||
(around the `create_runtime_gateways()` call at line 297), invoke the detector
|
||||
for the active thread and render the prompt above.
|
||||
|
||||
**Resume semantics:** selecting `r` re-invokes the graph with the same
|
||||
`thread_id`. LangGraph resumes from the last checkpoint automatically (this is
|
||||
the same mechanism `evoscientist --thread <id>` already relies on for session
|
||||
continuity). No new graph code needed.
|
||||
|
||||
**Acceptance:** After killing a TUI mid-run with Ctrl-C, relaunching with
|
||||
`--thread <id>` offers `[s]`/`[r]`/`[n]`; `[s]` prints the persisted final
|
||||
answer; `[r]` continues execution from where it stopped.
|
||||
|
||||
### C3 — WebUI mode hardening (`--web`)
|
||||
|
||||
Applies only when `ui_backend == "webui"`. The goal is to make the existing
|
||||
`deploy/webui.py` launch path robust and to mitigate (not eliminate) the SSE
|
||||
limitation.
|
||||
|
||||
**Files:**
|
||||
- `EvoScientist/deploy/webui.py` — three changes:
|
||||
1. **Process-group + signal cleanup.** Today `atexit.register(stop_langgraph_dev,
|
||||
...)` only fires on clean exit. Wrap the spawned `langgraph dev` (and the
|
||||
`npx` child) in a process group (`start_new_session=True` on POSIX; on
|
||||
Windows use a job object via the existing `_winloop.py` helpers) and
|
||||
install `SIGINT`/`SIGTERM` handlers that tear the group down. This closes
|
||||
the orphan-PID-15288 path for normal terminations. `SIGKILL` cannot run
|
||||
handlers on either platform — document that hard kills may still orphan
|
||||
the child, and the port-conflict UX in (2) is the user-facing recovery.
|
||||
2. **Port-conflict UX.** The current "Port X is occupied by another process"
|
||||
message is already correct; extend it to detect *whether the occupant is a
|
||||
`langgraph dev`* (via the existing `/ok` health probe) and, if so, offer to
|
||||
reuse it rather than bail. If the occupant is foreign, print the `lsof`
|
||||
hint already present.
|
||||
3. **Resume window.** Set `RESUMABLE_STREAM_TTL_SECONDS=600` in the env passed
|
||||
to `start_langgraph_dev` so a browser that does reconnect can replay events
|
||||
for runs up to 10 minutes long (the observed run was 127 s vs. the 120 s
|
||||
default).
|
||||
- `EvoScientist/deploy/server.py` — `start_langgraph_dev` already accepts env;
|
||||
pass `RESUMABLE_STREAM_TTL_SECONDS` through.
|
||||
|
||||
**What C3 does NOT do:** it cannot make the browser auto-resume, because that
|
||||
logic lives in `@evoscientist/webui`. It only widens the server-side window and
|
||||
stops the orphan process. Documented in the WebUI exit message: *for long runs,
|
||||
prefer the TUI (`evoscientist --tui`); the browser client does not resume a
|
||||
dropped stream.*
|
||||
|
||||
### C4 — Documentation
|
||||
|
||||
- `README` / CLI `--help`: `--web` (browser), `--tui` (terminal, default,
|
||||
recommended for long runs).
|
||||
- A short "Why did my WebUI output stop?" note pointing at this design doc, so
|
||||
future users reading the symptom can find the explanation.
|
||||
|
||||
## 6. Data Flow
|
||||
|
||||
### Default (TUI) path
|
||||
```
|
||||
evoscientist ─▶ resolve_ui_backend ─▶ "tui"
|
||||
│
|
||||
(no langgraph dev, no port bound)
|
||||
▼
|
||||
create_runtime_gateways(backend="local")
|
||||
│
|
||||
LocalGraphGateway.stream_events
|
||||
│
|
||||
stream_agent_events (in-process) ── checkpoint per super-step ──▶ sessions.db
|
||||
│
|
||||
on interrupt ──▶ get_latest_assistant_message(thread_id) ──▶ render
|
||||
```
|
||||
|
||||
### WebUI (`--web`) path
|
||||
unchanged from today (langgraph dev + `npx @evoscientist/webui`), plus C3's
|
||||
process-group cleanup and `RESUMABLE_STREAM_TTL_SECONDS=600`.
|
||||
|
||||
## 7. Error Handling
|
||||
|
||||
| Failure | TUI default path | WebUI `--web` path |
|
||||
|---|---|---|
|
||||
| Ctrl-C mid-run | C2a renders latest persisted assistant message | Browser truncates; user runs `evoscientist --tui --thread <id>` → `[s]` to recover |
|
||||
| Crash / SIGHUP | Checkpoint on disk; next `--thread <id>` shows `[s]`/`[r]` prompt (C2b) | Same recovery via TUI |
|
||||
| Port 6174 bound | N/A — no port bound | C3 reuses if `langgraph dev`, else prints `lsof` hint |
|
||||
| Orphaned `langgraph dev` | Cannot originate from TUI path | C3 process-group + signal handlers tear it down |
|
||||
|
||||
## 8. Testing
|
||||
|
||||
- **Unit** (`tests/test_sessions_*.py`): `get_latest_assistant_message` returns
|
||||
the full final answer for a thread whose last run completed; returns `None`
|
||||
for a thread interrupted before the model node.
|
||||
- **Unit**: `detect_interrupted_run` returns pending `next` nodes for an
|
||||
interrupted thread; `None` for an idle one.
|
||||
- **Integration** (new `tests/test_tui_interrupt_recovery.py`): start a TUI run
|
||||
against a graph whose model node sleeps, send `SIGINT` mid-run, assert the
|
||||
process exits 0 after printing the recovered message.
|
||||
- **Integration**: relaunch with `--thread <id>`, assert the `[s]/[r]/[n]`
|
||||
prompt appears and `[r]` produces a completed run.
|
||||
- **Launch** (`tests/test_cli_launch.py`, extend): `evoscientist` with default
|
||||
config binds no port; `--web` binds 6174 and tears it down on `SIGTERM`.
|
||||
- **WebUI** (manual / smoke): `--web`, kill parent with `SIGKILL`, assert no
|
||||
orphan `langgraph dev` remains after the process-group change (best-effort —
|
||||
`SIGKILL` cannot run handlers, but the process group lets the OS reap
|
||||
children when the session ends; document this limit).
|
||||
|
||||
## 9. Rollout / Migration
|
||||
|
||||
- User's config currently has `ui_backend = "webui"`. After this change, that
|
||||
config still selects WebUI (back-compatible). The user can run `evoscientist
|
||||
--tui` immediately to get the resilient path, or `EvoSci config set ui_backend
|
||||
tui` to make it permanent.
|
||||
- No migration of `sessions.db` — schema is unchanged.
|
||||
- No breaking change to `EvoSci deploy`.
|
||||
|
||||
## 10. Open Questions / Future Work
|
||||
|
||||
- **`@evoscientist/webui` client-side resume.** The front-end could re-POST
|
||||
`/runs/stream` with `last-event-id` (or use the thread's SSE endpoint with
|
||||
`since`) after a disconnect. This is the only way to fully fix the WebUI
|
||||
symptom; it lives in the npm repo and is out of scope here. File a
|
||||
cross-repo issue referencing this design.
|
||||
- Whether `[r]` resume should replay the *interrupted* model call or only
|
||||
continue from the *next* node. LangGraph's default (resume from pending
|
||||
writes) is correct for most cases; revisit if users hit re-execution of
|
||||
expensive tool calls.
|
||||
- `cli` backend vs `tui` backend distinction — both are in-process; consider
|
||||
deprecating the `cli`/`tui` split in a follow-up once the resilience work
|
||||
lands, to reduce the three-way default confusion that caused the original
|
||||
`DEFAULT_UI_BACKEND` divergence.
|
||||
@@ -0,0 +1,196 @@
|
||||
# WebUI SSE Truncation Recovery via Checkpoint Fallback (α)
|
||||
|
||||
**Date:** 2026-07-06
|
||||
**Status:** Design — pending implementation plan
|
||||
**Scope:** `@evoscientist/webui` front-end (separate npm repo) + one optional convenience route in this Python repo
|
||||
|
||||
## 1. Background & Root Cause
|
||||
|
||||
A WebUI user saw a 127 s research run render truncated output, ending mid-token at
|
||||
`O-8282d7e`. Investigation established:
|
||||
|
||||
- The server-side run **completed cleanly**; the thread checkpoint in
|
||||
`~/.evoscientist/sessions.db` holds the full 6818-char final answer
|
||||
(`status: idle`, `error: None`).
|
||||
- The disconnect is **browser-side**: `langgraph dev` sends an SSE heartbeat every
|
||||
5 s, `send_timeout` is `None`, and the server supports resume via
|
||||
`last-event-id` (`langgraph_api/api/runs.py:623`). But `/runs/stream` is
|
||||
**POST**, so the browser reads it via `fetch()`; native `EventSource`
|
||||
auto-reconnect does not apply, and `@evoscientist/webui` does not re-POST to
|
||||
resume. When the connection drops mid-run, the front-end is left showing the
|
||||
last partial token and never fetches the completed answer.
|
||||
|
||||
**Core insight:** the server already has the complete answer in the checkpoint.
|
||||
The fix is to make the front-end fall back to that checkpoint when the stream
|
||||
ends abnormally. This is small, surgical, and directly addresses the symptom.
|
||||
|
||||
## 2. Goal
|
||||
|
||||
**The user always sees the complete final answer in the WebUI, even if the SSE
|
||||
stream drops mid-run.**
|
||||
|
||||
## 3. Non-Goals
|
||||
|
||||
- Seamless stream resume (tracking `last-event-id` / `seq` and replaying
|
||||
buffered events). That is option β — better UX, ~3–5× the code, deferred.
|
||||
- The TUI pivot / launch-mode decoupling (option γ). Separable; only relevant if
|
||||
the port-conflict / orphan-process issues need addressing.
|
||||
- Patching `langgraph_api` upstream.
|
||||
|
||||
## 4. Detection Logic — What Counts as "Truncated"
|
||||
|
||||
The langgraph SSE protocol emits terminal events when a run finishes
|
||||
(success → `end`, failure → `error`; exact names to be confirmed against the
|
||||
`@langchain/langgraph-sdk` version during implementation). The front-end's
|
||||
streaming reader loop currently processes events until the `fetch()` ReadableStream
|
||||
closes.
|
||||
|
||||
**Truncation** = the stream closed (reader returned) **without** a terminal event
|
||||
having been received. Causes: browser tab throttled, network blip, proxy idle
|
||||
timeout. On truncation, the in-progress assistant bubble is left showing a prefix
|
||||
of the real answer.
|
||||
|
||||
## 5. Components
|
||||
|
||||
### F1 — Truncation detector (front-end)
|
||||
|
||||
In the streaming reader loop, track a `terminalSeen` flag. Set it when a
|
||||
terminal event arrives. When the reader returns, if `!terminalSeen`, mark the
|
||||
run `truncated` and trigger F2.
|
||||
|
||||
**File:** `@evoscientist/webui` — the run-stream consumer (the module that wraps
|
||||
the langgraph-ts SDK's `runs.stream` and renders into the message list).
|
||||
|
||||
### F2 — Checkpoint fallback fetch (front-end)
|
||||
|
||||
On `truncated`, fetch the complete final assistant message and **replace** the
|
||||
in-progress bubble's content with it (not append — the partial tokens are a
|
||||
prefix of the same message).
|
||||
|
||||
Two implementation choices, pick one:
|
||||
|
||||
- **F2a (zero server change):** call the standard langgraph endpoint
|
||||
`GET /threads/{thread_id}/state`, walk `values.messages` backwards to the last
|
||||
`AIMessage`, join its content blocks. No new route, but the front-end parses
|
||||
raw state and reimplements "find latest assistant text" logic.
|
||||
- **F2b (recommended, uses S1 below):** call `GET /api/threads/{id}/final-answer`
|
||||
→ `{content, completed_at, complete}`. Server owns the parsing; front-end
|
||||
stays dumb.
|
||||
|
||||
Render a subtle affordance (e.g., a dim "⚠ stream dropped — recovered from
|
||||
checkpoint" line above the message) so the user knows a disconnect happened and
|
||||
the displayed text is authoritative, not stale streaming.
|
||||
|
||||
### F3 — Retry until the run settles (front-end)
|
||||
|
||||
The browser may drop while the run is **still executing** server-side. A single
|
||||
fallback fetch at drop-time could return a not-yet-final message. So F2 retries
|
||||
with backoff until either:
|
||||
|
||||
- `complete == true` (run finished — render and stop), or
|
||||
- a max wait elapses (default 240 s — comfortably exceeds the observed 127 s
|
||||
run even if the drop happens at t=0), then render whatever is latest and
|
||||
stop retrying.
|
||||
|
||||
Backoff: poll at 1 s, 2 s, 4 s, then every 5 s up to the max. Each poll replaces
|
||||
the bubble with the latest content, so if the run finishes mid-retry the user
|
||||
sees it update to the final form.
|
||||
|
||||
### S1 — Convenience route `GET /api/threads/{id}/final-answer` (this repo, optional but recommended)
|
||||
|
||||
Enables F2b. Mounted on the existing custom ASGI app already wired via
|
||||
`langgraph.json` → `EvoScientist.langgraph_dev.http:app` (currently only serves
|
||||
`/api/models`).
|
||||
|
||||
**Contract:**
|
||||
|
||||
```
|
||||
GET /api/threads/{thread_id}/final-answer
|
||||
200 → { "content": str, # last AIMessage text, blocks joined
|
||||
"completed_at": iso8601, # run end timestamp, or null
|
||||
"complete": bool } # true iff run reached terminal state
|
||||
404 → thread not found
|
||||
```
|
||||
|
||||
**`complete` derivation** (run reached a terminal state):
|
||||
`thread.status == "idle"` **or** latest checkpoint's `next` tuple is empty.
|
||||
|
||||
**File:** `EvoScientist/langgraph_dev/http.py` — add `get_final_answer` route
|
||||
beside the existing `get_models`. Implementation reads the thread state via the
|
||||
shared checkpointer (`AsyncSqliteSaver` at `~/.evoscientist/sessions.db`,
|
||||
already used by `sessions.py`). If direct checkpointer access is awkward from
|
||||
inside the mounted app, fall back to an internal `langgraph_sdk.get_client()`
|
||||
call to `GET /threads/{id}/state` on localhost — the contract stays identical.
|
||||
|
||||
**Why prefer F2b+S1 over F2a:** the "find latest AIMessage + join content blocks
|
||||
+ decide complete?" logic is non-trivial (content is a list of typed blocks,
|
||||
including `reasoning` blocks that should be excluded from the rendered answer —
|
||||
see the checkpoint: block[0] was `reasoning`, block[1] was `text`). Centralizing
|
||||
it server-side keeps the front-end thin and makes the same logic reusable by the
|
||||
TUI's future recovery path if γ is ever pursued.
|
||||
|
||||
## 6. Data Flow
|
||||
|
||||
```
|
||||
browser fetch(/runs/stream, POST) ──SSE──▶ langgraph dev
|
||||
│ │
|
||||
│ (mid-run, connection drops) │ run continues to completion
|
||||
▼ │
|
||||
reader returns, no terminal event ──▶ truncated │
|
||||
│ │
|
||||
│ GET /api/threads/{id}/final-answer │
|
||||
▼ ▼
|
||||
http.py::get_final_answer ──read checkpoint──▶ sessions.db
|
||||
│
|
||||
▼
|
||||
{content, complete} ──▶ if !complete, retry (F3); else render (F2)
|
||||
```
|
||||
|
||||
## 7. Error Handling
|
||||
|
||||
| Case | Behavior |
|
||||
|---|---|
|
||||
| Stream ends normally (`end`/`error` received) | F1 sets `terminalSeen`; no fallback; existing render path unchanged |
|
||||
| Stream drops, run already finished server-side | First fallback fetch returns `complete=true`; render immediately |
|
||||
| Stream drops, run still executing | F3 retries with backoff; each retry renders latest; stops when `complete` or 120 s max |
|
||||
| Thread unknown / deleted | 404; front-end shows "stream interrupted and the session could not be recovered" |
|
||||
| Convenience route unreachable (older server without S1) | Front-end falls back to F2a (raw state endpoint) — degrade gracefully |
|
||||
|
||||
## 8. Testing
|
||||
|
||||
**Front-end (`@evoscientist/webui`):**
|
||||
- Unit: feed the reader a synthetic event stream that closes without a terminal
|
||||
event → assert `truncated == true`; feed one with `end` → `false`.
|
||||
- Integration: mock `fetch` to drop mid-stream; assert the fallback fetch fires
|
||||
and the bubble's content is replaced with the checkpoint value.
|
||||
|
||||
**Server (this repo, if S1):** new `tests/test_http_final_answer.py`
|
||||
- Completed thread → 200, `complete=true`, content matches the last AIMessage
|
||||
text (not the reasoning block).
|
||||
- Mid-run thread (forced `busy` / non-empty `next`) → 200, `complete=false`.
|
||||
- Unknown thread → 404.
|
||||
- Content with mixed `reasoning` + `text` blocks → only `text` in `content`.
|
||||
|
||||
## 9. Cross-Repo Coordination & Rollout
|
||||
|
||||
- **Order:** land S1 in this Python repo first (route + tests, behind no flag —
|
||||
it's a pure addition). Then F1/F2/F3 in `@evoscientist/webui`.
|
||||
- **Versioning:** the WebUI launches via `npx @evoscientist/webui@latest`, so
|
||||
the front-end fix reaches users on their next launch automatically once
|
||||
published. No coordinated upgrade required on the Python side.
|
||||
- **Backward compatibility:** if a user runs a new front-end against an older
|
||||
Python server without S1, the front-end must detect the 404 and fall back to
|
||||
F2a (raw state endpoint). If an old front-end runs against a new server, S1
|
||||
is simply unused — no harm.
|
||||
|
||||
## 10. Open Questions / Future Work
|
||||
|
||||
- **Exact terminal-event names** for the SDK version in use (`end`/`error` vs.
|
||||
`done`/`result`). Confirm against `@langchain/langgraph-sdk` during
|
||||
implementation; the detector is parameterized on this.
|
||||
- **True resume (β):** track `last-event-id` (or the V2 thread-stream `seq`) and
|
||||
replay buffered events on reconnect — seamless, no visible "recovered" flash.
|
||||
Larger front-end effort; defer until α is validated.
|
||||
- **TUI pivot (γ):** separable work to make the in-process TUI the default and
|
||||
eliminate the port-conflict / orphan-process issues. Independent spec if
|
||||
pursued.
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "EvoScientist"
|
||||
version = "0.2.1"
|
||||
version = "0.2.2"
|
||||
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
|
||||
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
from langchain_core.messages import AIMessage, HumanMessage
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from EvoScientist.config import EvoScientistConfig
|
||||
@@ -161,3 +162,143 @@ def test_ollama_discovery_skipped_when_base_url_absent():
|
||||
{"name": n, "model_id": m, "provider": p}
|
||||
for n, m, p in list_models_by_provider()
|
||||
]
|
||||
|
||||
|
||||
def test_final_answer_extracts_latest_ai_text_blocks():
|
||||
async def fake_metadata(_thread_id):
|
||||
return {"updated_at": "2026-07-06T14:14:53+00:00"}
|
||||
|
||||
async def fake_messages(_thread_id):
|
||||
return [
|
||||
HumanMessage(content="question"),
|
||||
AIMessage(content="old answer"),
|
||||
AIMessage(
|
||||
content=[
|
||||
{"type": "reasoning", "text": "internal"},
|
||||
{"type": "text", "text": "Part A"},
|
||||
{"type": "tool_use", "name": "search"},
|
||||
{"type": "output_text", "text": "Part B"},
|
||||
]
|
||||
),
|
||||
]
|
||||
|
||||
async def fake_runtime(_request, _thread_id):
|
||||
return {
|
||||
"found": True,
|
||||
"complete": True,
|
||||
"completed_at": "2026-07-06T14:15:00+00:00",
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||
new=fake_metadata,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
|
||||
new=fake_messages,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
|
||||
new=fake_runtime,
|
||||
),
|
||||
):
|
||||
resp = client.get("/api/threads/thread-1/final-answer")
|
||||
|
||||
assert resp.status_code == 200
|
||||
assert resp.json() == {
|
||||
"content": "Part A\n\nPart B",
|
||||
"completed_at": "2026-07-06T14:15:00+00:00",
|
||||
"complete": True,
|
||||
}
|
||||
|
||||
|
||||
def test_final_answer_skips_tool_selection_json_text():
|
||||
async def fake_metadata(_thread_id):
|
||||
return {"updated_at": "2026-07-06T14:14:53+00:00"}
|
||||
|
||||
async def fake_messages(_thread_id):
|
||||
return [
|
||||
HumanMessage(content="question"),
|
||||
AIMessage(content="stable answer"),
|
||||
AIMessage(
|
||||
content=(
|
||||
'{"tools":["search_papers","get_abstract"]}'
|
||||
'{"tools":["web_search_exa"]}'
|
||||
)
|
||||
),
|
||||
]
|
||||
|
||||
async def fake_runtime(_request, _thread_id):
|
||||
return {
|
||||
"found": True,
|
||||
"complete": True,
|
||||
"completed_at": "2026-07-06T14:15:00+00:00",
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||
new=fake_metadata,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
|
||||
new=fake_messages,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
|
||||
new=fake_runtime,
|
||||
),
|
||||
):
|
||||
resp = client.get("/api/threads/thread-1/final-answer")
|
||||
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["content"] == "stable answer"
|
||||
|
||||
|
||||
def test_final_answer_returns_404_for_unknown_thread():
|
||||
async def fake_metadata(_thread_id):
|
||||
return None
|
||||
|
||||
with patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||
new=fake_metadata,
|
||||
):
|
||||
resp = client.get("/api/threads/missing/final-answer")
|
||||
|
||||
assert resp.status_code == 404
|
||||
assert resp.json() == {"error": "thread not found"}
|
||||
|
||||
|
||||
def test_final_answer_does_not_mark_complete_when_runtime_state_fails():
|
||||
async def fake_metadata(_thread_id):
|
||||
return {"updated_at": "2026-07-06T14:14:53+00:00"}
|
||||
|
||||
async def fake_messages(_thread_id):
|
||||
return [AIMessage(content="checkpoint answer")]
|
||||
|
||||
async def fake_runtime(_request, _thread_id):
|
||||
raise RuntimeError("langgraph runtime unavailable")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||
new=fake_metadata,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
|
||||
new=fake_messages,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
|
||||
new=fake_runtime,
|
||||
),
|
||||
):
|
||||
resp = client.get("/api/threads/thread-1/final-answer")
|
||||
|
||||
assert resp.status_code == 200
|
||||
assert resp.json() == {
|
||||
"content": "checkpoint answer",
|
||||
"completed_at": None,
|
||||
"complete": False,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from EvoScientist.deploy.webui import _resolve_webui_package
|
||||
|
||||
|
||||
def test_resolve_webui_package_defaults_to_published_package(monkeypatch):
|
||||
monkeypatch.delenv("EVOSCIENTIST_WEBUI_PACKAGE", raising=False)
|
||||
|
||||
assert _resolve_webui_package() == "@evoscientist/webui@latest"
|
||||
|
||||
|
||||
def test_resolve_webui_package_allows_local_override(monkeypatch):
|
||||
monkeypatch.setenv("EVOSCIENTIST_WEBUI_PACKAGE", "/tmp/EvoScientist-WebUI")
|
||||
|
||||
assert _resolve_webui_package() == "/tmp/EvoScientist-WebUI"
|
||||
Reference in New Issue
Block a user