Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e0acc6155e |
@@ -48,3 +48,4 @@ conversation_history/
|
|||||||
*meals/
|
*meals/
|
||||||
botpy.log
|
botpy.log
|
||||||
large_tool_results/
|
large_tool_results/
|
||||||
|
runs/
|
||||||
|
|||||||
@@ -41,9 +41,15 @@ from ..stream.console import console
|
|||||||
|
|
||||||
# Front-end npm package + spec. ``@latest`` → always the newest published UI.
|
# Front-end npm package + spec. ``@latest`` → always the newest published UI.
|
||||||
_WEBUI_PACKAGE = "@evoscientist/webui@latest"
|
_WEBUI_PACKAGE = "@evoscientist/webui@latest"
|
||||||
|
_WEBUI_PACKAGE_ENV = "EVOSCIENTIST_WEBUI_PACKAGE"
|
||||||
_DEFAULT_WEBUI_PORT = 4716
|
_DEFAULT_WEBUI_PORT = 4716
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_webui_package() -> str:
|
||||||
|
"""Return the npm package spec to launch for the WebUI front-end."""
|
||||||
|
return os.getenv(_WEBUI_PACKAGE_ENV) or _WEBUI_PACKAGE
|
||||||
|
|
||||||
|
|
||||||
def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
||||||
"""Start the deploy-style backend + the WebUI front-end, then block.
|
"""Start the deploy-style backend + the WebUI front-end, then block.
|
||||||
|
|
||||||
@@ -203,6 +209,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
|||||||
"PORT": str(webui_port),
|
"PORT": str(webui_port),
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
webui_package = _resolve_webui_package()
|
||||||
console.print(
|
console.print(
|
||||||
Panel(
|
Panel(
|
||||||
Text.from_markup(
|
Text.from_markup(
|
||||||
@@ -211,7 +218,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
|||||||
f"[bold]WebUI:[/bold] http://localhost:{webui_port} "
|
f"[bold]WebUI:[/bold] http://localhost:{webui_port} "
|
||||||
f"[dim](opens in your browser)[/dim]\n"
|
f"[dim](opens in your browser)[/dim]\n"
|
||||||
f"[bold]Logs:[/bold] {_shorten(str(RUNTIME.log_file))}\n\n"
|
f"[bold]Logs:[/bold] {_shorten(str(RUNTIME.log_file))}\n\n"
|
||||||
f"[dim]Fetching {_WEBUI_PACKAGE} via npx (first run may take a "
|
f"[dim]Fetching {webui_package} via npx (first run may take a "
|
||||||
f"moment)… Press Ctrl+C to stop.[/dim]"
|
f"moment)… Press Ctrl+C to stop.[/dim]"
|
||||||
),
|
),
|
||||||
title="[bold green]✓ EvoScientist WebUI[/bold green]",
|
title="[bold green]✓ EvoScientist WebUI[/bold green]",
|
||||||
@@ -228,7 +235,7 @@ def run_webui(config: Any, workspace_dir: str | None = None) -> None:
|
|||||||
popen_kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP
|
popen_kwargs["creationflags"] = subprocess.CREATE_NEW_PROCESS_GROUP
|
||||||
try:
|
try:
|
||||||
webui_proc = subprocess.Popen(
|
webui_proc = subprocess.Popen(
|
||||||
[npx, "--yes", _WEBUI_PACKAGE, "--port", str(webui_port)],
|
[npx, "--yes", webui_package, "--port", str(webui_port)],
|
||||||
**popen_kwargs,
|
**popen_kwargs,
|
||||||
)
|
)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
|
|||||||
@@ -23,7 +23,17 @@ memory.
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import sqlite3
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import aiosqlite
|
||||||
|
import httpx
|
||||||
|
from langgraph.checkpoint.serde.jsonplus import JsonPlusSerializer
|
||||||
|
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
|
||||||
from starlette.applications import Starlette
|
from starlette.applications import Starlette
|
||||||
from starlette.requests import Request
|
from starlette.requests import Request
|
||||||
from starlette.responses import JSONResponse
|
from starlette.responses import JSONResponse
|
||||||
@@ -31,6 +41,15 @@ from starlette.routing import Route
|
|||||||
|
|
||||||
from EvoScientist.config import get_effective_config
|
from EvoScientist.config import get_effective_config
|
||||||
from EvoScientist.llm.models import list_model_picker_entries
|
from EvoScientist.llm.models import list_model_picker_entries
|
||||||
|
from EvoScientist.sessions import (
|
||||||
|
MAIN_THREAD_FILTER_PARAMS,
|
||||||
|
MAIN_THREAD_FILTER_SQL,
|
||||||
|
_load_checkpoint_messages,
|
||||||
|
_table_exists,
|
||||||
|
_to_short_path,
|
||||||
|
)
|
||||||
|
|
||||||
|
_logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
async def get_models(_request: Request) -> JSONResponse:
|
async def get_models(_request: Request) -> JSONResponse:
|
||||||
@@ -75,8 +94,270 @@ async def get_models(_request: Request) -> JSONResponse:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _message_type(message: Any) -> str | None:
|
||||||
|
if isinstance(message, dict):
|
||||||
|
role = message.get("role")
|
||||||
|
if role == "assistant":
|
||||||
|
return "ai"
|
||||||
|
if role == "user":
|
||||||
|
return "human"
|
||||||
|
value = message.get("type")
|
||||||
|
return str(value) if value is not None else None
|
||||||
|
value = getattr(message, "type", None)
|
||||||
|
return str(value) if value is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _message_content(message: Any) -> Any:
|
||||||
|
if isinstance(message, dict):
|
||||||
|
return message.get("content")
|
||||||
|
return getattr(message, "content", None)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_text_content(content: Any) -> str:
|
||||||
|
if isinstance(content, str):
|
||||||
|
return content
|
||||||
|
if not isinstance(content, list):
|
||||||
|
return ""
|
||||||
|
|
||||||
|
parts: list[str] = []
|
||||||
|
for block in content:
|
||||||
|
if isinstance(block, str):
|
||||||
|
parts.append(block)
|
||||||
|
continue
|
||||||
|
if not isinstance(block, dict):
|
||||||
|
continue
|
||||||
|
block_type = block.get("type")
|
||||||
|
if block_type not in {"text", "output_text"}:
|
||||||
|
continue
|
||||||
|
text = block.get("text") or block.get("content")
|
||||||
|
if isinstance(text, str):
|
||||||
|
parts.append(text)
|
||||||
|
return "\n\n".join(part for part in parts if part)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_tool_selection_payload(raw: str) -> bool:
|
||||||
|
try:
|
||||||
|
parsed = json.loads(raw)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
return False
|
||||||
|
return (
|
||||||
|
isinstance(parsed, dict)
|
||||||
|
and set(parsed) == {"tools"}
|
||||||
|
and isinstance(parsed["tools"], list)
|
||||||
|
and all(isinstance(tool, str) for tool in parsed["tools"])
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _split_json_objects(raw: str) -> list[str] | None:
|
||||||
|
objects: list[str] = []
|
||||||
|
depth = 0
|
||||||
|
start = -1
|
||||||
|
in_string = False
|
||||||
|
escaping = False
|
||||||
|
|
||||||
|
for i, char in enumerate(raw):
|
||||||
|
if in_string:
|
||||||
|
if escaping:
|
||||||
|
escaping = False
|
||||||
|
elif char == "\\":
|
||||||
|
escaping = True
|
||||||
|
elif char == '"':
|
||||||
|
in_string = False
|
||||||
|
continue
|
||||||
|
|
||||||
|
if char == '"':
|
||||||
|
if depth == 0:
|
||||||
|
return None
|
||||||
|
in_string = True
|
||||||
|
continue
|
||||||
|
if char == "{":
|
||||||
|
if depth == 0:
|
||||||
|
start = i
|
||||||
|
depth += 1
|
||||||
|
continue
|
||||||
|
if char == "}":
|
||||||
|
depth -= 1
|
||||||
|
if depth < 0 or start < 0:
|
||||||
|
return None
|
||||||
|
if depth == 0:
|
||||||
|
objects.append(raw[start : i + 1])
|
||||||
|
start = -1
|
||||||
|
continue
|
||||||
|
if depth == 0 and not char.isspace():
|
||||||
|
return None
|
||||||
|
|
||||||
|
if depth != 0 or in_string or not objects:
|
||||||
|
return None
|
||||||
|
return objects
|
||||||
|
|
||||||
|
|
||||||
|
def _is_tool_selection_text(text: str) -> bool:
|
||||||
|
stripped = text.strip()
|
||||||
|
if not stripped or '"tools"' not in stripped:
|
||||||
|
return False
|
||||||
|
if _is_tool_selection_payload(stripped):
|
||||||
|
return True
|
||||||
|
objects = _split_json_objects(stripped)
|
||||||
|
return objects is not None and all(_is_tool_selection_payload(obj) for obj in objects)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_final_answer(messages: list[Any]) -> str:
|
||||||
|
"""Return displayable text from the latest AI message in *messages*."""
|
||||||
|
for message in reversed(messages):
|
||||||
|
if _message_type(message) != "ai":
|
||||||
|
continue
|
||||||
|
content = _extract_text_content(_message_content(message)).strip()
|
||||||
|
if content and not _is_tool_selection_text(content):
|
||||||
|
return content
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def _sessions_db_path_for_http() -> Path:
|
||||||
|
data_dir = os.getenv("EVOSCIENTIST_DATA_DIR")
|
||||||
|
base = Path(data_dir).expanduser() if data_dir else Path.home() / ".evoscientist"
|
||||||
|
return Path(_to_short_path(str(base))) / "sessions.db"
|
||||||
|
|
||||||
|
|
||||||
|
async def _get_thread_metadata_for_http(thread_id: str) -> dict | None:
|
||||||
|
try:
|
||||||
|
async with aiosqlite.connect(
|
||||||
|
str(_sessions_db_path_for_http()), timeout=30.0
|
||||||
|
) as conn:
|
||||||
|
if not await _table_exists(conn, "checkpoints"):
|
||||||
|
return None
|
||||||
|
query = f"""
|
||||||
|
SELECT json_extract(metadata, '$.workspace_dir') as workspace_dir,
|
||||||
|
json_extract(metadata, '$.model') as model,
|
||||||
|
json_extract(metadata, '$.updated_at') as updated_at
|
||||||
|
FROM checkpoints
|
||||||
|
WHERE thread_id = ?
|
||||||
|
AND {MAIN_THREAD_FILTER_SQL}
|
||||||
|
ORDER BY checkpoint_id DESC
|
||||||
|
LIMIT 1
|
||||||
|
"""
|
||||||
|
async with conn.execute(
|
||||||
|
query, (thread_id, *MAIN_THREAD_FILTER_PARAMS)
|
||||||
|
) as cur:
|
||||||
|
row = await cur.fetchone()
|
||||||
|
except (OSError, sqlite3.Error):
|
||||||
|
return None
|
||||||
|
if not row:
|
||||||
|
return None
|
||||||
|
return {
|
||||||
|
"workspace_dir": row[0],
|
||||||
|
"model": row[1],
|
||||||
|
"updated_at": row[2],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def _get_thread_messages_for_http(thread_id: str) -> list:
|
||||||
|
try:
|
||||||
|
async with aiosqlite.connect(
|
||||||
|
str(_sessions_db_path_for_http()), timeout=30.0
|
||||||
|
) as conn:
|
||||||
|
if not await _table_exists(conn, "checkpoints"):
|
||||||
|
return []
|
||||||
|
check = f"""
|
||||||
|
SELECT 1 FROM checkpoints
|
||||||
|
WHERE thread_id = ? AND {MAIN_THREAD_FILTER_SQL}
|
||||||
|
LIMIT 1
|
||||||
|
"""
|
||||||
|
async with conn.execute(
|
||||||
|
check, (thread_id, *MAIN_THREAD_FILTER_PARAMS)
|
||||||
|
) as cur:
|
||||||
|
if not await cur.fetchone():
|
||||||
|
return []
|
||||||
|
serde = JsonPlusSerializer()
|
||||||
|
saver = AsyncSqliteSaver(conn, serde=serde)
|
||||||
|
return await _load_checkpoint_messages(saver, thread_id)
|
||||||
|
except (OSError, sqlite3.Error):
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
async def _read_thread_runtime_state(request: Request, thread_id: str) -> dict[str, Any]:
|
||||||
|
"""Read thread status from the co-hosted langgraph-api endpoints."""
|
||||||
|
base_url = f"{request.url.scheme}://{request.url.netloc}"
|
||||||
|
timeout = httpx.Timeout(2.0, connect=0.5)
|
||||||
|
async with httpx.AsyncClient(base_url=base_url, timeout=timeout) as client:
|
||||||
|
thread_resp = await client.get(f"/threads/{thread_id}")
|
||||||
|
if thread_resp.status_code == 404:
|
||||||
|
return {"found": False}
|
||||||
|
thread_resp.raise_for_status()
|
||||||
|
thread = thread_resp.json()
|
||||||
|
|
||||||
|
state: dict[str, Any] = {}
|
||||||
|
state_resp = await client.get(f"/threads/{thread_id}/state")
|
||||||
|
if state_resp.status_code == 404:
|
||||||
|
return {"found": False}
|
||||||
|
if state_resp.status_code < 400:
|
||||||
|
state = state_resp.json()
|
||||||
|
|
||||||
|
next_nodes = state.get("next")
|
||||||
|
is_terminal_checkpoint = isinstance(next_nodes, (list, tuple)) and not next_nodes
|
||||||
|
status = thread.get("status")
|
||||||
|
complete = status == "idle" or is_terminal_checkpoint
|
||||||
|
completed_at = None
|
||||||
|
if complete:
|
||||||
|
completed_at = (
|
||||||
|
thread.get("state_updated_at")
|
||||||
|
or thread.get("updated_at")
|
||||||
|
or (state.get("metadata") or {}).get("updated_at")
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"found": True,
|
||||||
|
"complete": complete,
|
||||||
|
"completed_at": completed_at,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def get_final_answer(request: Request) -> JSONResponse:
|
||||||
|
"""Return the latest checkpointed assistant answer for a WebUI thread.
|
||||||
|
|
||||||
|
This is a recovery surface for the WebUI stream consumer: when browser-side
|
||||||
|
SSE is interrupted but the langgraph run continues server-side, the final
|
||||||
|
answer is already persisted in ``sessions.db``. The route centralizes the
|
||||||
|
non-trivial "latest AIMessage text only" extraction so the browser does not
|
||||||
|
render reasoning blocks, tool calls, or tool-selection JSON fragments.
|
||||||
|
"""
|
||||||
|
thread_id = request.path_params["thread_id"]
|
||||||
|
metadata = await _get_thread_metadata_for_http(thread_id)
|
||||||
|
if metadata is None:
|
||||||
|
return JSONResponse({"error": "thread not found"}, status_code=404)
|
||||||
|
|
||||||
|
messages = await _get_thread_messages_for_http(thread_id)
|
||||||
|
content = _extract_final_answer(messages)
|
||||||
|
|
||||||
|
runtime: dict[str, Any] = {"found": True, "complete": False, "completed_at": None}
|
||||||
|
try:
|
||||||
|
runtime = await _read_thread_runtime_state(request, thread_id)
|
||||||
|
except Exception as exc:
|
||||||
|
_logger.debug(
|
||||||
|
"Could not read langgraph runtime state for thread %s: %s",
|
||||||
|
thread_id,
|
||||||
|
exc,
|
||||||
|
)
|
||||||
|
if runtime.get("found") is False:
|
||||||
|
return JSONResponse({"error": "thread not found"}, status_code=404)
|
||||||
|
|
||||||
|
completed_at = runtime.get("completed_at")
|
||||||
|
if completed_at is None and runtime.get("complete"):
|
||||||
|
completed_at = metadata.get("updated_at")
|
||||||
|
return JSONResponse(
|
||||||
|
{
|
||||||
|
"content": content,
|
||||||
|
"completed_at": completed_at,
|
||||||
|
"complete": bool(runtime.get("complete")),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
app = Starlette(
|
app = Starlette(
|
||||||
routes=[
|
routes=[
|
||||||
Route("/api/models", get_models, methods=["GET"]),
|
Route("/api/models", get_models, methods=["GET"]),
|
||||||
|
Route(
|
||||||
|
"/api/threads/{thread_id}/final-answer",
|
||||||
|
get_final_answer,
|
||||||
|
methods=["GET"],
|
||||||
|
),
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -474,6 +474,15 @@ def get_chat_model(
|
|||||||
# Even native thinking models like kimi-k2-thinking operate in non-thinking mode.
|
# Even native thinking models like kimi-k2-thinking operate in non-thinking mode.
|
||||||
if provider == "moonshot":
|
if provider == "moonshot":
|
||||||
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
|
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
|
||||||
|
# custom-openai: some Codex-orientated reseller gateways (e.g.
|
||||||
|
# hunnuapi.top) blacklist the openai SDK's default
|
||||||
|
# "OpenAI/Python" User-Agent at the app layer and return
|
||||||
|
# 403 "Your request was blocked." to any langchain-openai client.
|
||||||
|
# Spoof a Codex-CLI UA so requests go through.
|
||||||
|
if provider == "custom-openai":
|
||||||
|
kwargs.setdefault("default_headers", {}).setdefault(
|
||||||
|
"User-Agent", "codex_cli_rs/0.0.0"
|
||||||
|
)
|
||||||
provider = "openai"
|
provider = "openai"
|
||||||
|
|
||||||
# OpenRouter → native ChatOpenRouter via init_chat_model.
|
# OpenRouter → native ChatOpenRouter via init_chat_model.
|
||||||
|
|||||||
@@ -10,7 +10,7 @@
|
|||||||
<a href="https://pypi.org/project/EvoScientist/"><picture>
|
<a href="https://pypi.org/project/EvoScientist/"><picture>
|
||||||
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg">
|
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg">
|
||||||
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-dark.svg">
|
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-dark.svg">
|
||||||
<img alt="PyPI v0.2.1" src="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg" height="28">
|
<img alt="PyPI v0.2.2" src="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-pypi-light.svg" height="28">
|
||||||
</picture></a><a href="https://EvoScientist.github.io/"><picture>
|
</picture></a><a href="https://EvoScientist.github.io/"><picture>
|
||||||
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-light.svg">
|
<source media="(prefers-color-scheme: light)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-light.svg">
|
||||||
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-dark.svg">
|
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/EvoScientist/EvoScientist/main/.github/assets/badge-website-dark.svg">
|
||||||
@@ -137,6 +137,34 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
|
|||||||
> [!TIP]
|
> [!TIP]
|
||||||
> Looking for ready-to-use research skills? Check out [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) — powered by [**EvoScientist**](https://github.com/EvoScientist/EvoScientist)'s engine and installable skills, the entire end-to-end research lifecycle is covered out of the box. [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) are also compatible with other CLI coding agents.
|
> Looking for ready-to-use research skills? Check out [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) — powered by [**EvoScientist**](https://github.com/EvoScientist/EvoScientist)'s engine and installable skills, the entire end-to-end research lifecycle is covered out of the box. [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) are also compatible with other CLI coding agents.
|
||||||
|
|
||||||
|
## WebUI Streaming Fix Notes
|
||||||
|
|
||||||
|
The WebUI streaming fix is based on one rule: a browser-side SSE disconnect is
|
||||||
|
only a transport event, not proof that the LangGraph run has finished.
|
||||||
|
|
||||||
|
- Keep the composer action button as **Stop** while the backend thread still has
|
||||||
|
pending work. The UI should not switch back to **Send** merely because the
|
||||||
|
live SSE stream ended.
|
||||||
|
- Derive the visible busy state from both the live stream and backend thread
|
||||||
|
state. In the patched WebUI this is represented as `isRunActive`, backed by
|
||||||
|
`threads.getState(threadId)`.
|
||||||
|
- Restore **Send** only after the backend reports a terminal state: idle,
|
||||||
|
completed final answer, failed, or cancelled.
|
||||||
|
- Preserve human-in-the-loop controls during tool approvals. Approval buttons
|
||||||
|
remain actionable, while the bottom composer still shows **Stop** to avoid
|
||||||
|
duplicate submissions.
|
||||||
|
- Treat leaked `{"tools":[...]}` payloads as internal tool-selection control
|
||||||
|
messages, not assistant text, and filter them from the transcript.
|
||||||
|
- Recover dropped long-response tails through the checkpoint-backed
|
||||||
|
`/api/threads/{thread_id}/final-answer` endpoint instead of relying only on
|
||||||
|
browser stream state.
|
||||||
|
- Use a long enough final-answer polling window for real research runs; short
|
||||||
|
fallback windows can expire before the backend completes and make the browser
|
||||||
|
appear truncated until refresh.
|
||||||
|
- Validate with real browser flows: normal streaming, simulated SSE disconnect
|
||||||
|
and reload, pending tool approval, rejection/cancellation, and final transition
|
||||||
|
back to **Send**.
|
||||||
|
|
||||||
## 🔥 News
|
## 🔥 News
|
||||||
- **[03 Jun 2026]** 🥈 Ranked #2 overall — and 🥇 #1 among `GPT-5.4`-based agents — on [ResearchClawBench](https://github.com/InternScience/ResearchClawBench) (Agent Mode)! [**Leaderboard**](https://internscience.github.io/ResearchClawBench-Home/) 👈
|
- **[03 Jun 2026]** 🥈 Ranked #2 overall — and 🥇 #1 among `GPT-5.4`-based agents — on [ResearchClawBench](https://github.com/InternScience/ResearchClawBench) (Agent Mode)! [**Leaderboard**](https://internscience.github.io/ResearchClawBench-Home/) 👈
|
||||||
- **[18 Apr 2026]** 🥇 Ranked #1 on [DeepResearch Bench](https://deepresearch-bench.github.io/) at submission time! [**Leaderboard**](https://huggingface.co/spaces/muset-ai/DeepResearch-Bench-Leaderboard) 👈
|
- **[18 Apr 2026]** 🥇 Ranked #1 on [DeepResearch Bench](https://deepresearch-bench.github.io/) at submission time! [**Leaderboard**](https://huggingface.co/spaces/muset-ai/DeepResearch-Bench-Leaderboard) 👈
|
||||||
@@ -151,6 +179,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
|
|||||||
<details>
|
<details>
|
||||||
<summary>📦 Release Highlights — version changelog</summary>
|
<summary>📦 Release Highlights — version changelog</summary>
|
||||||
|
|
||||||
|
- **[07 Jul 2026]** **[v0.2.2](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.2)** — WebUI streaming resilience hotfix: SSE disconnects no longer imply run completion; the composer stays on **Stop** until backend thread state is terminal; final-answer checkpoint recovery fills dropped response tails; tool-selection JSON payloads are filtered from live transcripts; `EVOSCIENTIST_WEBUI_PACKAGE` lets the launcher run a local patched WebUI package for validation before npm publication.
|
||||||
- **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5.
|
- **[05 Jul 2026]** **[v0.2.1](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.1)** — AutoSkills: EvoMemory drafts reusable skills from its own observation clusters for you to review via `/autoskills`; a new `--output-format stream-json` for headless / SDK clients; richer slash-command completions; Windows UTF-8 config reads; a TUI welcome-banner fix; langchain-openrouter 0.2.5.
|
||||||
- **[26 Jun 2026]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — Scheduled tasks: cron-style recurring runs via `/schedule` or natural language, run unattended with shell-access gating; self-linking memory that connects observations into a knowledge graph (complements / contradicts / supersedes); a read-only `GET /api/models` endpoint for the WebUI model picker.
|
- **[26 Jun 2026]** **[v0.2.0](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.2.0)** — Scheduled tasks: cron-style recurring runs via `/schedule` or natural language, run unattended with shell-access gating; self-linking memory that connects observations into a knowledge graph (complements / contradicts / supersedes); a read-only `GET /api/models` endpoint for the WebUI model picker.
|
||||||
- **[23 Jun 2026]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — Hotfix for fresh installs: the first message crashed with `The subagent `task` tool cannot be exposed via `ptc`` after deepagents 0.6.11 / langchain-quickjs 0.3 reserved `task` as the REPL global. Removed `task` from the code-interpreter PTC allowlist (`task()` stays available as the REPL global; async dispatch stays in PTC) and pinned `deepagents[quickjs]~=0.6.11`.
|
- **[23 Jun 2026]** **[v0.1.9](https://github.com/EvoScientist/EvoScientist/releases/tag/v0.1.9)** — Hotfix for fresh installs: the first message crashed with `The subagent `task` tool cannot be exposed via `ptc`` after deepagents 0.6.11 / langchain-quickjs 0.3 reserved `task` as the REPL global. Removed `task` from the code-interpreter PTC allowlist (`task()` stays available as the REPL global; async dispatch stays in PTC) and pinned `deepagents[quickjs]~=0.6.11`.
|
||||||
@@ -180,6 +209,7 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human
|
|||||||
- [📦 Installation](#-installation)
|
- [📦 Installation](#-installation)
|
||||||
- [🔑 Configuration](#-configuration)
|
- [🔑 Configuration](#-configuration)
|
||||||
- [⚡ Quick Start](#-quick-start)
|
- [⚡ Quick Start](#-quick-start)
|
||||||
|
- [WebUI Streaming Fix Notes](#webui-streaming-fix-notes)
|
||||||
- [⏰ Scheduled Tasks](#-scheduled-tasks)
|
- [⏰ Scheduled Tasks](#-scheduled-tasks)
|
||||||
- [🍪 Examples & Recipes](#-examples--recipes)
|
- [🍪 Examples & Recipes](#-examples--recipes)
|
||||||
- [🔌 MCP Integration](#-mcp-integration)
|
- [🔌 MCP Integration](#-mcp-integration)
|
||||||
|
|||||||
@@ -0,0 +1,303 @@
|
|||||||
|
# TUI as Primary Interface + Run Resilience
|
||||||
|
|
||||||
|
**Date:** 2026-07-06
|
||||||
|
**Status:** Design — pending implementation plan
|
||||||
|
**Scope:** EvoScientist CLI/TUI launch path, run-interruption recovery, WebUI opt-in hardening
|
||||||
|
|
||||||
|
## 1. Background & Root Cause
|
||||||
|
|
||||||
|
A user running the WebUI backend (`ui_backend = "webui"`) saw a long research run
|
||||||
|
(127 s) render truncated output in the browser, ending mid-token at `O-8282d7e`.
|
||||||
|
Investigation established:
|
||||||
|
|
||||||
|
- **The server-side run completed cleanly.** The thread checkpoint in
|
||||||
|
`~/.evoscientist/sessions.db` holds the full 6818-char final answer
|
||||||
|
(`status: idle`, `error: None`). The text `O-8282d7e` does not appear at the
|
||||||
|
end of any stored content — it is a mid-stream artifact.
|
||||||
|
- **The disconnect is browser-side.** `langgraph dev` sends an SSE heartbeat
|
||||||
|
every 5 s (`langgraph_api/sse.py:102`), `send_timeout` is `None` (never kills
|
||||||
|
slow sends), and the server supports resume via the `last-event-id` header
|
||||||
|
(`langgraph_api/api/runs.py:623`). The `runs/stream` endpoint is **POST**, so
|
||||||
|
the browser uses `fetch()` to read it; native `EventSource` auto-reconnect does
|
||||||
|
not apply, and the `@evoscientist/webui` front-end does not re-POST with
|
||||||
|
`last-event-id` to resume.
|
||||||
|
- **`stream_resumable=True`** lets the server-side run finish after the browser
|
||||||
|
drops, which is why "server succeeded" ≠ "browser received".
|
||||||
|
- A separate symptom from the same session — `Port 6174 cannot be bound` — was a
|
||||||
|
leftover `langgraph dev` process (PID 15288) whose `atexit` cleanup never ran
|
||||||
|
(parent killed hard).
|
||||||
|
|
||||||
|
**Conclusion:** the SSE disconnect is structural to the WebUI path. The TUI path
|
||||||
|
does not use SSE at all — `tui_interactive.py:297` calls `create_runtime_gateways()`
|
||||||
|
with no `backend`, so it defaults to `backend="local"` (`gateway/runtime.py`),
|
||||||
|
which runs the compiled graph **in-process** and is immune.
|
||||||
|
|
||||||
|
## 2. Goals
|
||||||
|
|
||||||
|
1. **Make the in-process TUI the primary interface.** Long runs no longer touch
|
||||||
|
SSE, so the class of failure the user hit cannot occur.
|
||||||
|
2. **Guarantee the user always sees the complete answer.** Even if a TUI run is
|
||||||
|
interrupted (Ctrl-C, crash, terminal close), the next invocation can surface
|
||||||
|
the latest persisted assistant message and optionally resume.
|
||||||
|
3. **Keep the WebUI available as opt-in** (`--web`), with the launch path
|
||||||
|
hardened against orphaned processes and port conflicts.
|
||||||
|
4. **Eliminate the port-6174 / orphaned-`langgraph dev` failure mode** for the
|
||||||
|
default (TUI) path: it must not start `langgraph dev` at all.
|
||||||
|
|
||||||
|
## 3. Non-Goals
|
||||||
|
|
||||||
|
- Patching `langgraph_api` upstream.
|
||||||
|
- Replacing the WebUI front-end's fetch-SSE transport. (`@evoscientist/webui`
|
||||||
|
lives in a separate npm repo with its own release cycle; a client-side resume
|
||||||
|
fix is noted as future work, not in scope here.)
|
||||||
|
- Changing the graph, middleware, or checkpointer semantics.
|
||||||
|
- Touching the `EvoSci deploy` standalone-server command.
|
||||||
|
|
||||||
|
## 4. Architecture
|
||||||
|
|
||||||
|
Three launch modes, selected by a single source of truth:
|
||||||
|
|
||||||
|
| Mode | `ui_backend` | Runs graph via | Starts `langgraph dev`? |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **TUI (default)** | `tui` | `LocalGraphGateway` (in-process) | No |
|
||||||
|
| CLI | `cli` | `LocalGraphGateway` (in-process) | No |
|
||||||
|
| WebUI (opt-in) | `webui` | `langgraph dev` over HTTP/SSE | Yes |
|
||||||
|
|
||||||
|
A new `--web` flag on the top-level `evoscientist` command forces `webui` for
|
||||||
|
that invocation regardless of config; `--tui` forces `tui`. This lets users keep
|
||||||
|
`ui_backend = "webui"` in config but drop into the TUI without editing config,
|
||||||
|
and vice-versa.
|
||||||
|
|
||||||
|
### 4.1 Launch-mode resolution (single source of truth)
|
||||||
|
|
||||||
|
Today the default is inconsistent: `config/settings.py:279` defaults `ui_backend`
|
||||||
|
to `"tui"`, but `cli/tui_runtime.py:12` has `DEFAULT_UI_BACKEND = "cli"`. This
|
||||||
|
design collapses to one rule, evaluated in order:
|
||||||
|
|
||||||
|
1. `--web` flag → `webui`
|
||||||
|
2. `--tui` flag → `tui`
|
||||||
|
3. `config.ui_backend` (resolved via existing `resolve_ui_backend`)
|
||||||
|
4. `"tui"` (hardcoded final fallback)
|
||||||
|
|
||||||
|
`resolve_ui_backend` (`cli/tui_runtime.py:41`) stays the single normalizer; the
|
||||||
|
new flags feed into it rather than bypassing it.
|
||||||
|
|
||||||
|
### 4.2 Why TUI is immune (and what can still interrupt it)
|
||||||
|
|
||||||
|
In-process execution via `stream_agent_events` has no network hop. The only
|
||||||
|
"loss" modes are process-level:
|
||||||
|
|
||||||
|
- Ctrl-C mid-run
|
||||||
|
- Terminal close / SIGHUP
|
||||||
|
- Python crash / OOM
|
||||||
|
|
||||||
|
Because the checkpointer is `AsyncSqliteSaver` writing to
|
||||||
|
`~/.evoscientist/sessions.db` (`sessions.py:108` `get_db_path`), every completed
|
||||||
|
super-step is on disk before the next begins. An interrupted run therefore
|
||||||
|
leaves a recoverable checkpoint whose `next` tuple is non-empty (pending nodes)
|
||||||
|
and whose latest assistant message — if the model node finished — is the same
|
||||||
|
content the WebUI failed to deliver.
|
||||||
|
|
||||||
|
## 5. Components
|
||||||
|
|
||||||
|
### C1 — Launch-mode flags + TUI default
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- `EvoScientist/cli/__init__.py` (Typer app entry) — add `--web` / `--tui`
|
||||||
|
options on the root command; thread the resolved `ui_backend` into the existing
|
||||||
|
dispatch.
|
||||||
|
- `EvoScientist/cli/tui_runtime.py:12` — remove the divergent
|
||||||
|
`DEFAULT_UI_BACKEND = "cli"`; route through `resolve_ui_backend` so
|
||||||
|
`settings.py`'s `"tui"` default wins.
|
||||||
|
- `EvoScientist/cli/commands.py` — wherever `ui_backend` is currently read, accept
|
||||||
|
the flag override.
|
||||||
|
|
||||||
|
**Behavior:** Running `evoscientist` with no flags and default config launches
|
||||||
|
the TUI and starts **no** `langgraph dev` subprocess. No port is bound; the
|
||||||
|
orphan-process class of bug disappears for the default path.
|
||||||
|
|
||||||
|
**Acceptance:** `evoscientist` (no args, default config) does not open a
|
||||||
|
listening socket on 6174 or any other port.
|
||||||
|
|
||||||
|
### C2 — TUI run resilience (Approach B)
|
||||||
|
|
||||||
|
Two cooperating pieces, both backed by `sessions.db`.
|
||||||
|
|
||||||
|
#### C2a — Interrupt-fallback in the streaming loop
|
||||||
|
|
||||||
|
Wrap the TUI streaming consumer so that on any interruption
|
||||||
|
(`KeyboardInterrupt`, `asyncio.CancelledError`, unexpected `Exception`) it reads
|
||||||
|
the thread's latest checkpoint and renders the latest assistant message before
|
||||||
|
exiting.
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- `EvoScientist/stream/display.py` — `_run_streaming` and
|
||||||
|
`_astream_to_console` (the two streaming entry points used by
|
||||||
|
`cli/tui_backends.py`). Add a `finally`/except guard that, after an
|
||||||
|
interruption, calls a new `render_latest_assistant(thread_id)` helper.
|
||||||
|
- `EvoScientist/sessions.py` — add `get_latest_assistant_message(thread_id) ->
|
||||||
|
str | None`: read the most recent checkpoint for the thread, walk `messages`
|
||||||
|
backwards to the last `AIMessage` with non-empty content, return its text
|
||||||
|
(joined content blocks). Reuses existing `AsyncSqliteSaver` access already in
|
||||||
|
this module.
|
||||||
|
|
||||||
|
**Behavior on Ctrl-C mid-run:**
|
||||||
|
1. The guard catches the interrupt.
|
||||||
|
2. Prints a one-line notice: `Run interrupted — recovering latest output from
|
||||||
|
checkpoint…`.
|
||||||
|
3. Renders the latest persisted assistant message via the same Markdown renderer
|
||||||
|
the normal completion path uses.
|
||||||
|
4. If no assistant message is persisted yet (model node hadn't completed),
|
||||||
|
prints `Run interrupted before any output was saved. Resume with: evoscientist
|
||||||
|
--thread <id>` and exits.
|
||||||
|
|
||||||
|
**Non-goal:** this does not *continue* the run; it only guarantees visibility of
|
||||||
|
whatever completed. Continuation is C2b.
|
||||||
|
|
||||||
|
#### C2b — Startup detection of interrupted runs
|
||||||
|
|
||||||
|
Runs only when the TUI is launched to **resume a specific thread** (via the
|
||||||
|
existing thread-resume flag, i.e. the command `resume_hint.py` already prints at
|
||||||
|
exit). For a brand-new session there is nothing to detect and the prompt is
|
||||||
|
skipped. When resuming, check the thread's latest checkpoint: `next` is
|
||||||
|
non-empty **and** no run is currently active for that thread. If so, prompt:
|
||||||
|
|
||||||
|
```
|
||||||
|
You have an unfinished run in thread <id>:
|
||||||
|
[s] show what completed
|
||||||
|
[r] resume the run from its last checkpoint
|
||||||
|
[n] start fresh
|
||||||
|
```
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- `EvoScientist/sessions.py` — add `detect_interrupted_run(thread_id) ->
|
||||||
|
InterruptedRunInfo | None` returning the checkpoint id, the pending node names
|
||||||
|
(`next`), and the timestamp. Implemented as a thin read over the existing
|
||||||
|
checkpointer's `aget_tuple`.
|
||||||
|
- `EvoScientist/cli/tui_interactive.py` — near the existing thread-load path
|
||||||
|
(around the `create_runtime_gateways()` call at line 297), invoke the detector
|
||||||
|
for the active thread and render the prompt above.
|
||||||
|
|
||||||
|
**Resume semantics:** selecting `r` re-invokes the graph with the same
|
||||||
|
`thread_id`. LangGraph resumes from the last checkpoint automatically (this is
|
||||||
|
the same mechanism `evoscientist --thread <id>` already relies on for session
|
||||||
|
continuity). No new graph code needed.
|
||||||
|
|
||||||
|
**Acceptance:** After killing a TUI mid-run with Ctrl-C, relaunching with
|
||||||
|
`--thread <id>` offers `[s]`/`[r]`/`[n]`; `[s]` prints the persisted final
|
||||||
|
answer; `[r]` continues execution from where it stopped.
|
||||||
|
|
||||||
|
### C3 — WebUI mode hardening (`--web`)
|
||||||
|
|
||||||
|
Applies only when `ui_backend == "webui"`. The goal is to make the existing
|
||||||
|
`deploy/webui.py` launch path robust and to mitigate (not eliminate) the SSE
|
||||||
|
limitation.
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- `EvoScientist/deploy/webui.py` — three changes:
|
||||||
|
1. **Process-group + signal cleanup.** Today `atexit.register(stop_langgraph_dev,
|
||||||
|
...)` only fires on clean exit. Wrap the spawned `langgraph dev` (and the
|
||||||
|
`npx` child) in a process group (`start_new_session=True` on POSIX; on
|
||||||
|
Windows use a job object via the existing `_winloop.py` helpers) and
|
||||||
|
install `SIGINT`/`SIGTERM` handlers that tear the group down. This closes
|
||||||
|
the orphan-PID-15288 path for normal terminations. `SIGKILL` cannot run
|
||||||
|
handlers on either platform — document that hard kills may still orphan
|
||||||
|
the child, and the port-conflict UX in (2) is the user-facing recovery.
|
||||||
|
2. **Port-conflict UX.** The current "Port X is occupied by another process"
|
||||||
|
message is already correct; extend it to detect *whether the occupant is a
|
||||||
|
`langgraph dev`* (via the existing `/ok` health probe) and, if so, offer to
|
||||||
|
reuse it rather than bail. If the occupant is foreign, print the `lsof`
|
||||||
|
hint already present.
|
||||||
|
3. **Resume window.** Set `RESUMABLE_STREAM_TTL_SECONDS=600` in the env passed
|
||||||
|
to `start_langgraph_dev` so a browser that does reconnect can replay events
|
||||||
|
for runs up to 10 minutes long (the observed run was 127 s vs. the 120 s
|
||||||
|
default).
|
||||||
|
- `EvoScientist/deploy/server.py` — `start_langgraph_dev` already accepts env;
|
||||||
|
pass `RESUMABLE_STREAM_TTL_SECONDS` through.
|
||||||
|
|
||||||
|
**What C3 does NOT do:** it cannot make the browser auto-resume, because that
|
||||||
|
logic lives in `@evoscientist/webui`. It only widens the server-side window and
|
||||||
|
stops the orphan process. Documented in the WebUI exit message: *for long runs,
|
||||||
|
prefer the TUI (`evoscientist --tui`); the browser client does not resume a
|
||||||
|
dropped stream.*
|
||||||
|
|
||||||
|
### C4 — Documentation
|
||||||
|
|
||||||
|
- `README` / CLI `--help`: `--web` (browser), `--tui` (terminal, default,
|
||||||
|
recommended for long runs).
|
||||||
|
- A short "Why did my WebUI output stop?" note pointing at this design doc, so
|
||||||
|
future users reading the symptom can find the explanation.
|
||||||
|
|
||||||
|
## 6. Data Flow
|
||||||
|
|
||||||
|
### Default (TUI) path
|
||||||
|
```
|
||||||
|
evoscientist ─▶ resolve_ui_backend ─▶ "tui"
|
||||||
|
│
|
||||||
|
(no langgraph dev, no port bound)
|
||||||
|
▼
|
||||||
|
create_runtime_gateways(backend="local")
|
||||||
|
│
|
||||||
|
LocalGraphGateway.stream_events
|
||||||
|
│
|
||||||
|
stream_agent_events (in-process) ── checkpoint per super-step ──▶ sessions.db
|
||||||
|
│
|
||||||
|
on interrupt ──▶ get_latest_assistant_message(thread_id) ──▶ render
|
||||||
|
```
|
||||||
|
|
||||||
|
### WebUI (`--web`) path
|
||||||
|
unchanged from today (langgraph dev + `npx @evoscientist/webui`), plus C3's
|
||||||
|
process-group cleanup and `RESUMABLE_STREAM_TTL_SECONDS=600`.
|
||||||
|
|
||||||
|
## 7. Error Handling
|
||||||
|
|
||||||
|
| Failure | TUI default path | WebUI `--web` path |
|
||||||
|
|---|---|---|
|
||||||
|
| Ctrl-C mid-run | C2a renders latest persisted assistant message | Browser truncates; user runs `evoscientist --tui --thread <id>` → `[s]` to recover |
|
||||||
|
| Crash / SIGHUP | Checkpoint on disk; next `--thread <id>` shows `[s]`/`[r]` prompt (C2b) | Same recovery via TUI |
|
||||||
|
| Port 6174 bound | N/A — no port bound | C3 reuses if `langgraph dev`, else prints `lsof` hint |
|
||||||
|
| Orphaned `langgraph dev` | Cannot originate from TUI path | C3 process-group + signal handlers tear it down |
|
||||||
|
|
||||||
|
## 8. Testing
|
||||||
|
|
||||||
|
- **Unit** (`tests/test_sessions_*.py`): `get_latest_assistant_message` returns
|
||||||
|
the full final answer for a thread whose last run completed; returns `None`
|
||||||
|
for a thread interrupted before the model node.
|
||||||
|
- **Unit**: `detect_interrupted_run` returns pending `next` nodes for an
|
||||||
|
interrupted thread; `None` for an idle one.
|
||||||
|
- **Integration** (new `tests/test_tui_interrupt_recovery.py`): start a TUI run
|
||||||
|
against a graph whose model node sleeps, send `SIGINT` mid-run, assert the
|
||||||
|
process exits 0 after printing the recovered message.
|
||||||
|
- **Integration**: relaunch with `--thread <id>`, assert the `[s]/[r]/[n]`
|
||||||
|
prompt appears and `[r]` produces a completed run.
|
||||||
|
- **Launch** (`tests/test_cli_launch.py`, extend): `evoscientist` with default
|
||||||
|
config binds no port; `--web` binds 6174 and tears it down on `SIGTERM`.
|
||||||
|
- **WebUI** (manual / smoke): `--web`, kill parent with `SIGKILL`, assert no
|
||||||
|
orphan `langgraph dev` remains after the process-group change (best-effort —
|
||||||
|
`SIGKILL` cannot run handlers, but the process group lets the OS reap
|
||||||
|
children when the session ends; document this limit).
|
||||||
|
|
||||||
|
## 9. Rollout / Migration
|
||||||
|
|
||||||
|
- User's config currently has `ui_backend = "webui"`. After this change, that
|
||||||
|
config still selects WebUI (back-compatible). The user can run `evoscientist
|
||||||
|
--tui` immediately to get the resilient path, or `EvoSci config set ui_backend
|
||||||
|
tui` to make it permanent.
|
||||||
|
- No migration of `sessions.db` — schema is unchanged.
|
||||||
|
- No breaking change to `EvoSci deploy`.
|
||||||
|
|
||||||
|
## 10. Open Questions / Future Work
|
||||||
|
|
||||||
|
- **`@evoscientist/webui` client-side resume.** The front-end could re-POST
|
||||||
|
`/runs/stream` with `last-event-id` (or use the thread's SSE endpoint with
|
||||||
|
`since`) after a disconnect. This is the only way to fully fix the WebUI
|
||||||
|
symptom; it lives in the npm repo and is out of scope here. File a
|
||||||
|
cross-repo issue referencing this design.
|
||||||
|
- Whether `[r]` resume should replay the *interrupted* model call or only
|
||||||
|
continue from the *next* node. LangGraph's default (resume from pending
|
||||||
|
writes) is correct for most cases; revisit if users hit re-execution of
|
||||||
|
expensive tool calls.
|
||||||
|
- `cli` backend vs `tui` backend distinction — both are in-process; consider
|
||||||
|
deprecating the `cli`/`tui` split in a follow-up once the resilience work
|
||||||
|
lands, to reduce the three-way default confusion that caused the original
|
||||||
|
`DEFAULT_UI_BACKEND` divergence.
|
||||||
@@ -0,0 +1,196 @@
|
|||||||
|
# WebUI SSE Truncation Recovery via Checkpoint Fallback (α)
|
||||||
|
|
||||||
|
**Date:** 2026-07-06
|
||||||
|
**Status:** Design — pending implementation plan
|
||||||
|
**Scope:** `@evoscientist/webui` front-end (separate npm repo) + one optional convenience route in this Python repo
|
||||||
|
|
||||||
|
## 1. Background & Root Cause
|
||||||
|
|
||||||
|
A WebUI user saw a 127 s research run render truncated output, ending mid-token at
|
||||||
|
`O-8282d7e`. Investigation established:
|
||||||
|
|
||||||
|
- The server-side run **completed cleanly**; the thread checkpoint in
|
||||||
|
`~/.evoscientist/sessions.db` holds the full 6818-char final answer
|
||||||
|
(`status: idle`, `error: None`).
|
||||||
|
- The disconnect is **browser-side**: `langgraph dev` sends an SSE heartbeat every
|
||||||
|
5 s, `send_timeout` is `None`, and the server supports resume via
|
||||||
|
`last-event-id` (`langgraph_api/api/runs.py:623`). But `/runs/stream` is
|
||||||
|
**POST**, so the browser reads it via `fetch()`; native `EventSource`
|
||||||
|
auto-reconnect does not apply, and `@evoscientist/webui` does not re-POST to
|
||||||
|
resume. When the connection drops mid-run, the front-end is left showing the
|
||||||
|
last partial token and never fetches the completed answer.
|
||||||
|
|
||||||
|
**Core insight:** the server already has the complete answer in the checkpoint.
|
||||||
|
The fix is to make the front-end fall back to that checkpoint when the stream
|
||||||
|
ends abnormally. This is small, surgical, and directly addresses the symptom.
|
||||||
|
|
||||||
|
## 2. Goal
|
||||||
|
|
||||||
|
**The user always sees the complete final answer in the WebUI, even if the SSE
|
||||||
|
stream drops mid-run.**
|
||||||
|
|
||||||
|
## 3. Non-Goals
|
||||||
|
|
||||||
|
- Seamless stream resume (tracking `last-event-id` / `seq` and replaying
|
||||||
|
buffered events). That is option β — better UX, ~3–5× the code, deferred.
|
||||||
|
- The TUI pivot / launch-mode decoupling (option γ). Separable; only relevant if
|
||||||
|
the port-conflict / orphan-process issues need addressing.
|
||||||
|
- Patching `langgraph_api` upstream.
|
||||||
|
|
||||||
|
## 4. Detection Logic — What Counts as "Truncated"
|
||||||
|
|
||||||
|
The langgraph SSE protocol emits terminal events when a run finishes
|
||||||
|
(success → `end`, failure → `error`; exact names to be confirmed against the
|
||||||
|
`@langchain/langgraph-sdk` version during implementation). The front-end's
|
||||||
|
streaming reader loop currently processes events until the `fetch()` ReadableStream
|
||||||
|
closes.
|
||||||
|
|
||||||
|
**Truncation** = the stream closed (reader returned) **without** a terminal event
|
||||||
|
having been received. Causes: browser tab throttled, network blip, proxy idle
|
||||||
|
timeout. On truncation, the in-progress assistant bubble is left showing a prefix
|
||||||
|
of the real answer.
|
||||||
|
|
||||||
|
## 5. Components
|
||||||
|
|
||||||
|
### F1 — Truncation detector (front-end)
|
||||||
|
|
||||||
|
In the streaming reader loop, track a `terminalSeen` flag. Set it when a
|
||||||
|
terminal event arrives. When the reader returns, if `!terminalSeen`, mark the
|
||||||
|
run `truncated` and trigger F2.
|
||||||
|
|
||||||
|
**File:** `@evoscientist/webui` — the run-stream consumer (the module that wraps
|
||||||
|
the langgraph-ts SDK's `runs.stream` and renders into the message list).
|
||||||
|
|
||||||
|
### F2 — Checkpoint fallback fetch (front-end)
|
||||||
|
|
||||||
|
On `truncated`, fetch the complete final assistant message and **replace** the
|
||||||
|
in-progress bubble's content with it (not append — the partial tokens are a
|
||||||
|
prefix of the same message).
|
||||||
|
|
||||||
|
Two implementation choices, pick one:
|
||||||
|
|
||||||
|
- **F2a (zero server change):** call the standard langgraph endpoint
|
||||||
|
`GET /threads/{thread_id}/state`, walk `values.messages` backwards to the last
|
||||||
|
`AIMessage`, join its content blocks. No new route, but the front-end parses
|
||||||
|
raw state and reimplements "find latest assistant text" logic.
|
||||||
|
- **F2b (recommended, uses S1 below):** call `GET /api/threads/{id}/final-answer`
|
||||||
|
→ `{content, completed_at, complete}`. Server owns the parsing; front-end
|
||||||
|
stays dumb.
|
||||||
|
|
||||||
|
Render a subtle affordance (e.g., a dim "⚠ stream dropped — recovered from
|
||||||
|
checkpoint" line above the message) so the user knows a disconnect happened and
|
||||||
|
the displayed text is authoritative, not stale streaming.
|
||||||
|
|
||||||
|
### F3 — Retry until the run settles (front-end)
|
||||||
|
|
||||||
|
The browser may drop while the run is **still executing** server-side. A single
|
||||||
|
fallback fetch at drop-time could return a not-yet-final message. So F2 retries
|
||||||
|
with backoff until either:
|
||||||
|
|
||||||
|
- `complete == true` (run finished — render and stop), or
|
||||||
|
- a max wait elapses (default 240 s — comfortably exceeds the observed 127 s
|
||||||
|
run even if the drop happens at t=0), then render whatever is latest and
|
||||||
|
stop retrying.
|
||||||
|
|
||||||
|
Backoff: poll at 1 s, 2 s, 4 s, then every 5 s up to the max. Each poll replaces
|
||||||
|
the bubble with the latest content, so if the run finishes mid-retry the user
|
||||||
|
sees it update to the final form.
|
||||||
|
|
||||||
|
### S1 — Convenience route `GET /api/threads/{id}/final-answer` (this repo, optional but recommended)
|
||||||
|
|
||||||
|
Enables F2b. Mounted on the existing custom ASGI app already wired via
|
||||||
|
`langgraph.json` → `EvoScientist.langgraph_dev.http:app` (currently only serves
|
||||||
|
`/api/models`).
|
||||||
|
|
||||||
|
**Contract:**
|
||||||
|
|
||||||
|
```
|
||||||
|
GET /api/threads/{thread_id}/final-answer
|
||||||
|
200 → { "content": str, # last AIMessage text, blocks joined
|
||||||
|
"completed_at": iso8601, # run end timestamp, or null
|
||||||
|
"complete": bool } # true iff run reached terminal state
|
||||||
|
404 → thread not found
|
||||||
|
```
|
||||||
|
|
||||||
|
**`complete` derivation** (run reached a terminal state):
|
||||||
|
`thread.status == "idle"` **or** latest checkpoint's `next` tuple is empty.
|
||||||
|
|
||||||
|
**File:** `EvoScientist/langgraph_dev/http.py` — add `get_final_answer` route
|
||||||
|
beside the existing `get_models`. Implementation reads the thread state via the
|
||||||
|
shared checkpointer (`AsyncSqliteSaver` at `~/.evoscientist/sessions.db`,
|
||||||
|
already used by `sessions.py`). If direct checkpointer access is awkward from
|
||||||
|
inside the mounted app, fall back to an internal `langgraph_sdk.get_client()`
|
||||||
|
call to `GET /threads/{id}/state` on localhost — the contract stays identical.
|
||||||
|
|
||||||
|
**Why prefer F2b+S1 over F2a:** the "find latest AIMessage + join content blocks
|
||||||
|
+ decide complete?" logic is non-trivial (content is a list of typed blocks,
|
||||||
|
including `reasoning` blocks that should be excluded from the rendered answer —
|
||||||
|
see the checkpoint: block[0] was `reasoning`, block[1] was `text`). Centralizing
|
||||||
|
it server-side keeps the front-end thin and makes the same logic reusable by the
|
||||||
|
TUI's future recovery path if γ is ever pursued.
|
||||||
|
|
||||||
|
## 6. Data Flow
|
||||||
|
|
||||||
|
```
|
||||||
|
browser fetch(/runs/stream, POST) ──SSE──▶ langgraph dev
|
||||||
|
│ │
|
||||||
|
│ (mid-run, connection drops) │ run continues to completion
|
||||||
|
▼ │
|
||||||
|
reader returns, no terminal event ──▶ truncated │
|
||||||
|
│ │
|
||||||
|
│ GET /api/threads/{id}/final-answer │
|
||||||
|
▼ ▼
|
||||||
|
http.py::get_final_answer ──read checkpoint──▶ sessions.db
|
||||||
|
│
|
||||||
|
▼
|
||||||
|
{content, complete} ──▶ if !complete, retry (F3); else render (F2)
|
||||||
|
```
|
||||||
|
|
||||||
|
## 7. Error Handling
|
||||||
|
|
||||||
|
| Case | Behavior |
|
||||||
|
|---|---|
|
||||||
|
| Stream ends normally (`end`/`error` received) | F1 sets `terminalSeen`; no fallback; existing render path unchanged |
|
||||||
|
| Stream drops, run already finished server-side | First fallback fetch returns `complete=true`; render immediately |
|
||||||
|
| Stream drops, run still executing | F3 retries with backoff; each retry renders latest; stops when `complete` or 120 s max |
|
||||||
|
| Thread unknown / deleted | 404; front-end shows "stream interrupted and the session could not be recovered" |
|
||||||
|
| Convenience route unreachable (older server without S1) | Front-end falls back to F2a (raw state endpoint) — degrade gracefully |
|
||||||
|
|
||||||
|
## 8. Testing
|
||||||
|
|
||||||
|
**Front-end (`@evoscientist/webui`):**
|
||||||
|
- Unit: feed the reader a synthetic event stream that closes without a terminal
|
||||||
|
event → assert `truncated == true`; feed one with `end` → `false`.
|
||||||
|
- Integration: mock `fetch` to drop mid-stream; assert the fallback fetch fires
|
||||||
|
and the bubble's content is replaced with the checkpoint value.
|
||||||
|
|
||||||
|
**Server (this repo, if S1):** new `tests/test_http_final_answer.py`
|
||||||
|
- Completed thread → 200, `complete=true`, content matches the last AIMessage
|
||||||
|
text (not the reasoning block).
|
||||||
|
- Mid-run thread (forced `busy` / non-empty `next`) → 200, `complete=false`.
|
||||||
|
- Unknown thread → 404.
|
||||||
|
- Content with mixed `reasoning` + `text` blocks → only `text` in `content`.
|
||||||
|
|
||||||
|
## 9. Cross-Repo Coordination & Rollout
|
||||||
|
|
||||||
|
- **Order:** land S1 in this Python repo first (route + tests, behind no flag —
|
||||||
|
it's a pure addition). Then F1/F2/F3 in `@evoscientist/webui`.
|
||||||
|
- **Versioning:** the WebUI launches via `npx @evoscientist/webui@latest`, so
|
||||||
|
the front-end fix reaches users on their next launch automatically once
|
||||||
|
published. No coordinated upgrade required on the Python side.
|
||||||
|
- **Backward compatibility:** if a user runs a new front-end against an older
|
||||||
|
Python server without S1, the front-end must detect the 404 and fall back to
|
||||||
|
F2a (raw state endpoint). If an old front-end runs against a new server, S1
|
||||||
|
is simply unused — no harm.
|
||||||
|
|
||||||
|
## 10. Open Questions / Future Work
|
||||||
|
|
||||||
|
- **Exact terminal-event names** for the SDK version in use (`end`/`error` vs.
|
||||||
|
`done`/`result`). Confirm against `@langchain/langgraph-sdk` during
|
||||||
|
implementation; the detector is parameterized on this.
|
||||||
|
- **True resume (β):** track `last-event-id` (or the V2 thread-stream `seq`) and
|
||||||
|
replay buffered events on reconnect — seamless, no visible "recovered" flash.
|
||||||
|
Larger front-end effort; defer until α is validated.
|
||||||
|
- **TUI pivot (γ):** separable work to make the in-process TUI the default and
|
||||||
|
eliminate the port-conflict / orphan-process issues. Independent spec if
|
||||||
|
pursued.
|
||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
[project]
|
[project]
|
||||||
name = "EvoScientist"
|
name = "EvoScientist"
|
||||||
version = "0.2.1"
|
version = "0.2.2"
|
||||||
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
|
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
requires-python = ">=3.11"
|
requires-python = ">=3.11"
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
from unittest.mock import patch
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from langchain_core.messages import AIMessage, HumanMessage
|
||||||
from starlette.testclient import TestClient
|
from starlette.testclient import TestClient
|
||||||
|
|
||||||
from EvoScientist.config import EvoScientistConfig
|
from EvoScientist.config import EvoScientistConfig
|
||||||
@@ -161,3 +162,143 @@ def test_ollama_discovery_skipped_when_base_url_absent():
|
|||||||
{"name": n, "model_id": m, "provider": p}
|
{"name": n, "model_id": m, "provider": p}
|
||||||
for n, m, p in list_models_by_provider()
|
for n, m, p in list_models_by_provider()
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_final_answer_extracts_latest_ai_text_blocks():
|
||||||
|
async def fake_metadata(_thread_id):
|
||||||
|
return {"updated_at": "2026-07-06T14:14:53+00:00"}
|
||||||
|
|
||||||
|
async def fake_messages(_thread_id):
|
||||||
|
return [
|
||||||
|
HumanMessage(content="question"),
|
||||||
|
AIMessage(content="old answer"),
|
||||||
|
AIMessage(
|
||||||
|
content=[
|
||||||
|
{"type": "reasoning", "text": "internal"},
|
||||||
|
{"type": "text", "text": "Part A"},
|
||||||
|
{"type": "tool_use", "name": "search"},
|
||||||
|
{"type": "output_text", "text": "Part B"},
|
||||||
|
]
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
async def fake_runtime(_request, _thread_id):
|
||||||
|
return {
|
||||||
|
"found": True,
|
||||||
|
"complete": True,
|
||||||
|
"completed_at": "2026-07-06T14:15:00+00:00",
|
||||||
|
}
|
||||||
|
|
||||||
|
with (
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||||
|
new=fake_metadata,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
|
||||||
|
new=fake_messages,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
|
||||||
|
new=fake_runtime,
|
||||||
|
),
|
||||||
|
):
|
||||||
|
resp = client.get("/api/threads/thread-1/final-answer")
|
||||||
|
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert resp.json() == {
|
||||||
|
"content": "Part A\n\nPart B",
|
||||||
|
"completed_at": "2026-07-06T14:15:00+00:00",
|
||||||
|
"complete": True,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_final_answer_skips_tool_selection_json_text():
|
||||||
|
async def fake_metadata(_thread_id):
|
||||||
|
return {"updated_at": "2026-07-06T14:14:53+00:00"}
|
||||||
|
|
||||||
|
async def fake_messages(_thread_id):
|
||||||
|
return [
|
||||||
|
HumanMessage(content="question"),
|
||||||
|
AIMessage(content="stable answer"),
|
||||||
|
AIMessage(
|
||||||
|
content=(
|
||||||
|
'{"tools":["search_papers","get_abstract"]}'
|
||||||
|
'{"tools":["web_search_exa"]}'
|
||||||
|
)
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
async def fake_runtime(_request, _thread_id):
|
||||||
|
return {
|
||||||
|
"found": True,
|
||||||
|
"complete": True,
|
||||||
|
"completed_at": "2026-07-06T14:15:00+00:00",
|
||||||
|
}
|
||||||
|
|
||||||
|
with (
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||||
|
new=fake_metadata,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
|
||||||
|
new=fake_messages,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
|
||||||
|
new=fake_runtime,
|
||||||
|
),
|
||||||
|
):
|
||||||
|
resp = client.get("/api/threads/thread-1/final-answer")
|
||||||
|
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert resp.json()["content"] == "stable answer"
|
||||||
|
|
||||||
|
|
||||||
|
def test_final_answer_returns_404_for_unknown_thread():
|
||||||
|
async def fake_metadata(_thread_id):
|
||||||
|
return None
|
||||||
|
|
||||||
|
with patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||||
|
new=fake_metadata,
|
||||||
|
):
|
||||||
|
resp = client.get("/api/threads/missing/final-answer")
|
||||||
|
|
||||||
|
assert resp.status_code == 404
|
||||||
|
assert resp.json() == {"error": "thread not found"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_final_answer_does_not_mark_complete_when_runtime_state_fails():
|
||||||
|
async def fake_metadata(_thread_id):
|
||||||
|
return {"updated_at": "2026-07-06T14:14:53+00:00"}
|
||||||
|
|
||||||
|
async def fake_messages(_thread_id):
|
||||||
|
return [AIMessage(content="checkpoint answer")]
|
||||||
|
|
||||||
|
async def fake_runtime(_request, _thread_id):
|
||||||
|
raise RuntimeError("langgraph runtime unavailable")
|
||||||
|
|
||||||
|
with (
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_metadata_for_http",
|
||||||
|
new=fake_metadata,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._get_thread_messages_for_http",
|
||||||
|
new=fake_messages,
|
||||||
|
),
|
||||||
|
patch(
|
||||||
|
"EvoScientist.langgraph_dev.http._read_thread_runtime_state",
|
||||||
|
new=fake_runtime,
|
||||||
|
),
|
||||||
|
):
|
||||||
|
resp = client.get("/api/threads/thread-1/final-answer")
|
||||||
|
|
||||||
|
assert resp.status_code == 200
|
||||||
|
assert resp.json() == {
|
||||||
|
"content": "checkpoint answer",
|
||||||
|
"completed_at": None,
|
||||||
|
"complete": False,
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from EvoScientist.deploy.webui import _resolve_webui_package
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_webui_package_defaults_to_published_package(monkeypatch):
|
||||||
|
monkeypatch.delenv("EVOSCIENTIST_WEBUI_PACKAGE", raising=False)
|
||||||
|
|
||||||
|
assert _resolve_webui_package() == "@evoscientist/webui@latest"
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_webui_package_allows_local_override(monkeypatch):
|
||||||
|
monkeypatch.setenv("EVOSCIENTIST_WEBUI_PACKAGE", "/tmp/EvoScientist-WebUI")
|
||||||
|
|
||||||
|
assert _resolve_webui_package() == "/tmp/EvoScientist-WebUI"
|
||||||
@@ -944,7 +944,7 @@ wheels = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "evoscientist"
|
name = "evoscientist"
|
||||||
version = "0.2.0"
|
version = "0.2.2"
|
||||||
source = { editable = "." }
|
source = { editable = "." }
|
||||||
dependencies = [
|
dependencies = [
|
||||||
{ name = "deepagents", extra = ["quickjs"] },
|
{ name = "deepagents", extra = ["quickjs"] },
|
||||||
|
|||||||
Reference in New Issue
Block a user