diff --git a/.github/assets/asta_bench_data.png b/.github/assets/asta_bench_data.png new file mode 100644 index 0000000..06cc39b Binary files /dev/null and b/.github/assets/asta_bench_data.png differ diff --git a/.github/assets/badge-pypi-dark.svg b/.github/assets/badge-pypi-dark.svg index cb4d54b..27ecbeb 100644 --- a/.github/assets/badge-pypi-dark.svg +++ b/.github/assets/badge-pypi-dark.svg @@ -5,5 +5,5 @@ v0.0.4 + font-size="13" font-weight="700" fill="#ffffff">v0.0.5 \ No newline at end of file diff --git a/.github/assets/badge-pypi-light.svg b/.github/assets/badge-pypi-light.svg index 05252c8..b7b70bf 100644 --- a/.github/assets/badge-pypi-light.svg +++ b/.github/assets/badge-pypi-light.svg @@ -5,5 +5,5 @@ v0.0.4 + font-size="13" font-weight="700" fill="#ffffff">v0.0.5 \ No newline at end of file diff --git a/EvoScientist/cli/interactive.py b/EvoScientist/cli/interactive.py index c42bcb1..ba6ccb5 100644 --- a/EvoScientist/cli/interactive.py +++ b/EvoScientist/cli/interactive.py @@ -120,6 +120,8 @@ def print_banner( info.append(" \u2022 Type ", style="#ffe082") info.append("/", style="#ffe082 bold") info.append(" for commands", style="#ffe082") + info.append(" \u2022 ", style="#ffe082") + info.append("@ files", style="#ffe082 bold") info.append(" \u2022 Ctrl+C ", style="#ffe082") info.append("interrupt", style="#ffe082 bold") console.print(info) diff --git a/EvoScientist/cli/tui_interactive.py b/EvoScientist/cli/tui_interactive.py index ca0fcee..fc61077 100644 --- a/EvoScientist/cli/tui_interactive.py +++ b/EvoScientist/cli/tui_interactive.py @@ -115,6 +115,8 @@ def _build_welcome_banner( info.append(" \u2022 Type ", style="#ffe082") info.append("/", style="#ffe082 bold") info.append(" for commands", style="#ffe082") + info.append(" \u2022 ", style="#ffe082") + info.append("@ files", style="#ffe082 bold") info.append(" \u2022 Ctrl+C ", style="#ffe082") info.append("interrupt", style="#ffe082 bold") banner.append_text(info) diff --git a/EvoScientist/config/onboard.py b/EvoScientist/config/onboard.py index 3cbae2c..06d8aec 100644 --- a/EvoScientist/config/onboard.py +++ b/EvoScientist/config/onboard.py @@ -337,7 +337,7 @@ def validate_siliconflow_key(api_key: str) -> tuple[bool, str]: def validate_openrouter_key(api_key: str) -> tuple[bool, str]: - """Validate an OpenRouter API key by making a test request. + """Validate an OpenRouter API key via the authenticated /auth/key endpoint. Returns: Tuple of (is_valid, message). @@ -346,20 +346,17 @@ def validate_openrouter_key(api_key: str) -> tuple[bool, str]: return True, "Skipped (no key provided)" try: - import openai + import httpx - client = openai.OpenAI(api_key=api_key, base_url="https://openrouter.ai/api/v1") - client.models.list() - return True, "Valid" + resp = httpx.get( + "https://openrouter.ai/api/v1/auth/key", + headers={"Authorization": f"Bearer {api_key}"}, + timeout=10, + ) + if resp.status_code == 200: + return True, "Valid" + return False, "Invalid API key" except Exception as e: - error_str = str(e).lower() - if ( - "401" in error_str - or "unauthorized" in error_str - or "invalid" in error_str - or "authentication" in error_str - ): - return False, "Invalid API key" return False, f"Error: {e}" @@ -1717,9 +1714,14 @@ def _step_tinytex() -> None: when none is found. The agent can auto-install missing LaTeX packages at runtime via ``tlmgr``, so only the base TinyTeX is needed here. """ - prepare = questionary.confirm( - "Prepare LaTeX environment? (needed to compile .tex → .pdf)", - default=True, + latex_choices = [ + Choice(title="No need (skip LaTeX setup)", value=False), + Choice(title="Install now (TinyTeX compiler)", value=True), + ] + prepare = questionary.select( + "LaTeX environment (needed to compile .tex → .pdf):", + choices=latex_choices, + default=False, style=WIZARD_STYLE, qmark=QMARK, ).ask() diff --git a/README.md b/README.md index dc62854..c14bce2 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ - PyPI v0.0.4 + PyPI v0.0.5 @@ -47,14 +47,14 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human + + +
- ICAIS 2025 Awards + AstaBench Data Analysis #1
- Best Paper & Appraisal Award + #1 on AstaBench Data Analysis
- Best Paper + AstaBench Code & Execution #1
- AI-Generated Best Paper + #1 on AstaBench Code & Execution
DeepResearch Bench II #1 @@ -62,9 +62,16 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human #1 on DeepResearch Bench II - AstaBench Code & Execution #1 + ICAIS 2025 Awards
- #1 on AstaBench Code & Execution + Best Paper & Appraisal Award +
+ Best Paper +
+ AI-Generated Best Paper
@@ -102,7 +109,8 @@ Moving beyond traditional human-in-the-loop systems, EvoScientist adopts a human > Looking for ready-to-use research skills? Check out [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) — powered by [**EvoScientist**](https://github.com/EvoScientist/EvoScientist)'s engine and installable skills, the entire end-to-end research lifecycle is covered out of the box. [**EvoSkills**](https://github.com/EvoScientist/EvoSkills) are also compatible with other CLI coding agents. ## 🔥 News -- **[25 Mar 2026]** 🥇 Ranked #1 on [AstaBench Code & Execution](https://huggingface.co/spaces/allenai/asta-bench-leaderboard) at submission time! [**Leaderboard**](https://allenai-asta-bench-leaderboard.hf.space/code-execution) 👈 +- **[26 Jun 2026]** 🥇 Ranked #1 on [AstaBench Data Analysis](https://allenai-asta-bench-leaderboard.hf.space/home) at submission time! [**Leaderboard**](https://allenai-asta-bench-leaderboard.hf.space/data-analysis) 👈 +- **[25 Mar 2026]** 🥇 Ranked #1 on [AstaBench Code & Execution](https://allenai-asta-bench-leaderboard.hf.space/home) at submission time! [**Leaderboard**](https://allenai-asta-bench-leaderboard.hf.space/code-execution) 👈 - **[13 Mar 2026]** 🚀 [**EvoScientist**](https://github.com/EvoScientist/EvoScientist) officially debuts! - **[11 Mar 2026]** ⛳ Technical Report is live! [**Check it out**](https://arxiv.org/abs/2603.08127) 👈 - **[06 Mar 2026]** 🥇 Ranked #1 on [DeepResearch Bench II](https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html#leaderboard) at submission time! [**Leaderboard**](https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html#leaderboard) 👈 diff --git a/README.zh-CN.md b/README.zh-CN.md index 5dcfa74..b4d8ed8 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -15,7 +15,7 @@
- PyPI v0.0.4 + PyPI v0.0.5 @@ -54,14 +54,14 @@ EvoScientist 超越了传统的人在回路(Human-in-the-Loop)模式,采 + + +
- ICAIS 2025 Awards + AstaBench Data Analysis #1
- Best Paper & Appraisal Award + AstaBench 数据分析榜 第一名
- Best Paper + AstaBench Code & Execution #1
- AI-Generated Best Paper + AstaBench 代码与执行榜 第一名
DeepResearch Bench II #1 @@ -69,9 +69,16 @@ EvoScientist 超越了传统的人在回路(Human-in-the-Loop)模式,采 DeepResearch Bench II 第一名 - AstaBench Code & Execution #1 + ICAIS 2025 Awards
- AstaBench 代码与执行榜 第一名 + Best Paper & Appraisal Award +
+ Best Paper +
+ AI-Generated Best Paper
@@ -111,7 +118,8 @@ EvoScientist 超越了传统的人在回路(Human-in-the-Loop)模式,采 ## 🔥 动态 -- **[2026 年 3 月 25 日]** 🥇 提交时在 [AstaBench 代码与执行](https://huggingface.co/spaces/allenai/asta-bench-leaderboard) 排名第一![**排行榜**](https://allenai-asta-bench-leaderboard.hf.space/code-execution) 👈 +- **[2026 年 6 月 26 日]** 🥇 提交时在 [AstaBench 数据分析](https://allenai-asta-bench-leaderboard.hf.space/home) 排名第一![**排行榜**](https://allenai-asta-bench-leaderboard.hf.space/data-analysis) 👈 +- **[2026 年 3 月 25 日]** 🥇 提交时在 [AstaBench 代码与执行](https://allenai-asta-bench-leaderboard.hf.space/home) 排名第一![**排行榜**](https://allenai-asta-bench-leaderboard.hf.space/code-execution) 👈 - **[2026 年 3 月 13 日]** 🚀 [**EvoScientist**](https://github.com/EvoScientist/EvoScientist) 正式亮相! - **[2026 年 3 月 11 日]** ⛳ 技术报告已上线![**查看详情**](https://arxiv.org/abs/2603.08127) 👈 - **[2026 年 3 月 6 日]** 🥇 提交时在 [DeepResearch Bench II](https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html#leaderboard) 排名第一![**排行榜**](https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html#leaderboard) 👈 diff --git a/pyproject.toml b/pyproject.toml index 4746270..9316da5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "EvoScientist" -version = "0.0.4" +version = "0.0.5" description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery" readme = "README.md" requires-python = ">=3.11" diff --git a/tests/test_onboard.py b/tests/test_onboard.py index e1c5317..4aad144 100644 --- a/tests/test_onboard.py +++ b/tests/test_onboard.py @@ -1248,7 +1248,7 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_step_skipped") as mock_ps, patch("EvoScientist.config.onboard.console"), ): - mock_q.confirm.return_value.ask.return_value = False + mock_q.select.return_value.ask.return_value = False _step_tinytex() mock_ps.assert_called_once_with("LaTeX", "skipped") @@ -1269,7 +1269,7 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_latex_status") as mock_status, patch("EvoScientist.config.onboard.console"), ): - mock_q.confirm.return_value.ask.return_value = True + mock_q.select.return_value.ask.return_value = True _step_tinytex() mock_status.assert_called_once() @@ -1291,7 +1291,7 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._auto_install_latexmk") as mock_auto, patch("EvoScientist.config.onboard.console"), ): - mock_q.confirm.return_value.ask.return_value = True + mock_q.select.return_value.ask.return_value = True _step_tinytex() mock_auto.assert_called_once() @@ -1319,6 +1319,7 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_latex_status"), patch("EvoScientist.config.onboard.console"), ): + mock_q.select.return_value.ask.return_value = True mock_q.confirm.return_value.ask.return_value = True _step_tinytex() mock_pr.assert_called_once_with("LaTeX", "TinyTeX installed") @@ -1341,8 +1342,8 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_step_skipped") as mock_ps, patch("EvoScientist.config.onboard.console"), ): - # First confirm (prepare) = True, second confirm (install) = False - mock_q.confirm.return_value.ask.side_effect = [True, False] + mock_q.select.return_value.ask.return_value = True + mock_q.confirm.return_value.ask.return_value = False _step_tinytex() mock_ps.assert_called_once_with("LaTeX", "skipped") @@ -1368,6 +1369,7 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_step_result") as mock_pr, patch("EvoScientist.config.onboard.console"), ): + mock_q.select.return_value.ask.return_value = True mock_q.confirm.return_value.ask.return_value = True _step_tinytex() mock_pr.assert_called_once_with( @@ -1396,6 +1398,7 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_step_result") as mock_pr, patch("EvoScientist.config.onboard.console") as mock_con, ): + mock_q.select.return_value.ask.return_value = True mock_q.confirm.return_value.ask.return_value = True _step_tinytex() path_warning_printed = any( @@ -1424,7 +1427,6 @@ class TestStepTinytex: patch("EvoScientist.config.onboard._print_step_skipped") as mock_ps, patch("EvoScientist.config.onboard.console"), ): - # Only one confirm call (prepare=True), no second confirm for manual - mock_q.confirm.return_value.ask.return_value = True + mock_q.select.return_value.ask.return_value = True _step_tinytex() mock_ps.assert_called_once_with("LaTeX", "manual install needed")