From 5aa835361348769237068220d95e5d0fd0728435 Mon Sep 17 00:00:00 2001 From: Yinhan Lu <143548426+Yinhan-Lu@users.noreply.github.com> Date: Fri, 3 Apr 2026 09:35:14 -0400 Subject: [PATCH] feat(llm): upgrade OpenAI reasoning effort from high to xhigh (#136) * feat(llm): upgrade OpenAI reasoning effort from high to xhigh The OpenAI Responses API supports "xhigh" as a reasoning effort level, which provides deeper reasoning than "high". This is already used by other CLI tools (e.g., OpenClaw) for OpenAI models. Only affects the direct API key path; the ccproxy/OAuth path is unchanged (reasoning is still skipped there). Co-Authored-By: Claude Opus 4.6 (1M context) * fix(llm): limit xhigh reasoning to gpt-5.4+ and codex models Only gpt-5.4 series and codex models support xhigh reasoning effort. Older models (gpt-5, gpt-5.1, gpt-5.2, gpt-5.3) fall back to high. Co-Authored-By: Claude Opus 4.6 (1M context) --------- Co-authored-by: Claude Opus 4.6 (1M context) Co-authored-by: Xi Zhang Co-authored-by: Xi Zhang <106144707+X-iZhang@users.noreply.github.com> --- EvoScientist/llm/models.py | 3 ++- tests/test_llm.py | 33 +++++++++++++++++++++++++++++---- 2 files changed, 31 insertions(+), 5 deletions(-) diff --git a/EvoScientist/llm/models.py b/EvoScientist/llm/models.py index 3efb6f5..b48dcf2 100644 --- a/EvoScientist/llm/models.py +++ b/EvoScientist/llm/models.py @@ -215,7 +215,8 @@ def _apply_auto_config( # ccproxy uses Chat Completions which doesn't support reasoning. pass else: - kwargs["reasoning"] = {"effort": "high", "summary": "auto"} + _eff = "xhigh" if ("5.4" in model_id or "codex" in model_id) else "high" + kwargs["reasoning"] = {"effort": _eff, "summary": "auto"} # Google GenAI: surface thinking traces if provider == "google-genai": diff --git a/tests/test_llm.py b/tests/test_llm.py index 3d535e2..37c91b6 100644 --- a/tests/test_llm.py +++ b/tests/test_llm.py @@ -771,15 +771,40 @@ class TestAutoConfig: assert call_kwargs["thinking"] == custom_thinking @patch("EvoScientist.llm.models.init_chat_model") - def test_openai_reasoning(self, mock_init, monkeypatch): - """Native OpenAI models get auto-reasoning.""" + def test_openai_reasoning_xhigh(self, mock_init, monkeypatch): + """gpt-5.4+ and codex models get xhigh reasoning.""" + mock_init.return_value = "mock_model" + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + + get_chat_model("gpt-5.4", provider="openai") + assert mock_init.call_args[1]["reasoning"] == { + "effort": "xhigh", + "summary": "auto", + } + + get_chat_model("gpt-5.3-codex", provider="openai") + assert mock_init.call_args[1]["reasoning"] == { + "effort": "xhigh", + "summary": "auto", + } + + @patch("EvoScientist.llm.models.init_chat_model") + def test_openai_reasoning_high_fallback(self, mock_init, monkeypatch): + """Other OpenAI models get high reasoning effort.""" mock_init.return_value = "mock_model" monkeypatch.delenv("OPENAI_BASE_URL", raising=False) get_chat_model("gpt-5-nano") + assert mock_init.call_args[1]["reasoning"] == { + "effort": "high", + "summary": "auto", + } - call_kwargs = mock_init.call_args[1] - assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"} + get_chat_model("gpt-5.2", provider="openai") + assert mock_init.call_args[1]["reasoning"] == { + "effort": "high", + "summary": "auto", + } @patch("EvoScientist.llm.models.init_chat_model") def test_openai_base_url_override(self, mock_init, monkeypatch):