fix(llm): bound routed reasoning and surface empty truncation (#425)

* fix(llm): bound routed reasoning and surface empty truncation

Default DashScope Qwen 3.8 Max requests to low reasoning effort and forward explicit reasoning controls to custom OpenAI-compatible endpoints.

Detect length-limited responses that exhaust their budget during reasoning without producing content or tool calls, and surface a provider-aware error instead of ending the turn silently.

Add regression coverage for routed reasoning configuration, truncated empty responses, and valid content/tool-call responses.

* fix(middleware): reject empty structured text blocks

* fix(llm): address reasoning truncation review feedback

* fix(llm): validate DashScope reasoning effort

* fix(llm): document DashScope reasoning support
This commit is contained in:
LinxXu
2026-08-18 15:28:19 +08:00
committed by GitHub
parent 1b324906fd
commit 47c8da3e5b
6 changed files with 513 additions and 6 deletions
@@ -14,10 +14,13 @@ import dataclasses
from types import SimpleNamespace
import pytest
from langchain.agents.middleware.types import ModelResponse
from langchain_core.messages import AIMessage
from EvoScientist.llm.errors import ProviderStreamError
from EvoScientist.middleware.error_normalization import (
ErrorNormalizationMiddleware,
ModelOutputTruncatedError,
_normalize,
)
@@ -421,3 +424,116 @@ class TestMiddleware:
req = _request(_openrouter_model())
mw = ErrorNormalizationMiddleware()
assert self._run_awrap(mw, req, handler) == "ok"
def test_empty_length_response_becomes_visible_provider_error(self):
"""Reasoning-only truncation must not look like a successful idle turn."""
response = ModelResponse(
result=[
AIMessage(
content="",
additional_kwargs={"reasoning_content": "still thinking"},
response_metadata={"finish_reason": "length"},
)
]
)
def handler(_req):
return response
req = _request(_openai_model(base_url="https://internal.corp/v1"))
with pytest.raises(ProviderStreamError) as excinfo:
ErrorNormalizationMiddleware().wrap_model_call(req, handler)
assert excinfo.value.provider == "openai_compat"
assert isinstance(excinfo.value.__cause__, ModelOutputTruncatedError)
assert "reasoning_effort" in str(excinfo.value)
def test_empty_incomplete_responses_api_result_is_detected(self):
response = ModelResponse(
result=[
AIMessage(
content=[],
response_metadata={
"status": "incomplete",
"incomplete_details": {"reason": "max_output_tokens"},
},
)
]
)
async def handler(_req):
return response
req = _request(_openai_model())
with pytest.raises(ProviderStreamError):
self._run_awrap(ErrorNormalizationMiddleware(), req, handler)
def test_empty_structured_text_block_with_length_is_detected(self):
response = ModelResponse(
result=[
AIMessage(
content=[{"type": "text", "text": " "}],
response_metadata={"finish_reason": "length"},
)
]
)
def handler(_req):
return response
with pytest.raises(ProviderStreamError) as excinfo:
ErrorNormalizationMiddleware().wrap_model_call(
_request(_openai_model()), handler
)
assert isinstance(excinfo.value.__cause__, ModelOutputTruncatedError)
def test_redacted_thinking_only_with_max_tokens_is_detected(self):
response = ModelResponse(
result=[
AIMessage(
content=[{"type": "redacted_thinking", "data": "opaque-payload"}],
response_metadata={"stop_reason": "max_tokens"},
)
]
)
def handler(_req):
return response
with pytest.raises(ProviderStreamError) as excinfo:
ErrorNormalizationMiddleware().wrap_model_call(
_request(_anthropic_model()), handler
)
assert isinstance(excinfo.value.__cause__, ModelOutputTruncatedError)
@pytest.mark.parametrize(
"message",
[
AIMessage(content="answer", response_metadata={"finish_reason": "length"}),
AIMessage(
content="",
tool_calls=[{"name": "search", "args": {}, "id": "call-1"}],
response_metadata={"finish_reason": "length"},
),
AIMessage(content="", response_metadata={"finish_reason": "stop"}),
AIMessage(
content=[
{"type": "redacted_thinking", "data": "opaque-payload"},
{"type": "text", "text": "answer"},
],
response_metadata={"stop_reason": "max_tokens"},
),
],
)
def test_nonempty_tool_and_normal_stop_responses_are_not_rejected(self, message):
response = ModelResponse(result=[message])
def handler(_req):
return response
result = ErrorNormalizationMiddleware().wrap_model_call(
_request(_openai_model()), handler
)
assert result is response
+112
View File
@@ -1077,6 +1077,34 @@ class TestThirdPartyRouting:
assert call_kwargs["base_url"] == "https://my-llm.example.com/v1"
assert call_kwargs["api_key"] == "custom-key-789"
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_openai_forwards_explicit_reasoning_effort(
self, mock_init, monkeypatch
):
"""User-owned compatible endpoints receive an explicit effort only."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://opencode.example/v1")
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "custom-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "low")
get_chat_model("reasoning-model", provider="custom-openai")
assert mock_init.call_args[1]["reasoning_effort"] == "low"
@patch("EvoScientist.llm.models.init_chat_model")
def test_custom_openai_omits_unconfigured_reasoning_effort(
self, mock_init, monkeypatch
):
"""Unknown compatible endpoints stay compatible by default."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("CUSTOM_OPENAI_BASE_URL", "https://plain.example/v1")
monkeypatch.setenv("CUSTOM_OPENAI_API_KEY", "custom-key")
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
get_chat_model("plain-model", provider="custom-openai")
assert "reasoning_effort" not in mock_init.call_args[1]
@patch("EvoScientist.llm.models.init_chat_model")
def test_anthropic_base_url_override(self, mock_init, monkeypatch):
"""Anthropic provider should support base_url override (e.g. ccproxy)."""
@@ -1167,6 +1195,90 @@ class TestThirdPartyRouting:
== "https://dashscope.aliyuncs.com/compatible-mode/v1"
)
assert call_kwargs["api_key"] == "ds-key-456"
assert "reasoning_effort" not in call_kwargs
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_uses_bounded_default(self, mock_init, monkeypatch):
"""Qwen 3.8 avoids the regular endpoint's xhigh default."""
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
get_chat_model("qwen3.8-max", provider="dashscope")
assert mock_init.call_args[1]["reasoning_effort"] == "medium"
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_respects_configured_reasoning_effort(
self, mock_init, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "medium")
get_chat_model("qwen3.8-max", provider="dashscope")
assert mock_init.call_args[1]["reasoning_effort"] == "medium"
@pytest.mark.parametrize(
"effort",
[
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max",
],
)
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_accepts_supported_reasoning_effort(
self, mock_init, effort, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", effort)
get_chat_model("qwen3.8-max", provider="dashscope")
assert mock_init.call_args[1]["reasoning_effort"] == effort
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_rejects_unsupported_reasoning_effort(
self, mock_init, monkeypatch
):
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "invalid")
with pytest.raises(ValueError, match="dashscope"):
get_chat_model("qwen3.8-max", provider="dashscope")
mock_init.assert_not_called()
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_explicit_effort_overrides_invalid_environment(
self, mock_init, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "ds-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "invalid")
get_chat_model("qwen3.8-max", provider="dashscope", reasoning_effort="low")
assert mock_init.call_args[1]["reasoning_effort"] == "low"
@patch("EvoScientist.llm.models.init_chat_model")
def test_qwen38_dashscope_code_omits_undocumented_reasoning_effort(
self, mock_init, monkeypatch
):
mock_init.return_value = "mock_model"
monkeypatch.setenv("DASHSCOPE_API_KEY", "sk-sp-key")
monkeypatch.setenv("EVOSCIENTIST_REASONING_EFFORT", "medium")
get_chat_model("qwen3.8-max", provider="dashscope-code")
assert "reasoning_effort" not in mock_init.call_args[1]
@patch("EvoScientist.llm.models.init_chat_model")
def test_dashscope_code_routes_through_openai(self, mock_init, monkeypatch):
+105
View File
@@ -10,6 +10,7 @@ from types import SimpleNamespace
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from langchain.agents.middleware.types import ModelResponse
from langchain_core.exceptions import ContextOverflowError
from langchain_core.messages import AIMessage, HumanMessage
@@ -42,6 +43,22 @@ def _fake_request():
AI_RESPONSE = AIMessage(content="ok")
def _truncated_response() -> ModelResponse:
return ModelResponse(
result=[
AIMessage(
content="",
additional_kwargs={"reasoning_content": "still thinking"},
response_metadata={"finish_reason": "length"},
)
]
)
def _successful_response() -> ModelResponse:
return ModelResponse(result=[AIMessage(content="ok")])
@pytest.fixture(autouse=True)
def _clean_chain():
"""Ensure a clean fallback chain for every test."""
@@ -454,6 +471,94 @@ class TestSynchronousFallback:
assert result is response
assert handler.call_count == 2
def test_truncated_primary_response_uses_fallback(self):
from EvoScientist.middleware.model_fallback import ModelFallbackMiddleware
add_fallback("fb", "prov")
req = _fake_request()
response = _successful_response()
handler = MagicMock(side_effect=[_truncated_response(), response])
with patch("EvoScientist.llm.models.get_chat_model") as mock_gcm:
mock_gcm.return_value = MagicMock()
result = ModelFallbackMiddleware().wrap_model_call(req, handler)
assert result is response
assert handler.call_count == 2
class TestTruncatedResponseFallback:
"""Empty truncated model results must participate in the fallback chain."""
async def test_truncated_primary_response_uses_fallback(self):
from EvoScientist.middleware.model_fallback import ModelFallbackMiddleware
add_fallback("fb", "prov")
req = _fake_request()
response = _successful_response()
handler = AsyncMock(side_effect=[_truncated_response(), response])
with patch("EvoScientist.llm.models.get_chat_model") as mock_gcm:
mock_gcm.return_value = MagicMock()
result = await ModelFallbackMiddleware().awrap_model_call(req, handler)
assert result is response
assert handler.await_count == 2
async def test_truncated_fallback_continues_to_next_model(self):
from EvoScientist.middleware.model_fallback import ModelFallbackMiddleware
add_fallback("fb-a", "prov-a")
add_fallback("fb-b", "prov-b")
req = _fake_request()
response = _successful_response()
handler = AsyncMock(
side_effect=[_truncated_response(), _truncated_response(), response]
)
with patch("EvoScientist.llm.models.get_chat_model") as mock_gcm:
mock_gcm.return_value = MagicMock()
result = await ModelFallbackMiddleware().awrap_model_call(req, handler)
assert result is response
assert handler.await_count == 3
assert mock_gcm.call_count == 2
async def test_exhausted_truncated_fallbacks_use_last_provider(self):
from EvoScientist.llm.errors import ProviderStreamError
from EvoScientist.middleware.error_normalization import (
ModelOutputTruncatedError,
)
from EvoScientist.middleware.model_fallback import ModelFallbackMiddleware
def _make_openai_model(base_url=None):
cls = type(
"ChatOpenAI",
(),
{"__module__": "langchain_openai.chat_models.base"},
)
model = cls()
model.openai_api_base = base_url
return model
add_fallback("moonshot-model", "moonshot")
req = _fake_request()
req.model = _make_openai_model()
fallback_model = _make_openai_model(base_url="https://api.moonshot.cn/v1")
req.override = MagicMock(
side_effect=lambda **kw: SimpleNamespace(model=kw.get("model", req.model))
)
handler = AsyncMock(return_value=_truncated_response())
with patch("EvoScientist.llm.models.get_chat_model") as mock_gcm:
mock_gcm.return_value = fallback_model
with pytest.raises(ProviderStreamError) as exc_info:
await ModelFallbackMiddleware().awrap_model_call(req, handler)
assert exc_info.value.provider == "moonshot"
assert isinstance(exc_info.value.__cause__, ModelOutputTruncatedError)
assert handler.await_count == 2
# ═════════════════════════════════════════════════════════════════
# 4. UI emit callback