diff --git a/plugins/web/keyless_mcp.py b/plugins/web/keyless_mcp.py index 1855c8897c..116feaff32 100644 --- a/plugins/web/keyless_mcp.py +++ b/plugins/web/keyless_mcp.py @@ -120,7 +120,12 @@ def use_keyless(name: str, api_key: str) -> bool: def _parse_mcp_body(body: str) -> str: """First text content item from an MCP tools/call response — plain-JSON bodies (Parallel) or SSE ``data: {...}`` lines (Exa). Raises :class:`KeylessMCPError` for - JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit).""" + JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit). + + SSE frames are split on ``\\n`` only: ``str.splitlines()`` also breaks on U+0085 / + U+2028 / U+2029, which legitimately occur inside CJK page text and would cut a + ``data:`` line in two. A parsed envelope with no text is reported as such — it is + the vendor's answer, not an unrecognized shape.""" def _from_payload(payload: str) -> Optional[str]: payload = payload.strip() @@ -134,19 +139,21 @@ def _parse_mcp_body(body: str) -> str: texts = [c.get("text", "") for c in result.get("content") or [] if isinstance(c, dict)] if result.get("isError"): raise KeylessMCPError(" ".join(t for t in texts if t) or "MCP tool call failed") - return next((str(t) for t in texts if t), None) + return next((str(t) for t in texts if t), "") stripped = body.strip() candidates = [stripped] if stripped.startswith("{") else [] - candidates += [line[len("data: "):] for line in body.splitlines() if line.startswith("data: ")] + candidates += [line[len("data: "):] for line in body.split("\n") if line.startswith("data: ")] + envelope_seen = False for candidate in candidates: try: text = _from_payload(candidate) except json.JSONDecodeError: continue - if text is not None: + if text: return text - raise KeylessMCPError("Unrecognized MCP response shape") + envelope_seen = envelope_seen or text == "" + raise KeylessMCPError("MCP response contained no text content" if envelope_seen else "Unrecognized MCP response shape") def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIMEOUT_SECONDS) -> str: @@ -161,7 +168,10 @@ def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIM raise KeylessMCPError(f"request failed: {exc}") from exc if response.status_code >= 400: raise KeylessMCPError(f"HTTP {response.status_code}: {response.text[:300]}") - return _parse_mcp_body(response.text) + # JSON-RPC and SSE bodies are UTF-8 by spec, but ``text/event-stream`` carries no charset and + # ``requests`` then decodes ``.text`` as ISO-8859-1 — mojibake for every non-ASCII result and, + # for CJK, stray U+0085 line breaks that made the envelope unparseable. + return _parse_mcp_body(response.content.decode("utf-8", errors="replace")) # --- Parallel (search.parallel.ai) — JSON text payloads ----------------------- diff --git a/tests/tools/test_web_keyless_fallback.py b/tests/tools/test_web_keyless_fallback.py index 1de9fba14b..762d3c52f7 100644 --- a/tests/tools/test_web_keyless_fallback.py +++ b/tests/tools/test_web_keyless_fallback.py @@ -89,6 +89,29 @@ class TestParseMcpBody: with pytest.raises(keyless_mcp.KeylessMCPError): keyless_mcp._parse_mcp_body("nope") + def test_cjk_sse_body_survives_charset_less_event_stream(self): + """Exa answers ``text/event-stream`` without a charset; ``requests`` then decodes ``.text`` as + ISO-8859-1, and the U+0085 inside CJK UTF-8 sequences split the ``data:`` line under + ``splitlines()`` — a valid CJK result surfaced as "Unrecognized MCP response shape".""" + import requests + + title = "光伏发电站组件清洗与性能监测规范" + payload = {"result": {"content": [{"type": "text", "text": f"Title: {title}\nURL: https://x.example"}]}} + response = requests.Response() + response.status_code = 200 + response.headers["Content-Type"] = "text/event-stream" + response.encoding = "ISO-8859-1" # what the adapter picks for text/* without a charset + response._content = f"event: message\ndata: {json.dumps(payload, ensure_ascii=False)}\n\n".encode("utf-8") + assert "\x85" in response.text # the vendor body really does decode to mojibake via .text + with patch.object(requests, "post", return_value=response): + text = keyless_mcp.mcp_call(keyless_mcp.EXA_MCP_URL, "web_search_exa", {"query": title}) + assert title in text + + def test_parsed_envelope_without_text_names_the_condition(self): + body = json.dumps({"result": {"content": []}}) + with pytest.raises(keyless_mcp.KeylessMCPError, match="no text content"): + keyless_mcp._parse_mcp_body(body) + class TestExaTextParsing: def test_parses_blocks(self):