From 28cdd728158ec37a963a9ed1d23dc7cbfa168694 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:05:21 -0700 Subject: [PATCH] fix(web): keyless Exa decodes its SSE body as UTF-8 so CJK searches work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Exa's MCP endpoint answers `text/event-stream` without a charset, so `requests` decoded `.text` as ISO-8859-1: every non-ASCII character came back as mojibake, and for CJK results the UTF-8 continuation byte 0x85 became U+0085, which `str.splitlines()` treats as a line break — the `data:` JSON line was cut in two and the perfectly valid result surfaced as "Unrecognized MCP response shape", after which the ring fell through to keyless Firecrawl (403). Decode the bytes as UTF-8 (JSON-RPC and SSE are UTF-8 by spec) and split SSE frames on newlines only. A parsed envelope that really carries no text is now reported as "no text content" instead of an unrecognized shape. --- plugins/web/keyless_mcp.py | 22 ++++++++++++++++------ tests/tools/test_web_keyless_fallback.py | 23 +++++++++++++++++++++++ 2 files changed, 39 insertions(+), 6 deletions(-) diff --git a/plugins/web/keyless_mcp.py b/plugins/web/keyless_mcp.py index 1855c8897c..116feaff32 100644 --- a/plugins/web/keyless_mcp.py +++ b/plugins/web/keyless_mcp.py @@ -120,7 +120,12 @@ def use_keyless(name: str, api_key: str) -> bool: def _parse_mcp_body(body: str) -> str: """First text content item from an MCP tools/call response — plain-JSON bodies (Parallel) or SSE ``data: {...}`` lines (Exa). Raises :class:`KeylessMCPError` for - JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit).""" + JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit). + + SSE frames are split on ``\\n`` only: ``str.splitlines()`` also breaks on U+0085 / + U+2028 / U+2029, which legitimately occur inside CJK page text and would cut a + ``data:`` line in two. A parsed envelope with no text is reported as such — it is + the vendor's answer, not an unrecognized shape.""" def _from_payload(payload: str) -> Optional[str]: payload = payload.strip() @@ -134,19 +139,21 @@ def _parse_mcp_body(body: str) -> str: texts = [c.get("text", "") for c in result.get("content") or [] if isinstance(c, dict)] if result.get("isError"): raise KeylessMCPError(" ".join(t for t in texts if t) or "MCP tool call failed") - return next((str(t) for t in texts if t), None) + return next((str(t) for t in texts if t), "") stripped = body.strip() candidates = [stripped] if stripped.startswith("{") else [] - candidates += [line[len("data: "):] for line in body.splitlines() if line.startswith("data: ")] + candidates += [line[len("data: "):] for line in body.split("\n") if line.startswith("data: ")] + envelope_seen = False for candidate in candidates: try: text = _from_payload(candidate) except json.JSONDecodeError: continue - if text is not None: + if text: return text - raise KeylessMCPError("Unrecognized MCP response shape") + envelope_seen = envelope_seen or text == "" + raise KeylessMCPError("MCP response contained no text content" if envelope_seen else "Unrecognized MCP response shape") def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIMEOUT_SECONDS) -> str: @@ -161,7 +168,10 @@ def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIM raise KeylessMCPError(f"request failed: {exc}") from exc if response.status_code >= 400: raise KeylessMCPError(f"HTTP {response.status_code}: {response.text[:300]}") - return _parse_mcp_body(response.text) + # JSON-RPC and SSE bodies are UTF-8 by spec, but ``text/event-stream`` carries no charset and + # ``requests`` then decodes ``.text`` as ISO-8859-1 — mojibake for every non-ASCII result and, + # for CJK, stray U+0085 line breaks that made the envelope unparseable. + return _parse_mcp_body(response.content.decode("utf-8", errors="replace")) # --- Parallel (search.parallel.ai) — JSON text payloads ----------------------- diff --git a/tests/tools/test_web_keyless_fallback.py b/tests/tools/test_web_keyless_fallback.py index 1de9fba14b..762d3c52f7 100644 --- a/tests/tools/test_web_keyless_fallback.py +++ b/tests/tools/test_web_keyless_fallback.py @@ -89,6 +89,29 @@ class TestParseMcpBody: with pytest.raises(keyless_mcp.KeylessMCPError): keyless_mcp._parse_mcp_body("nope") + def test_cjk_sse_body_survives_charset_less_event_stream(self): + """Exa answers ``text/event-stream`` without a charset; ``requests`` then decodes ``.text`` as + ISO-8859-1, and the U+0085 inside CJK UTF-8 sequences split the ``data:`` line under + ``splitlines()`` — a valid CJK result surfaced as "Unrecognized MCP response shape".""" + import requests + + title = "光伏发电站组件清洗与性能监测规范" + payload = {"result": {"content": [{"type": "text", "text": f"Title: {title}\nURL: https://x.example"}]}} + response = requests.Response() + response.status_code = 200 + response.headers["Content-Type"] = "text/event-stream" + response.encoding = "ISO-8859-1" # what the adapter picks for text/* without a charset + response._content = f"event: message\ndata: {json.dumps(payload, ensure_ascii=False)}\n\n".encode("utf-8") + assert "\x85" in response.text # the vendor body really does decode to mojibake via .text + with patch.object(requests, "post", return_value=response): + text = keyless_mcp.mcp_call(keyless_mcp.EXA_MCP_URL, "web_search_exa", {"query": title}) + assert title in text + + def test_parsed_envelope_without_text_names_the_condition(self): + body = json.dumps({"result": {"content": []}}) + with pytest.raises(keyless_mcp.KeylessMCPError, match="no text content"): + keyless_mcp._parse_mcp_body(body) + class TestExaTextParsing: def test_parses_blocks(self):