fix(web): keyless Exa decodes its SSE body as UTF-8 so CJK searches work
Exa's MCP endpoint answers `text/event-stream` without a charset, so `requests` decoded `.text` as ISO-8859-1: every non-ASCII character came back as mojibake, and for CJK results the UTF-8 continuation byte 0x85 became U+0085, which `str.splitlines()` treats as a line break — the `data:` JSON line was cut in two and the perfectly valid result surfaced as "Unrecognized MCP response shape", after which the ring fell through to keyless Firecrawl (403). Decode the bytes as UTF-8 (JSON-RPC and SSE are UTF-8 by spec) and split SSE frames on newlines only. A parsed envelope that really carries no text is now reported as "no text content" instead of an unrecognized shape.
This commit is contained in:
@@ -120,7 +120,12 @@ def use_keyless(name: str, api_key: str) -> bool:
|
||||
def _parse_mcp_body(body: str) -> str:
|
||||
"""First text content item from an MCP tools/call response — plain-JSON bodies
|
||||
(Parallel) or SSE ``data: {...}`` lines (Exa). Raises :class:`KeylessMCPError` for
|
||||
JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit)."""
|
||||
JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit).
|
||||
|
||||
SSE frames are split on ``\\n`` only: ``str.splitlines()`` also breaks on U+0085 /
|
||||
U+2028 / U+2029, which legitimately occur inside CJK page text and would cut a
|
||||
``data:`` line in two. A parsed envelope with no text is reported as such — it is
|
||||
the vendor's answer, not an unrecognized shape."""
|
||||
|
||||
def _from_payload(payload: str) -> Optional[str]:
|
||||
payload = payload.strip()
|
||||
@@ -134,19 +139,21 @@ def _parse_mcp_body(body: str) -> str:
|
||||
texts = [c.get("text", "") for c in result.get("content") or [] if isinstance(c, dict)]
|
||||
if result.get("isError"):
|
||||
raise KeylessMCPError(" ".join(t for t in texts if t) or "MCP tool call failed")
|
||||
return next((str(t) for t in texts if t), None)
|
||||
return next((str(t) for t in texts if t), "")
|
||||
|
||||
stripped = body.strip()
|
||||
candidates = [stripped] if stripped.startswith("{") else []
|
||||
candidates += [line[len("data: "):] for line in body.splitlines() if line.startswith("data: ")]
|
||||
candidates += [line[len("data: "):] for line in body.split("\n") if line.startswith("data: ")]
|
||||
envelope_seen = False
|
||||
for candidate in candidates:
|
||||
try:
|
||||
text = _from_payload(candidate)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if text is not None:
|
||||
if text:
|
||||
return text
|
||||
raise KeylessMCPError("Unrecognized MCP response shape")
|
||||
envelope_seen = envelope_seen or text == ""
|
||||
raise KeylessMCPError("MCP response contained no text content" if envelope_seen else "Unrecognized MCP response shape")
|
||||
|
||||
|
||||
def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIMEOUT_SECONDS) -> str:
|
||||
@@ -161,7 +168,10 @@ def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIM
|
||||
raise KeylessMCPError(f"request failed: {exc}") from exc
|
||||
if response.status_code >= 400:
|
||||
raise KeylessMCPError(f"HTTP {response.status_code}: {response.text[:300]}")
|
||||
return _parse_mcp_body(response.text)
|
||||
# JSON-RPC and SSE bodies are UTF-8 by spec, but ``text/event-stream`` carries no charset and
|
||||
# ``requests`` then decodes ``.text`` as ISO-8859-1 — mojibake for every non-ASCII result and,
|
||||
# for CJK, stray U+0085 line breaks that made the envelope unparseable.
|
||||
return _parse_mcp_body(response.content.decode("utf-8", errors="replace"))
|
||||
|
||||
|
||||
# --- Parallel (search.parallel.ai) — JSON text payloads -----------------------
|
||||
|
||||
@@ -89,6 +89,29 @@ class TestParseMcpBody:
|
||||
with pytest.raises(keyless_mcp.KeylessMCPError):
|
||||
keyless_mcp._parse_mcp_body("<html>nope</html>")
|
||||
|
||||
def test_cjk_sse_body_survives_charset_less_event_stream(self):
|
||||
"""Exa answers ``text/event-stream`` without a charset; ``requests`` then decodes ``.text`` as
|
||||
ISO-8859-1, and the U+0085 inside CJK UTF-8 sequences split the ``data:`` line under
|
||||
``splitlines()`` — a valid CJK result surfaced as "Unrecognized MCP response shape"."""
|
||||
import requests
|
||||
|
||||
title = "光伏发电站组件清洗与性能监测规范"
|
||||
payload = {"result": {"content": [{"type": "text", "text": f"Title: {title}\nURL: https://x.example"}]}}
|
||||
response = requests.Response()
|
||||
response.status_code = 200
|
||||
response.headers["Content-Type"] = "text/event-stream"
|
||||
response.encoding = "ISO-8859-1" # what the adapter picks for text/* without a charset
|
||||
response._content = f"event: message\ndata: {json.dumps(payload, ensure_ascii=False)}\n\n".encode("utf-8")
|
||||
assert "\x85" in response.text # the vendor body really does decode to mojibake via .text
|
||||
with patch.object(requests, "post", return_value=response):
|
||||
text = keyless_mcp.mcp_call(keyless_mcp.EXA_MCP_URL, "web_search_exa", {"query": title})
|
||||
assert title in text
|
||||
|
||||
def test_parsed_envelope_without_text_names_the_condition(self):
|
||||
body = json.dumps({"result": {"content": []}})
|
||||
with pytest.raises(keyless_mcp.KeylessMCPError, match="no text content"):
|
||||
keyless_mcp._parse_mcp_body(body)
|
||||
|
||||
|
||||
class TestExaTextParsing:
|
||||
def test_parses_blocks(self):
|
||||
|
||||
Reference in New Issue
Block a user