fix(web): keyless Exa decodes its SSE body as UTF-8 so CJK searches work

Exa's MCP endpoint answers `text/event-stream` without a charset, so
`requests` decoded `.text` as ISO-8859-1: every non-ASCII character came back
as mojibake, and for CJK results the UTF-8 continuation byte 0x85 became
U+0085, which `str.splitlines()` treats as a line break — the `data:` JSON
line was cut in two and the perfectly valid result surfaced as "Unrecognized
MCP response shape", after which the ring fell through to keyless Firecrawl
(403). Decode the bytes as UTF-8 (JSON-RPC and SSE are UTF-8 by spec) and
split SSE frames on newlines only. A parsed envelope that really carries no
text is now reported as "no text content" instead of an unrecognized shape.
This commit is contained in:
teknium1
2026-09-15 12:05:21 -07:00
committed by Teknium
parent 80f76edfaf
commit 28cdd72815
2 changed files with 39 additions and 6 deletions
+16 -6
View File
@@ -120,7 +120,12 @@ def use_keyless(name: str, api_key: str) -> bool:
def _parse_mcp_body(body: str) -> str:
"""First text content item from an MCP tools/call response — plain-JSON bodies
(Parallel) or SSE ``data: {...}`` lines (Exa). Raises :class:`KeylessMCPError` for
JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit)."""
JSON-RPC errors and ``isError`` tool results (e.g. Exa's free-tier rate limit).
SSE frames are split on ``\\n`` only: ``str.splitlines()`` also breaks on U+0085 /
U+2028 / U+2029, which legitimately occur inside CJK page text and would cut a
``data:`` line in two. A parsed envelope with no text is reported as such — it is
the vendor's answer, not an unrecognized shape."""
def _from_payload(payload: str) -> Optional[str]:
payload = payload.strip()
@@ -134,19 +139,21 @@ def _parse_mcp_body(body: str) -> str:
texts = [c.get("text", "") for c in result.get("content") or [] if isinstance(c, dict)]
if result.get("isError"):
raise KeylessMCPError(" ".join(t for t in texts if t) or "MCP tool call failed")
return next((str(t) for t in texts if t), None)
return next((str(t) for t in texts if t), "")
stripped = body.strip()
candidates = [stripped] if stripped.startswith("{") else []
candidates += [line[len("data: "):] for line in body.splitlines() if line.startswith("data: ")]
candidates += [line[len("data: "):] for line in body.split("\n") if line.startswith("data: ")]
envelope_seen = False
for candidate in candidates:
try:
text = _from_payload(candidate)
except json.JSONDecodeError:
continue
if text is not None:
if text:
return text
raise KeylessMCPError("Unrecognized MCP response shape")
envelope_seen = envelope_seen or text == ""
raise KeylessMCPError("MCP response contained no text content" if envelope_seen else "Unrecognized MCP response shape")
def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIMEOUT_SECONDS) -> str:
@@ -161,7 +168,10 @@ def mcp_call(url: str, tool: str, arguments: Dict[str, Any], timeout: int = _TIM
raise KeylessMCPError(f"request failed: {exc}") from exc
if response.status_code >= 400:
raise KeylessMCPError(f"HTTP {response.status_code}: {response.text[:300]}")
return _parse_mcp_body(response.text)
# JSON-RPC and SSE bodies are UTF-8 by spec, but ``text/event-stream`` carries no charset and
# ``requests`` then decodes ``.text`` as ISO-8859-1 — mojibake for every non-ASCII result and,
# for CJK, stray U+0085 line breaks that made the envelope unparseable.
return _parse_mcp_body(response.content.decode("utf-8", errors="replace"))
# --- Parallel (search.parallel.ai) — JSON text payloads -----------------------
+23
View File
@@ -89,6 +89,29 @@ class TestParseMcpBody:
with pytest.raises(keyless_mcp.KeylessMCPError):
keyless_mcp._parse_mcp_body("<html>nope</html>")
def test_cjk_sse_body_survives_charset_less_event_stream(self):
"""Exa answers ``text/event-stream`` without a charset; ``requests`` then decodes ``.text`` as
ISO-8859-1, and the U+0085 inside CJK UTF-8 sequences split the ``data:`` line under
``splitlines()`` — a valid CJK result surfaced as "Unrecognized MCP response shape"."""
import requests
title = "光伏发电站组件清洗与性能监测规范"
payload = {"result": {"content": [{"type": "text", "text": f"Title: {title}\nURL: https://x.example"}]}}
response = requests.Response()
response.status_code = 200
response.headers["Content-Type"] = "text/event-stream"
response.encoding = "ISO-8859-1" # what the adapter picks for text/* without a charset
response._content = f"event: message\ndata: {json.dumps(payload, ensure_ascii=False)}\n\n".encode("utf-8")
assert "\x85" in response.text # the vendor body really does decode to mojibake via .text
with patch.object(requests, "post", return_value=response):
text = keyless_mcp.mcp_call(keyless_mcp.EXA_MCP_URL, "web_search_exa", {"query": title})
assert title in text
def test_parsed_envelope_without_text_names_the_condition(self):
body = json.dumps({"result": {"content": []}})
with pytest.raises(keyless_mcp.KeylessMCPError, match="no text content"):
keyless_mcp._parse_mcp_body(body)
class TestExaTextParsing:
def test_parses_blocks(self):