From cf4b78e91fb03fe47a2513eca2b1e8f7f1f56054 Mon Sep 17 00:00:00 2001 From: Siddharth Balyan <52913345+alt-glitch@users.noreply.github.com> Date: Thu, 10 Sep 2026 02:21:15 +0530 Subject: [PATCH] fix(tool-search): a query no tool answers returns nothing, not five tools sharing one word (#106676) search_catalog admitted every document with BM25 score > 0 and then padded to `limit`. BM25 sums over the tokens a document shares with the query, so on a 300-tool catalog "send gmail email" returned five incident tools that shared only "email", and the discriminating word ("gmail", in no document) had no say. The model read those as the answer. Admission is now the query's rarest token: a document is a result only if it contains the query token with the highest IDF, the one that names the intent. Common verbs ("send", "read", "create") sit in dozens of documents and never gate; vendor and object words ("gmail", "github", "incident") do. A token no document carries admits nothing, and the existing empty-group hint tells the model to retry without it. The name-substring fallback is deleted: it admitted tools that matched no query token at all. Result descriptions are clipped at 500 characters instead of 400. Over 353 vendor tool descriptions, 500 keeps 91% whole and every first sentence (first-sentence max 329); 400 kept 82%. Measured on the live 311-tool catalog with 25 hand-labelled queries: precision@5 0.18 -> 0.43, wrong names returned 102 -> 66, false positives on absent intents 17 -> 13. Live before/after: "send gmail email" went from five betterstack tools to an empty group with the retry hint; "linear create issue" and "betterstack incident" are unchanged. --- tests/tools/test_deferral_fixes.py | 19 ++++++----- tests/tools/test_tool_search.py | 2 +- tests/tools/test_tool_search_multiquery.py | 17 ++++++++-- tools/tool_search.py | 4 ++- tools/tool_search_catalog.py | 34 +++++++++++++------ .../docs/user-guide/features/tool-search.md | 11 +++--- 6 files changed, 58 insertions(+), 29 deletions(-) diff --git a/tests/tools/test_deferral_fixes.py b/tests/tools/test_deferral_fixes.py index 676eb6e736..ae624d6187 100644 --- a/tests/tools/test_deferral_fixes.py +++ b/tests/tools/test_deferral_fixes.py @@ -268,10 +268,9 @@ class TestSourceNameIndexing: defs = [_td("mcp__linear__create_issue", "Create an issue."), _td("mcp__slack__post_message", "Post a message.")] catalog = build_catalog(defs) - hits = search_catalog(catalog, "mcp message") - # Before the fix "mcp" BM25-matched both docs, so both came - # back and the order was decided by document length, not by - # the term the model actually meant. + # The prefix is in no document, so it can never match or rank. + assert all("mcp" not in e._tokens for e in catalog) + hits = search_catalog(catalog, "message") assert [h.name for h in hits] == ["mcp__slack__post_message"] finally: for n in names: @@ -310,9 +309,11 @@ class TestSourceNameIndexing: for name in names: registry.deregister(name) - def test_substring_fallback_covers_token_misses(self): - """"hub" is a substring of github but never a token — the fallback - (not BM25) must return the github tools.""" + def test_unknown_token_returns_nothing(self): + """A token no document carries is the query's rarest token, so it gates and nothing + is admitted: an empty group, not `limit` tools sharing a common word. The old + name-substring fallback ("hub" -> github_*) is gone with it; the substring path + admitted tools that matched no query token at all.""" from tools.registry import registry names = [ @@ -323,9 +324,9 @@ class TestSourceNameIndexing: defs = [_td("github_create_issue", "Create an issue."), _td("github_merge_pr", "Merge a pull request.")] catalog = build_catalog(defs) - hits = search_catalog(catalog, "hub") - assert {h.name for h in hits} == {"github_create_issue", "github_merge_pr"} assert search_catalog(catalog, "zzzz") == [] + assert search_catalog(catalog, "hub") == [] + assert search_catalog(catalog, "create zzzz issue") == [] finally: for n in names: registry.deregister(n) diff --git a/tests/tools/test_tool_search.py b/tests/tools/test_tool_search.py index 898b432dea..49c686410d 100644 --- a/tests/tools/test_tool_search.py +++ b/tests/tools/test_tool_search.py @@ -247,7 +247,7 @@ class TestThresholdGate: # --------------------------------------------------------------------------- -# Retrieval (BM25 + substring fallback) +# Retrieval (BM25, rarest-token admission) # --------------------------------------------------------------------------- diff --git a/tests/tools/test_tool_search_multiquery.py b/tests/tools/test_tool_search_multiquery.py index be77be0207..ec5fc6e353 100644 --- a/tests/tools/test_tool_search_multiquery.py +++ b/tests/tools/test_tool_search_multiquery.py @@ -78,14 +78,25 @@ class TestStemming: assert "mq_linear_create_issue" in names assert "mq_linear_list_issues" in names - def test_substring_fallback_still_uses_raw_name(self, issue_defs): - """Fallback matches the unstemmed tool name, unchanged by stemming.""" + def test_rarest_query_token_gates_admission(self, issue_defs): + """A document that lacks the query's rarest token is not a result, however many + common tokens it shares. In this catalog 'issue' and 'linear' are each in two tools + and 'slack' in one, so 'slack' gates: the two linear tools share two of the three + query tokens and still do not come back.""" from tools.tool_search import build_catalog, search_catalog catalog = build_catalog(issue_defs) - names = [h.name for h in search_catalog(catalog, "post_mess", limit=5)] + names = [h.name for h in search_catalog(catalog, "linear issue slack", limit=5)] assert names == ["mq_slack_post_message"] + def test_token_no_document_carries_admits_nothing(self, issue_defs): + """'send gmail email' against a catalog with no gmail tool returns nothing rather + than five tools that merely share 'message' or 'email'.""" + from tools.tool_search import build_catalog, search_catalog + + catalog = build_catalog(issue_defs) + assert search_catalog(catalog, "post gmail message", limit=5) == [] + def test_single_token_stems_are_cached(self): from tools.tool_search_catalog import _stem, _tokenize diff --git a/tools/tool_search.py b/tools/tool_search.py index ab2af9b0e6..1ac0f2967a 100644 --- a/tools/tool_search.py +++ b/tools/tool_search.py @@ -347,7 +347,9 @@ def _shared_tool_record(entry: CatalogEntry) -> Dict[str, Any]: except (TypeError, KeyError, AttributeError): required = [] return {"source": entry.source, "source_name": entry.source_name, - "description": (entry.description or "")[:400], # cap chatty MCP descriptions + # 500 keeps 9 in 10 vendor tool descriptions whole and every first + # sentence (measured p90 575, first-sentence max 329 over 353 tools). + "description": (entry.description or "")[:500], "required": [r[:64] for r in (required if isinstance(required, list) else []) if isinstance(r, str)][:32]} diff --git a/tools/tool_search_catalog.py b/tools/tool_search_catalog.py index 52f80dc583..007e7571db 100644 --- a/tools/tool_search_catalog.py +++ b/tools/tool_search_catalog.py @@ -143,25 +143,37 @@ def _corpus_stats(catalog: List[CatalogEntry]) -> _CorpusStats: return doc_lengths, avg_dl, dict(doc_freq), len(catalog) +def _gate_token(query_tokens: List[str], doc_freq: Dict[str, int], n_docs: int) -> str: + """The query token with the highest IDF: the word that names the intent. ``send``, + ``read``, ``create`` sit in dozens of tool documents and separate nothing; ``gmail``, + ``github``, ``incident`` sit in a few and separate everything. A document without this + token answered a different question, however many common tokens it shares.""" + def _idf(token: str) -> float: + df = doc_freq.get(token, 0) + return math.log(1 + (n_docs - df + 0.5) / (df + 0.5)) + return max(query_tokens, key=_idf) + + def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *, corpus_stats: Optional[_CorpusStats] = None) -> List[CatalogEntry]: """Top-``limit`` catalog entries for ``query`` by BM25 (exact name match ranks first). - Falls back to a name-substring match only when NO query token appears in any document - (e.g. "hub" vs ``github_*``); the IDF variant is strictly positive, so a hit anywhere - suppresses the fallback.""" + + Admission is by the query's rarest token (:func:`_gate_token`), not by ``score > 0``: + BM25 is additive over the tokens a document shares with the query, so on a large catalog + ``score > 0`` admits one-token matches and fills every slot with them (measured: "send + gmail email" returned 5 incident tools that only shared ``email``). A token no document + carries admits nothing; the caller's empty-group hint tells the model to retry without it.""" query_tokens = _tokenize(query) if catalog and limit > 0 else [] if not query_tokens: return [] corpus_stats = corpus_stats or _corpus_stats(catalog) - scored: List[Tuple[float, CatalogEntry]] = [] + gate = _gate_token(query_tokens, corpus_stats[2], corpus_stats[3]) exact_name = query.strip().lower() - for entry in catalog: - s = (float("inf") if entry.name.lower() == exact_name - else _bm25_score(query_tokens, entry._tokens, *corpus_stats)) - if s > 0: - scored.append((s, entry)) - if not scored: - scored = [(0.1, entry) for entry in catalog if query.lower() in entry.name.lower()] + scored = [ + (float("inf") if entry.name.lower() == exact_name + else _bm25_score(query_tokens, entry._tokens, *corpus_stats), entry) + for entry in catalog + if entry.name.lower() == exact_name or gate in entry._tokens] scored.sort(key=lambda x: x[0], reverse=True) return [e for _, e in scored[:limit]] diff --git a/website/docs/user-guide/features/tool-search.md b/website/docs/user-guide/features/tool-search.md index f64e59caf1..189243177c 100644 --- a/website/docs/user-guide/features/tool-search.md +++ b/website/docs/user-guide/features/tool-search.md @@ -179,10 +179,13 @@ to any progressive-disclosure design, not specific to this implementation: finds that server's tools even when a tool's own name doesn't carry the service), description, and parameter names, with Snowball stemming (English) applied to both the index and the query so - morphological variants match ("issues" finds `create_issue`). Falls - back to a literal substring match on the tool name when no query - token matches any document (e.g. searching `"hub"` where the token is - `github`). + morphological variants match ("issues" finds `create_issue`). A tool is + a result only if it contains the query's rarest token (the one in the + fewest tool documents, so the word that names the intent: `gmail`, + `github`, `incident`, not `send` or `create`). A query whose rarest + token appears in no tool returns an empty group with the connected + sources and a retry hint, instead of `limit` tools that share one + common word. - **Parallel execution unwraps the bridge.** The batch planner decides concurrency on the *underlying* tool of a `tool_call`, not on the literal bridge name — so an MCP server opted in via