diff --git a/tests/tools/test_deferral_fixes.py b/tests/tools/test_deferral_fixes.py index 676eb6e736..ae624d6187 100644 --- a/tests/tools/test_deferral_fixes.py +++ b/tests/tools/test_deferral_fixes.py @@ -268,10 +268,9 @@ class TestSourceNameIndexing: defs = [_td("mcp__linear__create_issue", "Create an issue."), _td("mcp__slack__post_message", "Post a message.")] catalog = build_catalog(defs) - hits = search_catalog(catalog, "mcp message") - # Before the fix "mcp" BM25-matched both docs, so both came - # back and the order was decided by document length, not by - # the term the model actually meant. + # The prefix is in no document, so it can never match or rank. + assert all("mcp" not in e._tokens for e in catalog) + hits = search_catalog(catalog, "message") assert [h.name for h in hits] == ["mcp__slack__post_message"] finally: for n in names: @@ -310,9 +309,11 @@ class TestSourceNameIndexing: for name in names: registry.deregister(name) - def test_substring_fallback_covers_token_misses(self): - """"hub" is a substring of github but never a token — the fallback - (not BM25) must return the github tools.""" + def test_unknown_token_returns_nothing(self): + """A token no document carries is the query's rarest token, so it gates and nothing + is admitted: an empty group, not `limit` tools sharing a common word. The old + name-substring fallback ("hub" -> github_*) is gone with it; the substring path + admitted tools that matched no query token at all.""" from tools.registry import registry names = [ @@ -323,9 +324,9 @@ class TestSourceNameIndexing: defs = [_td("github_create_issue", "Create an issue."), _td("github_merge_pr", "Merge a pull request.")] catalog = build_catalog(defs) - hits = search_catalog(catalog, "hub") - assert {h.name for h in hits} == {"github_create_issue", "github_merge_pr"} assert search_catalog(catalog, "zzzz") == [] + assert search_catalog(catalog, "hub") == [] + assert search_catalog(catalog, "create zzzz issue") == [] finally: for n in names: registry.deregister(n) diff --git a/tests/tools/test_tool_search.py b/tests/tools/test_tool_search.py index 898b432dea..49c686410d 100644 --- a/tests/tools/test_tool_search.py +++ b/tests/tools/test_tool_search.py @@ -247,7 +247,7 @@ class TestThresholdGate: # --------------------------------------------------------------------------- -# Retrieval (BM25 + substring fallback) +# Retrieval (BM25, rarest-token admission) # --------------------------------------------------------------------------- diff --git a/tests/tools/test_tool_search_multiquery.py b/tests/tools/test_tool_search_multiquery.py index be77be0207..ec5fc6e353 100644 --- a/tests/tools/test_tool_search_multiquery.py +++ b/tests/tools/test_tool_search_multiquery.py @@ -78,14 +78,25 @@ class TestStemming: assert "mq_linear_create_issue" in names assert "mq_linear_list_issues" in names - def test_substring_fallback_still_uses_raw_name(self, issue_defs): - """Fallback matches the unstemmed tool name, unchanged by stemming.""" + def test_rarest_query_token_gates_admission(self, issue_defs): + """A document that lacks the query's rarest token is not a result, however many + common tokens it shares. In this catalog 'issue' and 'linear' are each in two tools + and 'slack' in one, so 'slack' gates: the two linear tools share two of the three + query tokens and still do not come back.""" from tools.tool_search import build_catalog, search_catalog catalog = build_catalog(issue_defs) - names = [h.name for h in search_catalog(catalog, "post_mess", limit=5)] + names = [h.name for h in search_catalog(catalog, "linear issue slack", limit=5)] assert names == ["mq_slack_post_message"] + def test_token_no_document_carries_admits_nothing(self, issue_defs): + """'send gmail email' against a catalog with no gmail tool returns nothing rather + than five tools that merely share 'message' or 'email'.""" + from tools.tool_search import build_catalog, search_catalog + + catalog = build_catalog(issue_defs) + assert search_catalog(catalog, "post gmail message", limit=5) == [] + def test_single_token_stems_are_cached(self): from tools.tool_search_catalog import _stem, _tokenize diff --git a/tools/tool_search.py b/tools/tool_search.py index ab2af9b0e6..1ac0f2967a 100644 --- a/tools/tool_search.py +++ b/tools/tool_search.py @@ -347,7 +347,9 @@ def _shared_tool_record(entry: CatalogEntry) -> Dict[str, Any]: except (TypeError, KeyError, AttributeError): required = [] return {"source": entry.source, "source_name": entry.source_name, - "description": (entry.description or "")[:400], # cap chatty MCP descriptions + # 500 keeps 9 in 10 vendor tool descriptions whole and every first + # sentence (measured p90 575, first-sentence max 329 over 353 tools). + "description": (entry.description or "")[:500], "required": [r[:64] for r in (required if isinstance(required, list) else []) if isinstance(r, str)][:32]} diff --git a/tools/tool_search_catalog.py b/tools/tool_search_catalog.py index 52f80dc583..007e7571db 100644 --- a/tools/tool_search_catalog.py +++ b/tools/tool_search_catalog.py @@ -143,25 +143,37 @@ def _corpus_stats(catalog: List[CatalogEntry]) -> _CorpusStats: return doc_lengths, avg_dl, dict(doc_freq), len(catalog) +def _gate_token(query_tokens: List[str], doc_freq: Dict[str, int], n_docs: int) -> str: + """The query token with the highest IDF: the word that names the intent. ``send``, + ``read``, ``create`` sit in dozens of tool documents and separate nothing; ``gmail``, + ``github``, ``incident`` sit in a few and separate everything. A document without this + token answered a different question, however many common tokens it shares.""" + def _idf(token: str) -> float: + df = doc_freq.get(token, 0) + return math.log(1 + (n_docs - df + 0.5) / (df + 0.5)) + return max(query_tokens, key=_idf) + + def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *, corpus_stats: Optional[_CorpusStats] = None) -> List[CatalogEntry]: """Top-``limit`` catalog entries for ``query`` by BM25 (exact name match ranks first). - Falls back to a name-substring match only when NO query token appears in any document - (e.g. "hub" vs ``github_*``); the IDF variant is strictly positive, so a hit anywhere - suppresses the fallback.""" + + Admission is by the query's rarest token (:func:`_gate_token`), not by ``score > 0``: + BM25 is additive over the tokens a document shares with the query, so on a large catalog + ``score > 0`` admits one-token matches and fills every slot with them (measured: "send + gmail email" returned 5 incident tools that only shared ``email``). A token no document + carries admits nothing; the caller's empty-group hint tells the model to retry without it.""" query_tokens = _tokenize(query) if catalog and limit > 0 else [] if not query_tokens: return [] corpus_stats = corpus_stats or _corpus_stats(catalog) - scored: List[Tuple[float, CatalogEntry]] = [] + gate = _gate_token(query_tokens, corpus_stats[2], corpus_stats[3]) exact_name = query.strip().lower() - for entry in catalog: - s = (float("inf") if entry.name.lower() == exact_name - else _bm25_score(query_tokens, entry._tokens, *corpus_stats)) - if s > 0: - scored.append((s, entry)) - if not scored: - scored = [(0.1, entry) for entry in catalog if query.lower() in entry.name.lower()] + scored = [ + (float("inf") if entry.name.lower() == exact_name + else _bm25_score(query_tokens, entry._tokens, *corpus_stats), entry) + for entry in catalog + if entry.name.lower() == exact_name or gate in entry._tokens] scored.sort(key=lambda x: x[0], reverse=True) return [e for _, e in scored[:limit]] diff --git a/website/docs/user-guide/features/tool-search.md b/website/docs/user-guide/features/tool-search.md index f64e59caf1..189243177c 100644 --- a/website/docs/user-guide/features/tool-search.md +++ b/website/docs/user-guide/features/tool-search.md @@ -179,10 +179,13 @@ to any progressive-disclosure design, not specific to this implementation: finds that server's tools even when a tool's own name doesn't carry the service), description, and parameter names, with Snowball stemming (English) applied to both the index and the query so - morphological variants match ("issues" finds `create_issue`). Falls - back to a literal substring match on the tool name when no query - token matches any document (e.g. searching `"hub"` where the token is - `github`). + morphological variants match ("issues" finds `create_issue`). A tool is + a result only if it contains the query's rarest token (the one in the + fewest tool documents, so the word that names the intent: `gmail`, + `github`, `incident`, not `send` or `create`). A query whose rarest + token appears in no tool returns an empty group with the connected + sources and a retry hint, instead of `limit` tools that share one + common word. - **Parallel execution unwraps the bridge.** The batch planner decides concurrency on the *underlying* tool of a `tool_call`, not on the literal bridge name — so an MCP server opted in via