fix(tool-search): a query no tool answers returns nothing, not five tools sharing one word (#106676)

search_catalog admitted every document with BM25 score > 0 and then padded
to `limit`. BM25 sums over the tokens a document shares with the query, so
on a 300-tool catalog "send gmail email" returned five incident tools that
shared only "email", and the discriminating word ("gmail", in no document)
had no say. The model read those as the answer.

Admission is now the query's rarest token: a document is a result only if it
contains the query token with the highest IDF, the one that names the intent.
Common verbs ("send", "read", "create") sit in dozens of documents and never
gate; vendor and object words ("gmail", "github", "incident") do. A token no
document carries admits nothing, and the existing empty-group hint tells the
model to retry without it. The name-substring fallback is deleted: it admitted
tools that matched no query token at all.

Result descriptions are clipped at 500 characters instead of 400. Over 353
vendor tool descriptions, 500 keeps 91% whole and every first sentence
(first-sentence max 329); 400 kept 82%.

Measured on the live 311-tool catalog with 25 hand-labelled queries:
precision@5 0.18 -> 0.43, wrong names returned 102 -> 66, false positives on
absent intents 17 -> 13. Live before/after: "send gmail email" went from five
betterstack tools to an empty group with the retry hint; "linear create issue"
and "betterstack incident" are unchanged.
This commit is contained in:
Siddharth Balyan
2026-09-10 02:21:15 +05:30
committed by GitHub
parent ff76e65e14
commit cf4b78e91f
6 changed files with 58 additions and 29 deletions
+10 -9
View File
@@ -268,10 +268,9 @@ class TestSourceNameIndexing:
defs = [_td("mcp__linear__create_issue", "Create an issue."),
_td("mcp__slack__post_message", "Post a message.")]
catalog = build_catalog(defs)
hits = search_catalog(catalog, "mcp message")
# Before the fix "mcp" BM25-matched both docs, so both came
# back and the order was decided by document length, not by
# the term the model actually meant.
# The prefix is in no document, so it can never match or rank.
assert all("mcp" not in e._tokens for e in catalog)
hits = search_catalog(catalog, "message")
assert [h.name for h in hits] == ["mcp__slack__post_message"]
finally:
for n in names:
@@ -310,9 +309,11 @@ class TestSourceNameIndexing:
for name in names:
registry.deregister(name)
def test_substring_fallback_covers_token_misses(self):
""""hub" is a substring of github but never a token — the fallback
(not BM25) must return the github tools."""
def test_unknown_token_returns_nothing(self):
"""A token no document carries is the query's rarest token, so it gates and nothing
is admitted: an empty group, not `limit` tools sharing a common word. The old
name-substring fallback ("hub" -> github_*) is gone with it; the substring path
admitted tools that matched no query token at all."""
from tools.registry import registry
names = [
@@ -323,9 +324,9 @@ class TestSourceNameIndexing:
defs = [_td("github_create_issue", "Create an issue."),
_td("github_merge_pr", "Merge a pull request.")]
catalog = build_catalog(defs)
hits = search_catalog(catalog, "hub")
assert {h.name for h in hits} == {"github_create_issue", "github_merge_pr"}
assert search_catalog(catalog, "zzzz") == []
assert search_catalog(catalog, "hub") == []
assert search_catalog(catalog, "create zzzz issue") == []
finally:
for n in names:
registry.deregister(n)
+1 -1
View File
@@ -247,7 +247,7 @@ class TestThresholdGate:
# ---------------------------------------------------------------------------
# Retrieval (BM25 + substring fallback)
# Retrieval (BM25, rarest-token admission)
# ---------------------------------------------------------------------------
+14 -3
View File
@@ -78,14 +78,25 @@ class TestStemming:
assert "mq_linear_create_issue" in names
assert "mq_linear_list_issues" in names
def test_substring_fallback_still_uses_raw_name(self, issue_defs):
"""Fallback matches the unstemmed tool name, unchanged by stemming."""
def test_rarest_query_token_gates_admission(self, issue_defs):
"""A document that lacks the query's rarest token is not a result, however many
common tokens it shares. In this catalog 'issue' and 'linear' are each in two tools
and 'slack' in one, so 'slack' gates: the two linear tools share two of the three
query tokens and still do not come back."""
from tools.tool_search import build_catalog, search_catalog
catalog = build_catalog(issue_defs)
names = [h.name for h in search_catalog(catalog, "post_mess", limit=5)]
names = [h.name for h in search_catalog(catalog, "linear issue slack", limit=5)]
assert names == ["mq_slack_post_message"]
def test_token_no_document_carries_admits_nothing(self, issue_defs):
"""'send gmail email' against a catalog with no gmail tool returns nothing rather
than five tools that merely share 'message' or 'email'."""
from tools.tool_search import build_catalog, search_catalog
catalog = build_catalog(issue_defs)
assert search_catalog(catalog, "post gmail message", limit=5) == []
def test_single_token_stems_are_cached(self):
from tools.tool_search_catalog import _stem, _tokenize
+3 -1
View File
@@ -347,7 +347,9 @@ def _shared_tool_record(entry: CatalogEntry) -> Dict[str, Any]:
except (TypeError, KeyError, AttributeError):
required = []
return {"source": entry.source, "source_name": entry.source_name,
"description": (entry.description or "")[:400], # cap chatty MCP descriptions
# 500 keeps 9 in 10 vendor tool descriptions whole and every first
# sentence (measured p90 575, first-sentence max 329 over 353 tools).
"description": (entry.description or "")[:500],
"required": [r[:64] for r in (required if isinstance(required, list) else [])
if isinstance(r, str)][:32]}
+23 -11
View File
@@ -143,25 +143,37 @@ def _corpus_stats(catalog: List[CatalogEntry]) -> _CorpusStats:
return doc_lengths, avg_dl, dict(doc_freq), len(catalog)
def _gate_token(query_tokens: List[str], doc_freq: Dict[str, int], n_docs: int) -> str:
"""The query token with the highest IDF: the word that names the intent. ``send``,
``read``, ``create`` sit in dozens of tool documents and separate nothing; ``gmail``,
``github``, ``incident`` sit in a few and separate everything. A document without this
token answered a different question, however many common tokens it shares."""
def _idf(token: str) -> float:
df = doc_freq.get(token, 0)
return math.log(1 + (n_docs - df + 0.5) / (df + 0.5))
return max(query_tokens, key=_idf)
def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *,
corpus_stats: Optional[_CorpusStats] = None) -> List[CatalogEntry]:
"""Top-``limit`` catalog entries for ``query`` by BM25 (exact name match ranks first).
Falls back to a name-substring match only when NO query token appears in any document
(e.g. "hub" vs ``github_*``); the IDF variant is strictly positive, so a hit anywhere
suppresses the fallback."""
Admission is by the query's rarest token (:func:`_gate_token`), not by ``score > 0``:
BM25 is additive over the tokens a document shares with the query, so on a large catalog
``score > 0`` admits one-token matches and fills every slot with them (measured: "send
gmail email" returned 5 incident tools that only shared ``email``). A token no document
carries admits nothing; the caller's empty-group hint tells the model to retry without it."""
query_tokens = _tokenize(query) if catalog and limit > 0 else []
if not query_tokens:
return []
corpus_stats = corpus_stats or _corpus_stats(catalog)
scored: List[Tuple[float, CatalogEntry]] = []
gate = _gate_token(query_tokens, corpus_stats[2], corpus_stats[3])
exact_name = query.strip().lower()
for entry in catalog:
s = (float("inf") if entry.name.lower() == exact_name
else _bm25_score(query_tokens, entry._tokens, *corpus_stats))
if s > 0:
scored.append((s, entry))
if not scored:
scored = [(0.1, entry) for entry in catalog if query.lower() in entry.name.lower()]
scored = [
(float("inf") if entry.name.lower() == exact_name
else _bm25_score(query_tokens, entry._tokens, *corpus_stats), entry)
for entry in catalog
if entry.name.lower() == exact_name or gate in entry._tokens]
scored.sort(key=lambda x: x[0], reverse=True)
return [e for _, e in scored[:limit]]
@@ -179,10 +179,13 @@ to any progressive-disclosure design, not specific to this implementation:
finds that server's tools even when a tool's own name doesn't carry
the service), description, and parameter names, with Snowball
stemming (English) applied to both the index and the query so
morphological variants match ("issues" finds `create_issue`). Falls
back to a literal substring match on the tool name when no query
token matches any document (e.g. searching `"hub"` where the token is
`github`).
morphological variants match ("issues" finds `create_issue`). A tool is
a result only if it contains the query's rarest token (the one in the
fewest tool documents, so the word that names the intent: `gmail`,
`github`, `incident`, not `send` or `create`). A query whose rarest
token appears in no tool returns an empty group with the connected
sources and a retry hint, instead of `limit` tools that share one
common word.
- **Parallel execution unwraps the bridge.** The batch planner decides
concurrency on the *underlying* tool of a `tool_call`, not on the
literal bridge name — so an MCP server opted in via