diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index d814fe09ea..53a3e5c7bc 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2843,10 +2843,16 @@ DEFAULT_CONFIG = { # listing_max_tokens). Range 0..100. "threshold_pct": 5, # When the model calls tool_search without a ``limit`` argument, - # how many hits to return. Range 1..max_search_limit. + # how many hits to return PER QUERY. Range 1..max_search_limit. "search_default_limit": 5, - # Hard upper bound the model can request via ``limit``. Range 1..50. - "max_search_limit": 20, + # Hard upper bound the model can request via ``limit`` (per + # query). Range 1..50. + "max_search_limit": 25, + # Max queries per tool_search call / names per tool_describe + # call. Over-cap calls error and the model retries with fewer. + # Floor 1, no upper clamp. + "max_queries": 10, + "max_describe_names": 10, # Skills-style catalog listing embedded in the tool_search bridge # description: every deferred tool's name + first sentence of its # description (≤60 chars), grouped by MCP server / toolset. Keeps diff --git a/pyproject.toml b/pyproject.toml index 8631154845..f3ad159012 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -57,6 +57,11 @@ dependencies = [ "prompt_toolkit==3.0.52", # Cron scheduler (built-in feature — scheduled cron/interval jobs use croniter). "croniter==6.0.0", + # Snowball stemming for tool_search's BM25 (tools/tool_search.py) — + # official Snowball project package, pure Python, zero transitive deps. + # Applied at index AND query time so morphological variants match + # ("issues" finds create_issue). + "snowballstemmer==3.1.1", # ``packaging`` is imported directly on three production paths but was never # declared, so it only reached users transitively (pip/uv pull it for other # tools). The slim official Docker image ships without it, where the diff --git a/scripts/analyze_livetest.py b/scripts/analyze_livetest.py index 77028b99b7..7e55b5c7c7 100644 --- a/scripts/analyze_livetest.py +++ b/scripts/analyze_livetest.py @@ -38,10 +38,20 @@ def fmt_bridge_seq(calls): inner = (c.get("args") or {}).get("name", "?") parts.append(f"tool_call→{inner}") elif c["name"] == "tool_search": - q = (c.get("args") or {}).get("query", "?") + args = c.get("args") or {} + qs = args.get("queries") + if isinstance(qs, list): + q = "; ".join(str(x) for x in qs) + else: # legacy single-query transcripts + q = str(args.get("query", "?")) parts.append(f"search('{q[:30]}')") elif c["name"] == "tool_describe": - n = (c.get("args") or {}).get("name", "?") + args = c.get("args") or {} + ns = args.get("names") + if isinstance(ns, list): + n = ", ".join(str(x) for x in ns) + else: # legacy single-name transcripts + n = str(args.get("name", "?")) parts.append(f"describe({n})") return " → ".join(parts) diff --git a/tests/plugins/test_a2a_schema_registration.py b/tests/plugins/test_a2a_schema_registration.py index c1614c2451..1af53019d4 100644 --- a/tests/plugins/test_a2a_schema_registration.py +++ b/tests/plugins/test_a2a_schema_registration.py @@ -32,10 +32,10 @@ def test_a2a_call_schema_round_trips_through_tool_describe(monkeypatch): described = json.loads( tool_search.dispatch_tool_describe( - {"name": "a2a_call"}, + {"names": ["a2a_call"]}, current_tool_defs=definitions, ) - ) + )["tools"]["a2a_call"] assert described["description"] assert described["parameters"]["required"] == ["agent", "message"] diff --git a/tests/tools/test_tool_search.py b/tests/tools/test_tool_search.py index 9d1554b400..1a0c01eaaa 100644 --- a/tests/tools/test_tool_search.py +++ b/tests/tools/test_tool_search.py @@ -287,11 +287,24 @@ class TestAssembly: class TestBridgeDispatch: - def test_tool_search_requires_query(self): + def test_tool_search_requires_queries(self): from tools.tool_search import dispatch_tool_search result = dispatch_tool_search({}, current_tool_defs=[]) assert "error" in json.loads(result) + def test_tool_search_rejects_empty_and_overcap_queries(self): + from tools.tool_search import dispatch_tool_search, ToolSearchConfig + cfg = ToolSearchConfig.from_raw({}) + assert "error" in json.loads(dispatch_tool_search( + {"queries": []}, current_tool_defs=[], config=cfg)) + assert "error" in json.loads(dispatch_tool_search( + {"queries": [" ", ""]}, current_tool_defs=[], config=cfg)) + over = ["q"] * (cfg.max_queries + 1) + parsed = json.loads(dispatch_tool_search( + {"queries": over}, current_tool_defs=[], config=cfg)) + assert "error" in parsed + assert "too many queries" in parsed["error"] + def test_empty_search_keeps_connected_sources_discoverable(self): from tools.registry import registry from tools.tool_search import dispatch_tool_search @@ -306,11 +319,12 @@ class TestBridgeDispatch: ) result = json.loads(dispatch_tool_search( - {"query": "unrelated vocabulary"}, + {"queries": ["unrelated vocabulary"]}, current_tool_defs=[tool_def], )) - assert result["matches"] == [] + assert result["results"] == [{"query": "unrelated vocabulary", "matches": []}] + assert result["tools"] == {} assert result["total_available"] == 1 assert result["available_sources"] == [ {"name": "recovery-catalog", "tool_count": 1}, @@ -351,12 +365,12 @@ class TestHandleFunctionCallIntegration: import model_tools result = model_tools.handle_function_call( function_name="tool_search", - function_args={"query": "nothing matches this"}, + function_args={"queries": ["nothing matches this"]}, ) parsed = json.loads(result) # Without a real registry, the matches will be empty, but the # dispatch path completed without error. - assert "matches" in parsed or "error" in parsed + assert "results" in parsed or "error" in parsed def test_tool_search_emits_one_terminal_hook(self, monkeypatch): """Inline bridge results still complete the tool lifecycle.""" @@ -378,12 +392,12 @@ class TestHandleFunctionCallIntegration: monkeypatch.setattr( tool_search, "dispatch_tool_search", - lambda *args, **kwargs: json.dumps({"matches": []}), + lambda *args, **kwargs: json.dumps({"results": []}), ) result = model_tools.handle_function_call( function_name="tool_search", - function_args={"query": "private-query"}, + function_args={"queries": ["private-query"]}, session_id="private-session", task_id="private-task", turn_id="private-turn", @@ -391,7 +405,7 @@ class TestHandleFunctionCallIntegration: tool_call_id="private-call", ) - assert json.loads(result) == {"matches": []} + assert json.loads(result) == {"results": []} assert len(events) == 1 hook_name, payload = events[0] assert hook_name == "post_tool_call" @@ -494,7 +508,7 @@ class TestRegression_ToolsetScoping: # out-of-scope plugin tool (or any of the host registry). result = model_tools.handle_function_call( function_name="tool_search", - function_args={"query": "mcp_scoped_gh", "limit": 5}, + function_args={"queries": ["mcp_scoped_gh"], "limit": 5}, enabled_toolsets=["mcp-scoped-gh"], ) parsed = json.loads(result) @@ -502,7 +516,8 @@ class TestRegression_ToolsetScoping: f"expected scoped catalog of 12, got {parsed['total_available']} " "— catalog leaked tools outside the session's toolsets" ) - hit_names = {m["name"] for m in parsed["matches"]} + hit_names = set(parsed["tools"]) + assert hit_names == {n for g in parsed["results"] for n in g["matches"]} assert "scoped_oos_plugin" not in hit_names diff --git a/tests/tools/test_tool_search_multiquery.py b/tests/tools/test_tool_search_multiquery.py new file mode 100644 index 0000000000..1aacd68f70 --- /dev/null +++ b/tests/tools/test_tool_search_multiquery.py @@ -0,0 +1,289 @@ +"""Multi-query ``tool_search``, batched ``tool_describe``, and stemming. + +Covers the upgrade that replaced the single ``query`` string with +``queries: [str, ...]`` (grouped, split-shape response), the single +``name`` with ``names: [str, ...]`` (map response with ``not_found``), +and added Snowball stemming to the shared tokenizer. +""" + +import json + +import pytest + + +def _td(name, desc, props=None, required=None): + return { + "type": "function", + "function": { + "name": name, + "description": desc, + "parameters": { + "type": "object", + "properties": props or {}, + "required": required or [], + }, + }, + } + + +def _register(name, toolset, desc="Deferred capability.", props=None, required=None): + from tools.registry import registry + + registry.register( + name=name, + handler=lambda args, **kw: json.dumps({"ok": True}), + schema=_td(name, desc, props, required), + toolset=toolset, + ) + return _td(name, desc, props, required) + + +@pytest.fixture +def issue_defs(): + """A small deferred catalog registered under an MCP toolset.""" + return [ + _register("mq_linear_create_issue", "mcp-mq-linear", + "Create a new issue in a team.", + {"title": {"type": "string"}, "team": {"type": "string"}}, + ["title", "team"]), + _register("mq_linear_list_issues", "mcp-mq-linear", + "List issues in the workspace.", + {"query": {"type": "string"}}), + _register("mq_slack_post_message", "mcp-mq-slack", + "Post a message to a channel.", + {"channel": {"type": "string"}, "text": {"type": "string"}}, + ["channel", "text"]), + ] + + +# --------------------------------------------------------------------------- +# Stemming +# --------------------------------------------------------------------------- + + +class TestStemming: + def test_tokenize_stems_index_and_query_identically(self): + from tools.tool_search import _tokenize + # Same stem on both sides is the whole contract. + assert _tokenize("issues") == _tokenize("issue") + assert _tokenize("creating messages") == _tokenize("create message") + + def test_plural_query_finds_singular_tool_name(self, issue_defs): + """The measured miss on the old tokenizer: 'issues' skipped create_issue.""" + from tools.tool_search import build_catalog, search_catalog + + catalog = build_catalog(issue_defs) + names = [h.name for h in search_catalog(catalog, "issues", limit=5)] + assert "mq_linear_create_issue" in names + assert "mq_linear_list_issues" in names + + def test_substring_fallback_still_uses_raw_name(self, issue_defs): + """Fallback matches the unstemmed tool name, unchanged by stemming.""" + from tools.tool_search import build_catalog, search_catalog + + catalog = build_catalog(issue_defs) + names = [h.name for h in search_catalog(catalog, "post_mess", limit=5)] + assert names == ["mq_slack_post_message"] + + +# --------------------------------------------------------------------------- +# Multi-query dispatch_tool_search +# --------------------------------------------------------------------------- + + +class TestMultiQuerySearch: + def test_grouped_names_plus_shared_tool_map(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_search + + result = json.loads(dispatch_tool_search( + {"queries": ["create linear issue", "post slack message"]}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + + assert result["queries"] == ["create linear issue", "post slack message"] + assert result["total_available"] == 3 + # Groups carry NAMES only, in query order. + assert [g["query"] for g in result["results"]] == result["queries"] + for group in result["results"]: + for name in group["matches"]: + assert isinstance(name, str) + assert "mq_linear_create_issue" in result["results"][0]["matches"] + assert "mq_slack_post_message" in result["results"][1]["matches"] + # The shared map holds each matched tool exactly once, and nothing else. + matched = {n for g in result["results"] for n in g["matches"]} + assert set(result["tools"]) == matched + record = result["tools"]["mq_linear_create_issue"] + assert record["source"] == "mcp" + assert record["source_name"] == "mcp-mq-linear" + assert record["description"].startswith("Create a new issue") + assert record["required"] == ["title", "team"] + # All queries matched → no fallback block. + assert "available_sources" not in result + assert "hint" not in result + + def test_limit_applies_per_query(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_search + + result = json.loads(dispatch_tool_search( + {"queries": ["issues", "message"], "limit": 1}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + for group in result["results"]: + assert len(group["matches"]) <= 1 + + def test_partial_miss_adds_single_toplevel_fallback(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_search + + result = json.loads(dispatch_tool_search( + {"queries": ["issues", "zzzz nonsense qqqq"]}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + assert result["results"][1]["matches"] == [] + # ONE fallback block at top level — not per empty group. + assert "available_sources" in result + assert "Some queries" in result["hint"] + for group in result["results"]: + assert "available_sources" not in group + assert "hint" not in group + source_names = {s["name"] for s in result["available_sources"]} + assert {"mq-linear", "mq-slack"} <= source_names + + def test_bare_string_query_coerced_to_single_query(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_search + + result = json.loads(dispatch_tool_search( + {"queries": "post slack message"}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + assert result["queries"] == ["post slack message"] + + def test_max_queries_config_respected(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_search + + cfg = ToolSearchConfig.from_raw({"max_queries": 2}) + ok = json.loads(dispatch_tool_search( + {"queries": ["a b", "c d"]}, current_tool_defs=issue_defs, config=cfg)) + assert "error" not in ok + over = json.loads(dispatch_tool_search( + {"queries": ["a", "b", "c"]}, current_tool_defs=issue_defs, config=cfg)) + assert "too many queries" in over["error"] + + +# --------------------------------------------------------------------------- +# Batched dispatch_tool_describe +# --------------------------------------------------------------------------- + + +class TestBatchedDescribe: + def test_map_response_with_not_found(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_describe + + # Deferrable in the global registry, but NOT in this session's defs — + # the stale/out-of-scope case that lands in not_found. + _register("mq_out_of_scope_op", "mcp-mq-elsewhere") + + result = json.loads(dispatch_tool_describe( + {"names": ["mq_linear_create_issue", "mq_slack_post_message", + "mq_out_of_scope_op", "mcp__bogus__missing"]}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + assert set(result["tools"]) == {"mq_linear_create_issue", + "mq_slack_post_message"} + schema = result["tools"]["mq_linear_create_issue"] + assert schema["description"] == "Create a new issue in a team." + assert schema["parameters"]["required"] == ["title", "team"] + # Deferrable-but-absent collects in not_found; found ones still resolve. + assert result["not_found"] == ["mq_out_of_scope_op"] + assert "tool_search" in result["hint"] + # A name unknown to the registry keeps the spelling-check error. + assert "not a deferrable tool" in result["errors"]["mcp__bogus__missing"] + + def test_non_deferrable_name_keeps_per_name_error(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_describe + + result = json.loads(dispatch_tool_describe( + {"names": ["terminal", "mq_linear_create_issue"]}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + assert "mq_linear_create_issue" in result["tools"] + assert "not a deferrable tool" in result["errors"]["terminal"] + + def test_duplicates_deduped_silently(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_describe + + result = json.loads(dispatch_tool_describe( + {"names": ["mq_linear_create_issue", "mq_linear_create_issue"]}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + assert list(result["tools"]) == ["mq_linear_create_issue"] + assert "not_found" not in result + + def test_empty_and_overcap_names_error(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_describe + + cfg = ToolSearchConfig.from_raw({}) + assert "error" in json.loads(dispatch_tool_describe( + {}, current_tool_defs=issue_defs, config=cfg)) + assert "error" in json.loads(dispatch_tool_describe( + {"names": []}, current_tool_defs=issue_defs, config=cfg)) + over = ["n%d" % i for i in range(cfg.max_describe_names + 1)] + parsed = json.loads(dispatch_tool_describe( + {"names": over}, current_tool_defs=issue_defs, config=cfg)) + assert "too many names" in parsed["error"] + + def test_bare_string_name_coerced(self, issue_defs): + from tools.tool_search import ToolSearchConfig, dispatch_tool_describe + + result = json.loads(dispatch_tool_describe( + {"names": "mq_linear_create_issue"}, + current_tool_defs=issue_defs, + config=ToolSearchConfig.from_raw({}), + )) + assert "mq_linear_create_issue" in result["tools"] + + +# --------------------------------------------------------------------------- +# Config + bridge schema +# --------------------------------------------------------------------------- + + +class TestConfigAndSchema: + def test_new_caps_default_and_parse(self): + from tools.tool_search import ToolSearchConfig + + cfg = ToolSearchConfig.from_raw(None) + assert cfg.max_queries >= 1 + assert cfg.max_describe_names >= 1 + # Operator-tunable with no upper clamp; floored at 1. + big = ToolSearchConfig.from_raw({"max_queries": 500, + "max_describe_names": 500}) + assert big.max_queries == 500 + assert big.max_describe_names == 500 + floored = ToolSearchConfig.from_raw({"max_queries": 0, + "max_describe_names": -3}) + assert floored.max_queries == 1 + assert floored.max_describe_names == 1 + + def test_limit_default_within_cap(self): + from tools.tool_search import ToolSearchConfig + + cfg = ToolSearchConfig.from_raw({}) + assert 1 <= cfg.search_default_limit <= cfg.max_search_limit <= 50 + + def test_bridge_schema_declares_array_inputs(self): + from tools.tool_search import bridge_tool_schemas + + schemas = {s["function"]["name"]: s["function"] for s in bridge_tool_schemas(3)} + search_params = schemas["tool_search"]["parameters"] + assert search_params["required"] == ["queries"] + assert search_params["properties"]["queries"]["type"] == "array" + describe_params = schemas["tool_describe"]["parameters"] + assert describe_params["required"] == ["names"] + assert describe_params["properties"]["names"]["type"] == "array" diff --git a/tools/tool_search.py b/tools/tool_search.py index 8559cb4d96..809ad5c016 100644 --- a/tools/tool_search.py +++ b/tools/tool_search.py @@ -46,9 +46,12 @@ import json import logging import math import re +import threading from dataclasses import dataclass, field from typing import Any, Dict, Iterable, List, Optional, Tuple +import snowballstemmer + from tools.registry import tool_error logger = logging.getLogger("tools.tool_search") @@ -90,6 +93,12 @@ class ToolSearchConfig: threshold_pct: float # 0..100 search_default_limit: int max_search_limit: int + # Per-call caps on the model-facing array inputs. ``max_queries`` bounds + # ``tool_search``'s ``queries`` list; ``max_describe_names`` bounds + # ``tool_describe``'s ``names`` list. Over-cap calls error (the model + # repairs in one round-trip) rather than silently truncating. + max_queries: int = 10 + max_describe_names: int = 10 # Catalog listing ("skills-style" progressive disclosure): when active, # a grouped name + short-description manifest of every deferred tool is # embedded in the tool_search bridge description, so capabilities stay @@ -114,13 +123,13 @@ class ToolSearchConfig: """ if raw is True: return cls(enabled="auto", threshold_pct=5.0, - search_default_limit=5, max_search_limit=20) + search_default_limit=5, max_search_limit=25) if raw is False: return cls(enabled="off", threshold_pct=5.0, - search_default_limit=5, max_search_limit=20) + search_default_limit=5, max_search_limit=25) if not isinstance(raw, dict): return cls(enabled="auto", threshold_pct=5.0, - search_default_limit=5, max_search_limit=20) + search_default_limit=5, max_search_limit=25) enabled_raw = str(raw.get("enabled", "auto")).strip().lower() if enabled_raw in ("true", "1", "yes"): @@ -135,10 +144,15 @@ class ToolSearchConfig: threshold_pct = _safe_float(raw.get("threshold_pct"), 5.0) threshold_pct = max(0.0, min(100.0, threshold_pct)) - max_search_limit = max(1, min(50, _safe_int(raw.get("max_search_limit"), 20))) + max_search_limit = max(1, min(50, _safe_int(raw.get("max_search_limit"), 25))) search_default_limit = max(1, min(max_search_limit, _safe_int(raw.get("search_default_limit"), 5))) + # Per-call array caps. Floored at 1 (a non-positive cap would reject + # every call); no upper clamp — deliberately operator-tunable. + max_queries = max(1, _safe_int(raw.get("max_queries"), 10)) + max_describe_names = max(1, _safe_int(raw.get("max_describe_names"), 10)) + listing_raw = str(raw.get("listing", "auto")).strip().lower() if listing_raw in ("true", "1", "yes"): listing = "on" @@ -155,6 +169,8 @@ class ToolSearchConfig: threshold_pct=threshold_pct, search_default_limit=search_default_limit, max_search_limit=max_search_limit, + max_queries=max_queries, + max_describe_names=max_describe_names, listing=listing, listing_max_tokens=listing_max_tokens, ) @@ -353,11 +369,32 @@ class CatalogEntry: _TOKEN_RE = re.compile(r"[A-Za-z0-9]+") +# Snowball stemmer instances keep mutable parsing state, so they are not +# safe to share across threads — and bridge dispatch can run on parallel +# tool-call threads. One stemmer per thread, created lazily. +_thread_local = threading.local() + + +def _stemmer() -> Any: + st = getattr(_thread_local, "stemmer", None) + if st is None: + st = snowballstemmer.stemmer("english") + _thread_local.stemmer = st + return st + def _tokenize(text: str) -> List[str]: + """Lowercase alphanumeric tokens, Snowball-stemmed (English). + + Stemming is applied here so it hits BOTH the index path + (:func:`build_catalog` via :func:`_entry_search_text`) and the query + path (:func:`search_catalog`) identically — a query for "issues" + matches a tool named ``create_issue``. + """ if not text: return [] - return [t.lower() for t in _TOKEN_RE.findall(text)] + tokens = [t.lower() for t in _TOKEN_RE.findall(text)] + return list(_stemmer().stemWords(tokens)) def _entry_search_text(td: Dict[str, Any], source_label: str = "") -> str: @@ -687,9 +724,12 @@ def bridge_tool_schemas( """ desc_search = ( f"Search {deferred_count} additional tools that are loaded on demand. " - "Returns up to ``limit`` matches with name and description. Follow " - f"with `{TOOL_DESCRIBE_NAME}` to load a tool's full parameter schema, " - f"then `{TOOL_CALL_NAME}` to invoke it. Tools listed at the top of this " + "Takes a list of queries searched in parallel against the same " + "catalog; send one query per distinct capability you need. Returns " + "matching tool names grouped per query plus a shared map with each " + "tool's description. Follow with " + f"`{TOOL_DESCRIBE_NAME}` to load full parameter schemas, " + f"then `{TOOL_CALL_NAME}` to invoke. Tools listed at the top of this " "system prompt are already available and do not need to be searched." ) if listing and listing_form == "groups": @@ -715,8 +755,9 @@ def bridge_tool_schemas( ) desc_search += "\n\n" + listing desc_describe = ( - f"Load the full JSON schema for one tool returned by `{TOOL_SEARCH_NAME}`. " - f"Required before `{TOOL_CALL_NAME}` if the tool's parameters are unknown." + f"Load the full JSON schemas for tools returned by `{TOOL_SEARCH_NAME}`. " + f"Required before `{TOOL_CALL_NAME}` if a tool's parameters are unknown. " + "Batch every schema you need into one call." ) desc_call = ( "Invoke a deferred tool by name with the given arguments. Argument shape " @@ -733,16 +774,17 @@ def bridge_tool_schemas( "parameters": { "type": "object", "properties": { - "query": { - "type": "string", - "description": "Keywords describing the capability you need (e.g. 'create github issue').", + "queries": { + "type": "array", + "items": {"type": "string"}, + "description": "Search queries, each a few keywords describing one capability (e.g. ['create github issue', 'send slack message']). Searched in parallel; results come back grouped per query.", }, "limit": { "type": "integer", - "description": "Maximum number of results to return. Default 5.", + "description": "Maximum number of matches PER QUERY. Default 5.", }, }, - "required": ["query"], + "required": ["queries"], }, }, }, @@ -754,12 +796,13 @@ def bridge_tool_schemas( "parameters": { "type": "object", "properties": { - "name": { - "type": "string", - "description": "Exact tool name (as returned by tool_search).", + "names": { + "type": "array", + "items": {"type": "string"}, + "description": "Exact tool names (as returned by tool_search).", }, }, - "required": ["name"], + "required": ["names"], }, }, }, @@ -891,13 +934,25 @@ def is_bridge_tool(name: str) -> bool: return name in BRIDGE_TOOL_NAMES -def _format_search_hit(entry: CatalogEntry) -> Dict[str, Any]: +def _shared_tool_record(entry: CatalogEntry) -> Dict[str, Any]: + """One record for the response's shared ``tools`` map. + + Held once per tool no matter how many query groups matched it — the + per-query groups carry names only. ``required`` lists the schema's + required parameter names so the model can attempt a call without a + ``tool_describe`` round-trip when the required surface is trivial. + """ + fn = (entry.schema or {}).get("function") or {} + params = fn.get("parameters") or {} + required = params.get("required") + if not isinstance(required, list): + required = [] return { - "name": entry.name, "source": entry.source, "source_name": entry.source_name, # Cap description so a chatty MCP server doesn't blow up the result. "description": (entry.description or "")[:400], + "required": [r for r in required if isinstance(r, str)], } @@ -925,12 +980,42 @@ def dispatch_tool_search(args: Dict[str, Any], *, current_tool_defs: List[Dict[str, Any]], config: Optional[ToolSearchConfig] = None) -> str: - """Execute the ``tool_search`` bridge tool. Returns a JSON string.""" + """Execute the ``tool_search`` bridge tool. Returns a JSON string. + + Accepts ``queries: [str, ...]`` — each query is searched independently + against the same catalog. The response groups matching tool NAMES per + query and carries each matched tool's record exactly once in a shared + ``tools`` map:: + + { + "queries": ["...", "..."], + "total_available": 215, + "results": [{"query": "...", "matches": ["", ...]}, ...], + "tools": {"": {"source": ..., "source_name": ..., + "description": ..., "required": [...]}} + } + + ``limit`` applies PER QUERY. When one or more queries return no matches, + a single top-level ``available_sources`` + ``hint`` block is added so a + lexical miss is not mistaken for a missing capability. + """ if config is None: config = load_config() - query = str(args.get("query") or "").strip() - if not query: - return tool_error("query is required") + + raw_queries = args.get("queries") + if isinstance(raw_queries, str): + # A bare string is an understandable model slip; treat as one query. + raw_queries = [raw_queries] + if not isinstance(raw_queries, list): + return tool_error("queries is required and must be an array of strings") + queries = [str(q).strip() for q in raw_queries if str(q or "").strip()] + if not queries: + return tool_error("queries is required and must contain at least one non-empty string") + if len(queries) > config.max_queries: + return tool_error( + f"too many queries: {len(queries)} > max {config.max_queries}. " + "Retry with fewer, more targeted queries." + ) raw_limit = args.get("limit") if raw_limit is None: @@ -940,47 +1025,108 @@ def dispatch_tool_search(args: Dict[str, Any], _, deferrable = classify_tools(current_tool_defs) catalog = build_catalog(deferrable) - hits = search_catalog(catalog, query, limit=limit) + + results: List[Dict[str, Any]] = [] + tools_map: Dict[str, Dict[str, Any]] = {} + any_empty = False + for query in queries: + hits = search_catalog(catalog, query, limit=limit) + if not hits: + any_empty = True + for h in hits: + if h.name not in tools_map: + tools_map[h.name] = _shared_tool_record(h) + results.append({"query": query, "matches": [h.name for h in hits]}) + result: Dict[str, Any] = { - "query": query, + "queries": queries, "total_available": len(catalog), - "matches": [_format_search_hit(h) for h in hits], + "results": results, + "tools": tools_map, } - if not hits and catalog: + if any_empty and catalog: result["available_sources"] = _available_source_summary(catalog) result["hint"] = ( - "No lexical match was found, but the sources above are connected " - "and their tools remain available. Retry tool_search with the " - "service name plus a concrete action or object before concluding " - "the capability is unavailable." + "Some queries returned no lexical matches, but the sources above " + "are connected and their tools remain available. Retry " + "tool_search with the service name plus a concrete action or " + "object before concluding the capability is unavailable." ) return json.dumps(result, ensure_ascii=False) def dispatch_tool_describe(args: Dict[str, Any], *, - current_tool_defs: List[Dict[str, Any]]) -> str: - """Execute the ``tool_describe`` bridge tool. Returns a JSON string.""" - name = str(args.get("name") or "").strip() - if not name: - return tool_error("name is required") - if not is_deferrable_tool_name(name): + current_tool_defs: List[Dict[str, Any]], + config: Optional[ToolSearchConfig] = None) -> str: + """Execute the ``tool_describe`` bridge tool. Returns a JSON string. + + Accepts ``names: [str, ...]`` and returns a map keyed by tool name:: + + { + "tools": {"": {"description": ..., "parameters": {...}}, ...}, + "not_found": ["", ...], # only when some names missed + "errors": {"": "..."} # only for non-deferrable names + } + + Unknown names land in ``not_found`` instead of failing the whole call; + non-deferrable names (core tools, typos of visible tools) keep their + per-name error message in ``errors``. Duplicates are deduped silently. + """ + if config is None: + config = load_config() + + raw_names = args.get("names") + if isinstance(raw_names, str): + # A bare string is an understandable model slip; treat as one name. + raw_names = [raw_names] + if not isinstance(raw_names, list): + return tool_error("names is required and must be an array of strings") + names: List[str] = [] + for n in raw_names: + n = str(n or "").strip() + if n and n not in names: + names.append(n) + if not names: + return tool_error("names is required and must contain at least one non-empty string") + if len(names) > config.max_describe_names: return tool_error( - f"'{name}' is not a deferrable tool. If you see it in the tools list " - "already, call it directly; otherwise check the spelling against tool_search." + f"too many names: {len(names)} > max {config.max_describe_names}. " + "Retry with fewer names per call." ) + _, deferrable = classify_tools(current_tool_defs) + by_name: Dict[str, Dict[str, Any]] = {} for td in deferrable: fn = td.get("function") or {} - if fn.get("name") == name: - return json.dumps({ - "name": name, + if fn.get("name"): + by_name[fn["name"]] = fn + + tools: Dict[str, Dict[str, Any]] = {} + not_found: List[str] = [] + errors: Dict[str, str] = {} + for name in names: + fn = by_name.get(name) + if fn is not None: + tools[name] = { "description": fn.get("description", ""), "parameters": fn.get("parameters", {}), - }, ensure_ascii=False) - return tool_error( - f"'{name}' is not currently available. Re-run tool_search to refresh." - ) + } + elif not is_deferrable_tool_name(name): + errors[name] = ( + f"'{name}' is not a deferrable tool. If you see it in the tools list " + "already, call it directly; otherwise check the spelling against tool_search." + ) + else: + not_found.append(name) + + result: Dict[str, Any] = {"tools": tools} + if not_found: + result["not_found"] = not_found + result["hint"] = "Names in not_found are not currently available. Re-run tool_search to refresh." + if errors: + result["errors"] = errors + return json.dumps(result, ensure_ascii=False) def scoped_deferrable_names(tool_defs: List[Dict[str, Any]]) -> frozenset[str]: diff --git a/uv.lock b/uv.lock index 0b058b8e70..6cb9011ec9 100644 --- a/uv.lock +++ b/uv.lock @@ -1599,6 +1599,7 @@ dependencies = [ { name = "requests" }, { name = "rich" }, { name = "ruamel-yaml" }, + { name = "snowballstemmer" }, { name = "tenacity" }, { name = "tzdata", marker = "sys_platform == 'win32'" }, { name = "urllib3" }, @@ -1929,6 +1930,7 @@ requires-dist = [ { name = "slack-bolt", marker = "extra == 'slack'", specifier = "==1.30.0" }, { name = "slack-sdk", marker = "extra == 'messaging'", specifier = "==3.43.0" }, { name = "slack-sdk", marker = "extra == 'slack'", specifier = "==3.43.0" }, + { name = "snowballstemmer", specifier = "==3.1.1" }, { name = "sounddevice", marker = "extra == 'voice'", specifier = "==0.5.5" }, { name = "sounddevice", marker = "extra == 'wake'", specifier = "==0.5.5" }, { name = "starlette", marker = "extra == 'computer-use'", specifier = "==1.3.1" }, @@ -4256,6 +4258,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2", size = 10235, upload-time = "2024-02-25T23:20:01.196Z" }, ] +[[package]] +name = "snowballstemmer" +version = "3.1.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/43/f8/0a71edf031f03c40db17503cb8ca78a69a171254e568e7db241b0ab57ea1/snowballstemmer-3.1.1.tar.gz", hash = "sha256:e07bbc54a0d798fe6010a12398422e62a8bfbba95c394fd0956ef58cb4d3e260", size = 123314, upload-time = "2026-06-03T00:56:40.194Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/4c/07/2ebca9b11fb9be7340a818d8d6f63feaebb146be2c4afbd6061701d6df6e/snowballstemmer-3.1.1-py3-none-any.whl", hash = "sha256:7e207fa178741da09cdee59d3ecec3827ad5f92b1fc5c9ff3755b639f71f5752", size = 104164, upload-time = "2026-06-03T00:56:38.614Z" }, +] + [[package]] name = "socksio" version = "1.0.0" diff --git a/website/docs/user-guide/features/tool-search.md b/website/docs/user-guide/features/tool-search.md index 09fff56516..eef14d25b7 100644 --- a/website/docs/user-guide/features/tool-search.md +++ b/website/docs/user-guide/features/tool-search.md @@ -30,22 +30,39 @@ When Tool Search activates for a turn, the model sees three new tools in place of the deferred ones: ``` -tool_search(query, limit?) — search the deferred-tool catalog -tool_describe(name) — load the full schema for one tool +tool_search(queries, limit?) — search the deferred-tool catalog (one or more queries) +tool_describe(names) — load the full schemas for one or more tools tool_call(name, arguments) — invoke a deferred tool ``` A typical interaction looks like: ``` -Model: tool_search("create a github issue") - → { matches: [{ name: "mcp_github_create_issue", ... }, ...] } -Model: tool_describe("mcp_github_create_issue") - → { parameters: { type: "object", properties: { ... } } } +Model: tool_search(["create a github issue", "send a slack message"]) + → { results: [ { query: "create a github issue", + matches: ["mcp_github_create_issue", ...] }, + { query: "send a slack message", + matches: ["mcp_slack_post_message", ...] } ], + tools: { mcp_github_create_issue: { description: "...", + required: ["title"], ... }, + mcp_slack_post_message: { ... } } } +Model: tool_describe(["mcp_github_create_issue", "mcp_slack_post_message"]) + → { tools: { mcp_github_create_issue: { parameters: { ... } }, + mcp_slack_post_message: { parameters: { ... } } } } Model: tool_call("mcp_github_create_issue", { title: "...", body: "..." }) → { ok: true, issue_number: 42 } ``` +Each query in a `tool_search` call is searched independently against the +same catalog (`limit` applies per query); the per-query groups carry tool +names only, while the shared `tools` map holds each matched tool's +description and required parameter names once. Queries are stemmed, so +"issues" finds `create_issue`. When some queries return no matches, the +response includes a single `available_sources` summary of the connected +servers so a lexical miss is not mistaken for a missing capability. +`tool_describe` resolves every requested name in one call; unknown names +are reported in `not_found` without failing the rest of the batch. + When the model invokes `tool_call`, Hermes **unwraps the bridge** and dispatches the underlying tool exactly as if the model had called it directly. Pre-tool-call hooks, guardrails, approval prompts, and @@ -78,7 +95,9 @@ tools: enabled: auto # auto (default), on, or off threshold_pct: 5 # listing budget as a percentage of context search_default_limit: 5 - max_search_limit: 20 + max_search_limit: 25 + max_queries: 10 + max_describe_names: 10 listing: auto # embed a grouped name+description catalog manifest listing_max_tokens: 4000 ``` @@ -87,8 +106,10 @@ tools: | --- | --- | --- | | `enabled` | `auto` | `auto`/`on` activate whenever at least one deferrable tool exists; `off` disables entirely (everything stays eager). `auto` is currently an alias of `on` — it is reserved for a future mode that inlines schemas when they fit the context and defers only when they don't. Pin `on` or `off` if you want today's behavior guaranteed across upgrades. | | `threshold_pct` | `5` | Listing budget as a percentage of the active model's context length. Range 0–100. | -| `search_default_limit` | `5` | Hits returned when the model calls `tool_search` without a `limit`. | -| `max_search_limit` | `20` | Hard upper bound the model can request via `limit`. Range 1–50. | +| `search_default_limit` | `5` | Hits returned per query when the model calls `tool_search` without a `limit`. | +| `max_search_limit` | `25` | Hard upper bound the model can request via `limit` (per query). Range 1–50. | +| `max_queries` | `10` | Maximum queries per `tool_search` call. Over-cap calls return an error and the model retries with fewer. No upper clamp. | +| `max_describe_names` | `10` | Maximum names per `tool_describe` call. Same error-and-retry semantics. No upper clamp. | | `listing` | `auto` | Embed a skills-style manifest of every deferred tool (name + first sentence of its description, ≤60 chars, grouped by MCP server) in the `tool_search` bridge description. `auto` includes it when it fits the budget (falling back to names-only, then to the tier-2 server summary); `on`/`off` force either way. | | `listing_max_tokens` | `4000` | Absolute cap on the embedded listing, regardless of context size. Range 200–60000. Large catalogs degrade to names-only or per-server summaries, keeping full schemas available through search. | @@ -151,9 +172,12 @@ to any progressive-disclosure design, not specific to this implementation: - **Retrieval:** BM25 over tokenized tool name, source name (the MCP server or plugin toolset the tool belongs to, so searching `"linear"` finds that server's tools even when a tool's own name doesn't carry - the service), description, and parameter names. Falls back to a - literal substring match on the tool name when no query token matches - any document (e.g. searching `"hub"` where the token is `github`). + the service), description, and parameter names, with Snowball + stemming (English) applied to both the index and the query so + morphological variants match ("issues" finds `create_issue`). Falls + back to a literal substring match on the tool name when no query + token matches any document (e.g. searching `"hub"` where the token is + `github`). - **Parallel execution unwraps the bridge.** The batch planner decides concurrency on the *underlying* tool of a `tool_call`, not on the literal bridge name — so an MCP server opted in via