diff --git a/tests/tools/test_tool_search.py b/tests/tools/test_tool_search.py index 49c686410d..4c6725659d 100644 --- a/tests/tools/test_tool_search.py +++ b/tests/tools/test_tool_search.py @@ -290,6 +290,64 @@ class TestRetrieval: assert len(hits) <= 1 +class TestRelevanceFloor: + """Coverage floor layered on the rarest-token gate. + + The gate stops a query whose intent word no tool carries. It does not stop a long + hunt whose every word exists SOMEWHERE in the catalog while no single tool carries + more than one of them; those must return nothing rather than a plausible-looking + list the model rephrases against forever. + """ + + def _catalog(self): + from tools.tool_search import build_catalog + defs = [ + _td("github_rerun_failed_workflow_run_jobs", + "Re-run failed jobs in a workflow run", + {"run_id": {"type": "string"}}), + _td("github_create_issue", "Open a new issue in a GitHub repository", + {"title": {"type": "string"}, "body": {"type": "string"}}), + _td("github_list_issues", "List issues in a repository", + {"repo": {"type": "string"}}), + _td("slack_send_message", "Post a message into a Slack channel", + {"channel": {"type": "string"}, "text": {"type": "string"}}), + # Every hunt word below is answerable by SOME document, none by one + # document — the production catalog shape behind the 216-search trace. + _td("gist_save_snippet", "Save a shell command snippet as a gist", + {"content": {"type": "string"}}), + _td("codeql_scan", "Scan code for vulnerabilities and execute analysis", + {"repo": {"type": "string"}}), + ] + return build_catalog(defs) + + def test_incidental_single_term_match_is_filtered(self): + # Every term answerable (each in exactly one document), so the rarest-token + # gate admits the tool sharing that one word; the floor must not. + from tools.tool_search import search_catalog + hits = search_catalog(self._catalog(), "run shell command execute code", limit=5) + assert hits == [] + + def test_short_queries_are_untouched(self): + # Below 4 answerable terms wording legitimately differs by a word. + from tools.tool_search import search_catalog + hits = search_catalog(self._catalog(), "list issues", limit=5) + assert any(h.name == "github_list_issues" for h in hits) + hits = search_catalog(self._catalog(), "send message", limit=5) + assert any(h.name == "slack_send_message" for h in hits) + + def test_long_query_with_real_coverage_still_matches(self): + from tools.tool_search import search_catalog + hits = search_catalog( + self._catalog(), "create issue github repository title", limit=5) + assert hits and hits[0].name == "github_create_issue" + assert all(h.name != "github_rerun_failed_workflow_run_jobs" for h in hits) + + def test_exact_name_match_bypasses_coverage(self): + from tools.tool_search import search_catalog + hits = search_catalog(self._catalog(), "github_create_issue", limit=5) + assert hits and hits[0].name == "github_create_issue" + + # --------------------------------------------------------------------------- # Assembly — the full passthrough/activate decision. # --------------------------------------------------------------------------- diff --git a/tools/tool_search_catalog.py b/tools/tool_search_catalog.py index 007e7571db..18cc370146 100644 --- a/tools/tool_search_catalog.py +++ b/tools/tool_search_catalog.py @@ -154,6 +154,26 @@ def _gate_token(query_tokens: List[str], doc_freq: Dict[str, int], n_docs: int) return max(query_tokens, key=_idf) +# Relevance floor. The rarest-token gate stops queries whose intent word no tool carries; it +# does not stop a long hunt whose every word exists SOMEWHERE in the catalog while no single +# tool carries more than one of them (observed: "run shell command execute code python" -> a +# workflow-rerun tool sharing only "run"; the model read "results exist" as "it is in here" +# and re-searched 216 times). A document must also match MIN_QUERY_TERM_COVERAGE of the +# query's ANSWERABLE terms (present in at least one document) before it is offered. Coverage +# only engages from MIN_ANSWERABLE_TERMS_FOR_COVERAGE terms up: short queries ("list issues") +# legitimately differ from a tool by a word, and are where a coverage rule costs real recall. +MIN_QUERY_TERM_COVERAGE = 0.5 +MIN_ANSWERABLE_TERMS_FOR_COVERAGE = 4 + + +def _required_term_coverage(answerable_term_count: int) -> int: + """How many of a query's ANSWERABLE unique terms a document must match to be offered: + one below the engagement floor, else at least half, rounded up.""" + if answerable_term_count < MIN_ANSWERABLE_TERMS_FOR_COVERAGE: + return 1 + return math.ceil(answerable_term_count * MIN_QUERY_TERM_COVERAGE) + + def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *, corpus_stats: Optional[_CorpusStats] = None) -> List[CatalogEntry]: """Top-``limit`` catalog entries for ``query`` by BM25 (exact name match ranks first). @@ -162,18 +182,29 @@ def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *, BM25 is additive over the tokens a document shares with the query, so on a large catalog ``score > 0`` admits one-token matches and fills every slot with them (measured: "send gmail email" returned 5 incident tools that only shared ``email``). A token no document - carries admits nothing; the caller's empty-group hint tells the model to retry without it.""" + carries admits nothing; the caller's empty-group hint tells the model to retry without it. + Long queries additionally need :func:`_required_term_coverage` of their answerable terms.""" query_tokens = _tokenize(query) if catalog and limit > 0 else [] if not query_tokens: return [] corpus_stats = corpus_stats or _corpus_stats(catalog) - gate = _gate_token(query_tokens, corpus_stats[2], corpus_stats[3]) + doc_freq = corpus_stats[2] + gate = _gate_token(query_tokens, doc_freq, corpus_stats[3]) + answerable = {t for t in query_tokens if doc_freq.get(t, 0) > 0} + required_terms = _required_term_coverage(len(answerable)) exact_name = query.strip().lower() + + def _admitted(entry: CatalogEntry) -> bool: + """Carries the intent word AND enough of the answerable terms (exact name is exempt).""" + if entry.name.lower() == exact_name: + return True + tokens = set(entry._tokens) + return gate in tokens and sum(1 for t in answerable if t in tokens) >= required_terms + scored = [ (float("inf") if entry.name.lower() == exact_name else _bm25_score(query_tokens, entry._tokens, *corpus_stats), entry) - for entry in catalog - if entry.name.lower() == exact_name or gate in entry._tokens] + for entry in catalog if _admitted(entry)] scored.sort(key=lambda x: x[0], reverse=True) return [e for _, e in scored[:limit]] diff --git a/website/docs/user-guide/features/tool-search.md b/website/docs/user-guide/features/tool-search.md index ead7b6c5a7..2c6ac08e38 100644 --- a/website/docs/user-guide/features/tool-search.md +++ b/website/docs/user-guide/features/tool-search.md @@ -236,6 +236,14 @@ to any progressive-disclosure design, not specific to this implementation: token appears in no tool returns an empty group with the connected sources and a retry hint, instead of `limit` tools that share one common word. +- **Relevance floor:** a tool must match at least half of a query's + *answerable* terms (terms present anywhere in the catalog) before it + is offered — sharing one incidental word with a long query is not a + match. A hunt for a capability that doesn't exist returns no results + instead of a plausible-looking list the model rephrases against + forever. The floor only engages from four answerable terms up, so + short queries like "list issues" keep full recall, and an exact + tool-name query always matches. - **Parallel execution unwraps the bridge.** The batch planner decides concurrency on the *underlying* tool of a `tool_call`, not on the literal bridge name — so an MCP server opted in via