feat(tool_search): long hunts for nonexistent tools now return no results instead of incidental matches

Port from nearai/ironclaw#7965: BM25 admits any document scoring above
zero, i.e. sharing ONE term with the query. A long descriptive search
for a capability that does not exist therefore returned a plausible-
looking ranked list, and the model read 'results exist' as 'it is in
here somewhere' and rephrased instead of stopping (IronClaw production
trace: 652 tool calls, 216 of them tool_search, hunting a 'data' tool
that did not exist).

A document must now match at least half the query's ANSWERABLE terms
(terms present anywhere in the index) before it is offered. Coverage
only engages from four answerable terms up, preserving recall on short
queries; exact tool-name matches remain authoritative; the substring
fallback is unchanged.

Docs: relevance-floor bullet added to tool-search.md implementation
details.
This commit is contained in:
Teknium
2026-08-30 17:10:17 -07:00
parent ae596d805b
commit c63de5a231
3 changed files with 101 additions and 4 deletions

View File

@@ -290,6 +290,64 @@ class TestRetrieval:
assert len(hits) <= 1
class TestRelevanceFloor:
"""Coverage floor layered on the rarest-token gate.
The gate stops a query whose intent word no tool carries. It does not stop a long
hunt whose every word exists SOMEWHERE in the catalog while no single tool carries
more than one of them; those must return nothing rather than a plausible-looking
list the model rephrases against forever.
"""
def _catalog(self):
from tools.tool_search import build_catalog
defs = [
_td("github_rerun_failed_workflow_run_jobs",
"Re-run failed jobs in a workflow run",
{"run_id": {"type": "string"}}),
_td("github_create_issue", "Open a new issue in a GitHub repository",
{"title": {"type": "string"}, "body": {"type": "string"}}),
_td("github_list_issues", "List issues in a repository",
{"repo": {"type": "string"}}),
_td("slack_send_message", "Post a message into a Slack channel",
{"channel": {"type": "string"}, "text": {"type": "string"}}),
# Every hunt word below is answerable by SOME document, none by one
# document — the production catalog shape behind the 216-search trace.
_td("gist_save_snippet", "Save a shell command snippet as a gist",
{"content": {"type": "string"}}),
_td("codeql_scan", "Scan code for vulnerabilities and execute analysis",
{"repo": {"type": "string"}}),
]
return build_catalog(defs)
def test_incidental_single_term_match_is_filtered(self):
# Every term answerable (each in exactly one document), so the rarest-token
# gate admits the tool sharing that one word; the floor must not.
from tools.tool_search import search_catalog
hits = search_catalog(self._catalog(), "run shell command execute code", limit=5)
assert hits == []
def test_short_queries_are_untouched(self):
# Below 4 answerable terms wording legitimately differs by a word.
from tools.tool_search import search_catalog
hits = search_catalog(self._catalog(), "list issues", limit=5)
assert any(h.name == "github_list_issues" for h in hits)
hits = search_catalog(self._catalog(), "send message", limit=5)
assert any(h.name == "slack_send_message" for h in hits)
def test_long_query_with_real_coverage_still_matches(self):
from tools.tool_search import search_catalog
hits = search_catalog(
self._catalog(), "create issue github repository title", limit=5)
assert hits and hits[0].name == "github_create_issue"
assert all(h.name != "github_rerun_failed_workflow_run_jobs" for h in hits)
def test_exact_name_match_bypasses_coverage(self):
from tools.tool_search import search_catalog
hits = search_catalog(self._catalog(), "github_create_issue", limit=5)
assert hits and hits[0].name == "github_create_issue"
# ---------------------------------------------------------------------------
# Assembly — the full passthrough/activate decision.
# ---------------------------------------------------------------------------

View File

@@ -154,6 +154,26 @@ def _gate_token(query_tokens: List[str], doc_freq: Dict[str, int], n_docs: int)
return max(query_tokens, key=_idf)
# Relevance floor. The rarest-token gate stops queries whose intent word no tool carries; it
# does not stop a long hunt whose every word exists SOMEWHERE in the catalog while no single
# tool carries more than one of them (observed: "run shell command execute code python" -> a
# workflow-rerun tool sharing only "run"; the model read "results exist" as "it is in here"
# and re-searched 216 times). A document must also match MIN_QUERY_TERM_COVERAGE of the
# query's ANSWERABLE terms (present in at least one document) before it is offered. Coverage
# only engages from MIN_ANSWERABLE_TERMS_FOR_COVERAGE terms up: short queries ("list issues")
# legitimately differ from a tool by a word, and are where a coverage rule costs real recall.
MIN_QUERY_TERM_COVERAGE = 0.5
MIN_ANSWERABLE_TERMS_FOR_COVERAGE = 4
def _required_term_coverage(answerable_term_count: int) -> int:
"""How many of a query's ANSWERABLE unique terms a document must match to be offered:
one below the engagement floor, else at least half, rounded up."""
if answerable_term_count < MIN_ANSWERABLE_TERMS_FOR_COVERAGE:
return 1
return math.ceil(answerable_term_count * MIN_QUERY_TERM_COVERAGE)
def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *,
corpus_stats: Optional[_CorpusStats] = None) -> List[CatalogEntry]:
"""Top-``limit`` catalog entries for ``query`` by BM25 (exact name match ranks first).
@@ -162,18 +182,29 @@ def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *,
BM25 is additive over the tokens a document shares with the query, so on a large catalog
``score > 0`` admits one-token matches and fills every slot with them (measured: "send
gmail email" returned 5 incident tools that only shared ``email``). A token no document
carries admits nothing; the caller's empty-group hint tells the model to retry without it."""
carries admits nothing; the caller's empty-group hint tells the model to retry without it.
Long queries additionally need :func:`_required_term_coverage` of their answerable terms."""
query_tokens = _tokenize(query) if catalog and limit > 0 else []
if not query_tokens:
return []
corpus_stats = corpus_stats or _corpus_stats(catalog)
gate = _gate_token(query_tokens, corpus_stats[2], corpus_stats[3])
doc_freq = corpus_stats[2]
gate = _gate_token(query_tokens, doc_freq, corpus_stats[3])
answerable = {t for t in query_tokens if doc_freq.get(t, 0) > 0}
required_terms = _required_term_coverage(len(answerable))
exact_name = query.strip().lower()
def _admitted(entry: CatalogEntry) -> bool:
"""Carries the intent word AND enough of the answerable terms (exact name is exempt)."""
if entry.name.lower() == exact_name:
return True
tokens = set(entry._tokens)
return gate in tokens and sum(1 for t in answerable if t in tokens) >= required_terms
scored = [
(float("inf") if entry.name.lower() == exact_name
else _bm25_score(query_tokens, entry._tokens, *corpus_stats), entry)
for entry in catalog
if entry.name.lower() == exact_name or gate in entry._tokens]
for entry in catalog if _admitted(entry)]
scored.sort(key=lambda x: x[0], reverse=True)
return [e for _, e in scored[:limit]]

View File

@@ -236,6 +236,14 @@ to any progressive-disclosure design, not specific to this implementation:
token appears in no tool returns an empty group with the connected
sources and a retry hint, instead of `limit` tools that share one
common word.
- **Relevance floor:** a tool must match at least half of a query's
*answerable* terms (terms present anywhere in the catalog) before it
is offered — sharing one incidental word with a long query is not a
match. A hunt for a capability that doesn't exist returns no results
instead of a plausible-looking list the model rephrases against
forever. The floor only engages from four answerable terms up, so
short queries like "list issues" keep full recall, and an exact
tool-name query always matches.
- **Parallel execution unwraps the bridge.** The batch planner decides
concurrency on the *underlying* tool of a `tool_call`, not on the
literal bridge name — so an MCP server opted in via