feat(tool_search): long hunts for nonexistent tools now return no results instead of incidental matches
Port from nearai/ironclaw#7965: BM25 admits any document scoring above zero, i.e. sharing ONE term with the query. A long descriptive search for a capability that does not exist therefore returned a plausible- looking ranked list, and the model read 'results exist' as 'it is in here somewhere' and rephrased instead of stopping (IronClaw production trace: 652 tool calls, 216 of them tool_search, hunting a 'data' tool that did not exist). A document must now match at least half the query's ANSWERABLE terms (terms present anywhere in the index) before it is offered. Coverage only engages from four answerable terms up, preserving recall on short queries; exact tool-name matches remain authoritative; the substring fallback is unchanged. Docs: relevance-floor bullet added to tool-search.md implementation details.
This commit is contained in:
@@ -290,6 +290,64 @@ class TestRetrieval:
|
||||
assert len(hits) <= 1
|
||||
|
||||
|
||||
class TestRelevanceFloor:
|
||||
"""Coverage floor layered on the rarest-token gate.
|
||||
|
||||
The gate stops a query whose intent word no tool carries. It does not stop a long
|
||||
hunt whose every word exists SOMEWHERE in the catalog while no single tool carries
|
||||
more than one of them; those must return nothing rather than a plausible-looking
|
||||
list the model rephrases against forever.
|
||||
"""
|
||||
|
||||
def _catalog(self):
|
||||
from tools.tool_search import build_catalog
|
||||
defs = [
|
||||
_td("github_rerun_failed_workflow_run_jobs",
|
||||
"Re-run failed jobs in a workflow run",
|
||||
{"run_id": {"type": "string"}}),
|
||||
_td("github_create_issue", "Open a new issue in a GitHub repository",
|
||||
{"title": {"type": "string"}, "body": {"type": "string"}}),
|
||||
_td("github_list_issues", "List issues in a repository",
|
||||
{"repo": {"type": "string"}}),
|
||||
_td("slack_send_message", "Post a message into a Slack channel",
|
||||
{"channel": {"type": "string"}, "text": {"type": "string"}}),
|
||||
# Every hunt word below is answerable by SOME document, none by one
|
||||
# document — the production catalog shape behind the 216-search trace.
|
||||
_td("gist_save_snippet", "Save a shell command snippet as a gist",
|
||||
{"content": {"type": "string"}}),
|
||||
_td("codeql_scan", "Scan code for vulnerabilities and execute analysis",
|
||||
{"repo": {"type": "string"}}),
|
||||
]
|
||||
return build_catalog(defs)
|
||||
|
||||
def test_incidental_single_term_match_is_filtered(self):
|
||||
# Every term answerable (each in exactly one document), so the rarest-token
|
||||
# gate admits the tool sharing that one word; the floor must not.
|
||||
from tools.tool_search import search_catalog
|
||||
hits = search_catalog(self._catalog(), "run shell command execute code", limit=5)
|
||||
assert hits == []
|
||||
|
||||
def test_short_queries_are_untouched(self):
|
||||
# Below 4 answerable terms wording legitimately differs by a word.
|
||||
from tools.tool_search import search_catalog
|
||||
hits = search_catalog(self._catalog(), "list issues", limit=5)
|
||||
assert any(h.name == "github_list_issues" for h in hits)
|
||||
hits = search_catalog(self._catalog(), "send message", limit=5)
|
||||
assert any(h.name == "slack_send_message" for h in hits)
|
||||
|
||||
def test_long_query_with_real_coverage_still_matches(self):
|
||||
from tools.tool_search import search_catalog
|
||||
hits = search_catalog(
|
||||
self._catalog(), "create issue github repository title", limit=5)
|
||||
assert hits and hits[0].name == "github_create_issue"
|
||||
assert all(h.name != "github_rerun_failed_workflow_run_jobs" for h in hits)
|
||||
|
||||
def test_exact_name_match_bypasses_coverage(self):
|
||||
from tools.tool_search import search_catalog
|
||||
hits = search_catalog(self._catalog(), "github_create_issue", limit=5)
|
||||
assert hits and hits[0].name == "github_create_issue"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Assembly — the full passthrough/activate decision.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -154,6 +154,26 @@ def _gate_token(query_tokens: List[str], doc_freq: Dict[str, int], n_docs: int)
|
||||
return max(query_tokens, key=_idf)
|
||||
|
||||
|
||||
# Relevance floor. The rarest-token gate stops queries whose intent word no tool carries; it
|
||||
# does not stop a long hunt whose every word exists SOMEWHERE in the catalog while no single
|
||||
# tool carries more than one of them (observed: "run shell command execute code python" -> a
|
||||
# workflow-rerun tool sharing only "run"; the model read "results exist" as "it is in here"
|
||||
# and re-searched 216 times). A document must also match MIN_QUERY_TERM_COVERAGE of the
|
||||
# query's ANSWERABLE terms (present in at least one document) before it is offered. Coverage
|
||||
# only engages from MIN_ANSWERABLE_TERMS_FOR_COVERAGE terms up: short queries ("list issues")
|
||||
# legitimately differ from a tool by a word, and are where a coverage rule costs real recall.
|
||||
MIN_QUERY_TERM_COVERAGE = 0.5
|
||||
MIN_ANSWERABLE_TERMS_FOR_COVERAGE = 4
|
||||
|
||||
|
||||
def _required_term_coverage(answerable_term_count: int) -> int:
|
||||
"""How many of a query's ANSWERABLE unique terms a document must match to be offered:
|
||||
one below the engagement floor, else at least half, rounded up."""
|
||||
if answerable_term_count < MIN_ANSWERABLE_TERMS_FOR_COVERAGE:
|
||||
return 1
|
||||
return math.ceil(answerable_term_count * MIN_QUERY_TERM_COVERAGE)
|
||||
|
||||
|
||||
def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *,
|
||||
corpus_stats: Optional[_CorpusStats] = None) -> List[CatalogEntry]:
|
||||
"""Top-``limit`` catalog entries for ``query`` by BM25 (exact name match ranks first).
|
||||
@@ -162,18 +182,29 @@ def search_catalog(catalog: List[CatalogEntry], query: str, limit: int = 5, *,
|
||||
BM25 is additive over the tokens a document shares with the query, so on a large catalog
|
||||
``score > 0`` admits one-token matches and fills every slot with them (measured: "send
|
||||
gmail email" returned 5 incident tools that only shared ``email``). A token no document
|
||||
carries admits nothing; the caller's empty-group hint tells the model to retry without it."""
|
||||
carries admits nothing; the caller's empty-group hint tells the model to retry without it.
|
||||
Long queries additionally need :func:`_required_term_coverage` of their answerable terms."""
|
||||
query_tokens = _tokenize(query) if catalog and limit > 0 else []
|
||||
if not query_tokens:
|
||||
return []
|
||||
corpus_stats = corpus_stats or _corpus_stats(catalog)
|
||||
gate = _gate_token(query_tokens, corpus_stats[2], corpus_stats[3])
|
||||
doc_freq = corpus_stats[2]
|
||||
gate = _gate_token(query_tokens, doc_freq, corpus_stats[3])
|
||||
answerable = {t for t in query_tokens if doc_freq.get(t, 0) > 0}
|
||||
required_terms = _required_term_coverage(len(answerable))
|
||||
exact_name = query.strip().lower()
|
||||
|
||||
def _admitted(entry: CatalogEntry) -> bool:
|
||||
"""Carries the intent word AND enough of the answerable terms (exact name is exempt)."""
|
||||
if entry.name.lower() == exact_name:
|
||||
return True
|
||||
tokens = set(entry._tokens)
|
||||
return gate in tokens and sum(1 for t in answerable if t in tokens) >= required_terms
|
||||
|
||||
scored = [
|
||||
(float("inf") if entry.name.lower() == exact_name
|
||||
else _bm25_score(query_tokens, entry._tokens, *corpus_stats), entry)
|
||||
for entry in catalog
|
||||
if entry.name.lower() == exact_name or gate in entry._tokens]
|
||||
for entry in catalog if _admitted(entry)]
|
||||
scored.sort(key=lambda x: x[0], reverse=True)
|
||||
return [e for _, e in scored[:limit]]
|
||||
|
||||
|
||||
@@ -236,6 +236,14 @@ to any progressive-disclosure design, not specific to this implementation:
|
||||
token appears in no tool returns an empty group with the connected
|
||||
sources and a retry hint, instead of `limit` tools that share one
|
||||
common word.
|
||||
- **Relevance floor:** a tool must match at least half of a query's
|
||||
*answerable* terms (terms present anywhere in the catalog) before it
|
||||
is offered — sharing one incidental word with a long query is not a
|
||||
match. A hunt for a capability that doesn't exist returns no results
|
||||
instead of a plausible-looking list the model rephrases against
|
||||
forever. The floor only engages from four answerable terms up, so
|
||||
short queries like "list issues" keep full recall, and an exact
|
||||
tool-name query always matches.
|
||||
- **Parallel execution unwraps the bridge.** The batch planner decides
|
||||
concurrency on the *underlying* tool of a `tool_call`, not on the
|
||||
literal bridge name — so an MCP server opted in via
|
||||
|
||||
Reference in New Issue
Block a user