feat: MCP tool results spill at 50K and carry upstream-elision warnings
Composio-style MCP servers return un-paginated 22-47K-char payloads that
sail under the generic 100K per-result spillover threshold, bloating
context and ballooning per-turn reasoning time on long conversations.
Competitors cap harder (OpenCode/pi 50KB, Claude Code 30K, Codex ~10K
tokens). Three changes:
- mcp_* tools spill at a tighter 50K default (BudgetConfig.mcp_result_size,
config-overridable via tool_budget.mcp_result_size_chars; pinned and
per-tool overrides still win; capped by the context-scaled default).
- The persisted-output preview now teaches recovery: page the saved file
with read_file or process with execute_code instead of re-requesting the
same data from the remote API.
- Untrusted/MCP string results are scanned (bounded, first 64KB) for
provider-side elision markers ('...N more items', "has_more": true,
'saved to sandbox', data_preview) and get ONE cache-safe incompleteness
notice appended at result-construction time, before untrusted wrapping —
so the model stops treating provider-elided enumerations as complete.
- Hard 2M-char allocation cap in mcp_tool.py (text, error, and
structuredContent paths) so a pathological multi-MB server payload is
bounded before it propagates, while ordinary large results reach
spillover intact. Distilled from #56060/#56072/#56511 (issue #56059);
supersedes their 50K lossy truncation with spillover-friendly semantics.
Docs: configuration.md spillover-budget section + cli-config.yaml.example.
Co-authored-by: Stoltemberg <215755014+Stoltemberg@users.noreply.github.com>
Co-authored-by: AlexFucuson9 <295703459+AlexFucuson9@users.noreply.github.com>
Co-authored-by: Tranquil-Flow <66773372+Tranquil-Flow@users.noreply.github.com>
This commit is contained in:
@@ -557,7 +557,11 @@ def make_tool_result_message(
|
||||
The outer list itself is rebuilt rather than returned by identity, so
|
||||
callers should compare by value, not by ``is``.
|
||||
"""
|
||||
wrapped = _maybe_wrap_untrusted(name, content)
|
||||
# Order matters: detect provider-side elision on the RAW content and
|
||||
# append the notice first, THEN wrap — so the notice lives inside the
|
||||
# untrusted block next to the data it describes, appended exactly once
|
||||
# at construction time (cache-safe).
|
||||
wrapped = _maybe_wrap_untrusted(name, _maybe_append_elision_notice(name, content))
|
||||
message = stamp_message_timestamp({
|
||||
"role": "tool",
|
||||
"name": name,
|
||||
@@ -608,6 +612,70 @@ def _is_untrusted_tool(name: Optional[str]) -> bool:
|
||||
return any(name.startswith(p) for p in _UNTRUSTED_TOOL_PREFIXES)
|
||||
|
||||
|
||||
# --- Upstream-elision detection --------------------------------------------
|
||||
#
|
||||
# Some MCP servers elide data SERVER-SIDE and mark the elision inside the
|
||||
# payload itself (e.g. Composio: '...13 more items' inside a JSON array,
|
||||
# '"has_more": true', 'Complete response was large (N tokens). Full data
|
||||
# saved to sandbox in /mnt/files/...', 'data_preview' envelopes). Because the
|
||||
# result looks structurally complete, models treat the visible slice as the
|
||||
# whole dataset and falsely claim completeness. When one of these markers is
|
||||
# present, we append ONE compact notice at result-construction time — before
|
||||
# the message enters history, never mutated later, so prompt caching is safe.
|
||||
|
||||
# Conservative patterns only: each one is an explicit provider-side "there is
|
||||
# more data than what you can see" signal, not a generic truncation heuristic.
|
||||
_UPSTREAM_ELISION_PATTERNS = (
|
||||
re.compile(r"\.\.\.\s*\d+\s+more\s+items?", re.IGNORECASE),
|
||||
re.compile(r'"has_more"\s*:\s*true', re.IGNORECASE),
|
||||
re.compile(r"saved to sandbox", re.IGNORECASE),
|
||||
re.compile(r"data_preview", re.IGNORECASE),
|
||||
)
|
||||
|
||||
# Results smaller than this can't meaningfully hide an elided enumeration —
|
||||
# skip the scan entirely so tiny results pay nothing.
|
||||
_ELISION_SCAN_MIN_CHARS = 1_000
|
||||
|
||||
# Bound the regex scan: markers appear near the elided structure, which for
|
||||
# the payload sizes that matter (20-50K) is always inside the first 64KB.
|
||||
_ELISION_SCAN_MAX_CHARS = 65_536
|
||||
|
||||
_UPSTREAM_ELISION_NOTICE = (
|
||||
'\n[hermes note: this result contains provider-side elision markers '
|
||||
'(e.g. "...N more items" / has_more:true). The data shown is INCOMPLETE '
|
||||
'— page/fetch the remainder before treating any enumeration as complete.]'
|
||||
)
|
||||
|
||||
|
||||
def _detect_upstream_elision(content: Any) -> bool:
|
||||
"""True when a string tool result carries provider-side elision markers.
|
||||
|
||||
Cheap and safe by construction: non-string content is never scanned,
|
||||
results under ``_ELISION_SCAN_MIN_CHARS`` short-circuit, and the regex
|
||||
scan is capped at the first ``_ELISION_SCAN_MAX_CHARS`` chars.
|
||||
"""
|
||||
if not isinstance(content, str):
|
||||
return False
|
||||
if len(content) < _ELISION_SCAN_MIN_CHARS:
|
||||
return False
|
||||
window = content[:_ELISION_SCAN_MAX_CHARS]
|
||||
return any(p.search(window) for p in _UPSTREAM_ELISION_PATTERNS)
|
||||
|
||||
|
||||
def _maybe_append_elision_notice(name: str, content: Any) -> Any:
|
||||
"""Append the incompleteness notice to untrusted string results that
|
||||
embed upstream elision markers. Returns ``content`` unchanged otherwise.
|
||||
|
||||
Runs on the RAW result before untrusted-wrapping so the notice sits with
|
||||
the data it describes, and only at result-construction time (cache-safe).
|
||||
"""
|
||||
if not _is_untrusted_tool(name):
|
||||
return content
|
||||
if _detect_upstream_elision(content):
|
||||
return content + _UPSTREAM_ELISION_NOTICE
|
||||
return content
|
||||
|
||||
|
||||
def _tool_output_risk_metadata(name: str, content: Any) -> Optional[Dict[str, Any]]:
|
||||
"""Classify textual attacker-controlled output without retaining a copy.
|
||||
|
||||
@@ -729,5 +797,7 @@ __all__ = [
|
||||
"_extract_landed_file_mutation_paths",
|
||||
"_extract_error_preview",
|
||||
"_trajectory_normalize_msg",
|
||||
"_detect_upstream_elision",
|
||||
"_maybe_append_elision_notice",
|
||||
"make_tool_result_message",
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user