refactor(tools): terminal, execute_code, MCP and the bounded collector truncate through one head/tail helper

Four copies of the 40/60 head/tail algorithm with a near-identical notice
(terminal_tool_result, mcp_tool_content, code_execution_tool,
environments/base_output) collapse into tools/tool_output_truncate.py, so the
ratio and the `... [<LABEL> TRUNCATED - N <unit> omitted out of T total] ...`
marker are defined once. execute_code keeps byte mode + spill path and only
shares the notice/split. Visible change: the terminal notice now uses
thousands separators like the other three (`9,000 chars` not `9000 chars`).

kanban_specify._truncate: comment claimed escape stripping the body never did;
comment now says what the plain clamp is for.
This commit is contained in:
teknium1
2026-09-12 19:42:52 -07:00
committed by Teknium
parent 3f59b5c594
commit 1512edfb85
5 changed files with 13 additions and 31 deletions

View File

@@ -79,8 +79,7 @@ class SpecifyOutcome:
def _truncate(text: str, limit: int) -> str:
# Stored history is untrusted for display — remove escape sequences and control chars so a recap line
# can't clear the screen / retitle the window when echoed to a terminal (openai/codex#31494 bug class).
# Plain length clamp for LLM prompt fields; these never reach a terminal, so no escape stripping here.
if len(text) <= limit:
return text
return text[: limit - 1] + "…"

View File

@@ -31,6 +31,7 @@ from tools.registry import registry, tool_error
from hermes_time import get_timezone_name
from tools.code_execution_env import _resolve_child_cwd, _resolve_child_python
from tools.code_execution_rpc import _rpc_poll_loop
from tools.tool_output_truncate import head_tail_split, truncation_notice
logger = logging.getLogger(__name__)
@@ -62,10 +63,10 @@ def _truncate_stdout_text(stdout_text: str) -> Tuple[str, Dict[str, Any]]:
"stdout_bytes_total": total, "stdout_bytes_omitted": total - captured}
if total <= MAX_STDOUT_BYTES:
return stdout_bytes.decode("utf-8", errors="replace"), metadata
head_bytes = int(MAX_STDOUT_BYTES * 0.4)
head_bytes, tail_bytes = head_tail_split(MAX_STDOUT_BYTES)
text = (stdout_bytes[:head_bytes].decode("utf-8", errors="replace")
+ f"\n\n... [OUTPUT TRUNCATED - {total - captured:,} bytes omitted out of {total:,} total] ...\n\n"
+ stdout_bytes[head_bytes - MAX_STDOUT_BYTES:].decode("utf-8", errors="replace"))
+ truncation_notice(total - captured, total, unit="bytes")
+ stdout_bytes[-tail_bytes:].decode("utf-8", errors="replace"))
metadata["warning"] = ("execute_code stdout was truncated; the script did run, but only "
"the captured head/tail output is included. Re-run only with "
"narrower output if the omitted data is required.")

View File

@@ -16,6 +16,7 @@ from pathlib import Path
from typing import IO, Callable, Protocol
from hermes_constants import get_hermes_home
from tools.tool_output_truncate import head_tail_split, truncation_notice
from hermes_cli._subprocess_compat import windows_hide_flags
# Sentinel capacity for full-fidelity capture: large enough that the collector
@@ -154,16 +155,13 @@ class _BoundedOutputCollector:
notice = ""
for _ in range(4):
omitted = max(0, self._total_chars - max(0, available - len(notice)))
updated = (
f"\n\n... [OUTPUT TRUNCATED - {omitted:,} chars omitted "
f"out of {self._total_chars:,} total] ...\n\n")
updated = truncation_notice(omitted, self._total_chars)
if updated == notice:
break
notice = updated
content_budget = max(0, available - len(notice))
head_chars = int(content_budget * 0.4)
tail_chars = content_budget - head_chars
head_chars, tail_chars = head_tail_split(content_budget)
rendered_tail = tail[-tail_chars:] if tail_chars else ""
return head[:head_chars] + notice[:available] + rendered_tail + suffix

View File

@@ -9,6 +9,7 @@ from typing import Any, Dict, Optional, Tuple
from tools.ansi_strip import strip_unicode_tags
from tools.mcp_tool_common import mcp_field
from tools.mcp_tool_schema import mcp_prefixed_tool_name
from tools.tool_output_truncate import truncate_head_tail
logger = logging.getLogger("tools.mcp_tool")
@@ -28,18 +29,8 @@ _MCP_RESOURCE_MAX_B64_CHARS = _MCP_RESOURCE_MAX_BYTES * 4 // 3 + 4
def _truncate_mcp_text_result(text: str, max_chars: int = _MCP_HARD_RESULT_CAP_CHARS) -> str:
"""Pass text at or under ``max_chars`` unchanged; otherwise keep a 40% head / 60% tail
split with an omission notice between.
Bound pathological MCP text before it propagates (#56059).
"""
if len(text) <= max_chars:
return text
head_chars = int(max_chars * 0.4)
tail_chars = max_chars - head_chars
omitted = len(text) - head_chars - tail_chars
return (text[:head_chars] + f"\n\n... [MCP RESULT TRUNCATED - {omitted:,} chars omitted "
f"out of {len(text):,} total] ...\n\n" + text[-tail_chars:])
"""Bound pathological MCP text before it propagates (#56059)."""
return truncate_head_tail(text, max_chars, label="MCP RESULT")
def _is_reserved_mcp_meta_key(key: str) -> bool:

View File

@@ -145,16 +145,9 @@ def _apply_output_transform_hook(command, output, returncode, task_id, env_type)
def _truncate_head_tail(output: str) -> str:
"""Truncate keeping head (errors often appear early) and tail (most recent)."""
from tools.tool_output_limits import get_max_bytes
max_chars = get_max_bytes()
if len(output) <= max_chars:
return output
head_chars = int(max_chars * 0.4)
tail_chars = max_chars - head_chars
notice = (f"\n\n... [OUTPUT TRUNCATED - {len(output) - head_chars - tail_chars} "
f"chars omitted out of {len(output)} total] ...\n\n")
return output[:head_chars] + notice + output[-tail_chars:]
from tools.tool_output_truncate import truncate_head_tail
return truncate_head_tail(output, get_max_bytes())
def _failure_hint(command: str, returncode: int, output: str, exit_note) -> Optional[str]: