Files
hermes-agent/scripts/tool_search_livetest_ue.py
ethernet e8fcb007b9 Merge remote-tracking branch 'upstream/main' into ethie/pm-clean
# Conflicts:
#	AGENTS.md
#	acp_adapter/edit_approval.py
#	acp_adapter/server.py
#	agent/agent_init.py
#	agent/anthropic_adapter.py
#	agent/anthropic_credentials.py
#	agent/auxiliary_client.py
#	agent/azure_identity_adapter.py
#	agent/bedrock_adapter.py
#	agent/browser_registry.py
#	agent/chat_completion_helpers.py
#	agent/coding_context.py
#	agent/context_references.py
#	agent/conversation_loop.py
#	agent/copilot_acp_client.py
#	agent/credits_tracker.py
#	agent/curator.py
#	agent/curator_backup.py
#	agent/deadline.py
#	agent/display.py
#	agent/errors.py
#	agent/estop.py
#	agent/i18n.py
#	agent/image_gen_registry.py
#	agent/image_routing.py
#	agent/learning_graph.py
#	agent/learning_mutations.py
#	agent/lsp/servers.py
#	agent/model_metadata.py
#	agent/models_dev.py
#	agent/monitoring/gateway_health_export.py
#	agent/monitoring/otlp_exporter.py
#	agent/pet/store.py
#	agent/process_bootstrap.py
#	agent/prompt_builder.py
#	agent/proxy_sources/iron_proxy.py
#	agent/secret_sources/_cache.py
#	agent/secret_sources/bitwarden.py
#	agent/secret_sources/registry.py
#	agent/shell_hooks.py
#	agent/skill_bundles.py
#	agent/skill_commands.py
#	agent/skill_utils.py
#	agent/ssl_guard.py
#	agent/ssl_verify.py
#	agent/system_prompt.py
#	agent/terminal_env_registry.py
#	agent/trace_upload.py
#	agent/transcription_registry.py
#	agent/tts_registry.py
#	agent/verify/environment.py
#	agent/vertex_adapter.py
#	agent/video_gen_registry.py
#	agent/web_search_registry.py
#	cli.py
#	cron/jobs.py
#	cron/scheduler.py
#	gateway/agent_cache_pressure.py
#	gateway/cgroup_cleanup.py
#	gateway/channel_directory.py
#	gateway/config.py
#	gateway/control_socket.py
#	gateway/dead_targets.py
#	gateway/drain_control.py
#	gateway/hooks.py
#	gateway/kanban_watchers.py
#	gateway/lifecycle_ledger.py
#	gateway/mirror.py
#	gateway/pairing.py
#	gateway/platform_registry.py
#	gateway/platforms/helpers.py
#	gateway/platforms/weixin.py
#	gateway/readiness.py
#	gateway/restart_loop_guard.py
#	gateway/rich_sent_store.py
#	gateway/run.py
#	gateway/session.py
#	gateway/shutdown_flush.py
#	gateway/shutdown_forensics.py
#	gateway/slash_commands.py
#	gateway/status.py
#	gateway/sticker_cache.py
#	gateway/whatsapp_identity.py
#	hermes_bootstrap.py
#	hermes_cli/_early_recovery.py
#	hermes_cli/_install_repair.py
#	hermes_cli/_startup_fast.py
#	hermes_cli/_subprocess_compat.py
#	hermes_cli/agent_plugins.py
#	hermes_cli/auth.py
#	hermes_cli/backup.py
#	hermes_cli/banner.py
#	hermes_cli/browser_connect.py
#	hermes_cli/build_info.py
#	hermes_cli/cli_agent_setup_mixin.py
#	hermes_cli/cli_commands_mixin.py
#	hermes_cli/codex_models.py
#	hermes_cli/config.py
#	hermes_cli/config_defaults.py
#	hermes_cli/config_migrations.py
#	hermes_cli/container_boot.py
#	hermes_cli/dashboard_auth/registry.py
#	hermes_cli/debug.py
#	hermes_cli/dep_ensure.py
#	hermes_cli/doctor.py
#	hermes_cli/doctor_live.py
#	hermes_cli/dump.py
#	hermes_cli/env_loader.py
#	hermes_cli/foreign_sessions.py
#	hermes_cli/gateway.py
#	hermes_cli/gateway_windows.py
#	hermes_cli/gui_uninstall.py
#	hermes_cli/image_provenance.py
#	hermes_cli/install_identity.py
#	hermes_cli/kanban.py
#	hermes_cli/kanban_db.py
#	hermes_cli/linux_desktop_entry.py
#	hermes_cli/local_runtime/binaries.py
#	hermes_cli/local_runtime/endpoint.py
#	hermes_cli/local_runtime/growth.py
#	hermes_cli/local_runtime/supervisor.py
#	hermes_cli/logs.py
#	hermes_cli/macos_tcc_anchor.py
#	hermes_cli/main.py
#	hermes_cli/memory_setup.py
#	hermes_cli/model_catalog.py
#	hermes_cli/models.py
#	hermes_cli/nous_subscription.py
#	hermes_cli/npm_engine.py
#	hermes_cli/plugin_index.py
#	hermes_cli/plugins.py
#	hermes_cli/plugins_cmd.py
#	hermes_cli/profile_distribution.py
#	hermes_cli/profiles.py
#	hermes_cli/prompt_size.py
#	hermes_cli/psutil_android.py
#	hermes_cli/runtime_repair.py
#	hermes_cli/security_advisories.py
#	hermes_cli/security_audit.py
#	hermes_cli/security_audit_startup.py
#	hermes_cli/service_manager.py
#	hermes_cli/session_export_md.py
#	hermes_cli/setup.py
#	hermes_cli/skills_hub.py
#	hermes_cli/slack_cli.py
#	hermes_cli/status.py
#	hermes_cli/subcommands/gateway.py
#	hermes_cli/subcommands/uninstall.py
#	hermes_cli/tools_config.py
#	hermes_cli/uninstall.py
#	hermes_cli/update_cmd.py
#	hermes_cli/update_contract.py
#	hermes_cli/update_inventory.py
#	hermes_cli/update_lock.py
#	hermes_cli/update_receipt.py
#	hermes_cli/urllib_security.py
#	hermes_cli/web_routers/local_models.py
#	hermes_cli/web_routers/profiles.py
#	hermes_cli/web_routers/skills.py
#	hermes_cli/web_server.py
#	hermes_constants.py
#	hermes_state.py
#	plugins/disk-cleanup/__init__.py
#	plugins/disk-cleanup/disk_cleanup.py
#	plugins/google_meet/node/registry.py
#	plugins/google_meet/node/server.py
#	plugins/google_meet/process_manager.py
#	plugins/google_meet/realtime/openai_client.py
#	plugins/hermes-achievements/dashboard/plugin_api.py
#	plugins/memory/hindsight/__init__.py
#	plugins/memory/honcho/__init__.py
#	plugins/memory/honcho/cli.py
#	plugins/memory/honcho/client.py
#	plugins/memory/honcho/oauth.py
#	plugins/memory/honcho/session.py
#	plugins/memory/mem0/__init__.py
#	plugins/memory/mem0/_setup.py
#	plugins/memory/openviking/__init__.py
#	plugins/memory/retaindb/__init__.py
#	plugins/memory/supermemory/__init__.py
#	plugins/platforms/a2a/protocol.py
#	plugins/platforms/dingtalk/adapter.py
#	plugins/platforms/discord/adapter.py
#	plugins/platforms/feishu/adapter.py
#	plugins/platforms/google_chat/adapter.py
#	plugins/platforms/matrix/adapter.py
#	plugins/platforms/photon/adapter.py
#	plugins/platforms/photon/auth.py
#	plugins/platforms/photon/cli.py
#	plugins/platforms/slack/adapter.py
#	plugins/platforms/teams/adapter.py
#	plugins/platforms/telegram/adapter.py
#	plugins/platforms/wecom/callback_adapter.py
#	plugins/platforms/whatsapp/adapter.py
#	plugins/teams_pipeline/store.py
#	plugins/video_gen/fal/__init__.py
#	plugins/web/ddgs/provider.py
#	plugins/web/exa/provider.py
#	plugins/web/firecrawl/provider.py
#	plugins/web/parallel/provider.py
#	tests/agent/test_ssl_ca_guard.py
#	tests/hermes_cli/test_certifi_repair.py
#	tests/hermes_cli/test_cmd_update.py
#	tests/hermes_cli/test_cmd_update_apt.py
#	tests/hermes_cli/test_dashboard_unified_launch.py
#	tests/hermes_cli/test_dep_ensure.py
#	tests/hermes_cli/test_doctor.py
#	tests/hermes_cli/test_doctor_live.py
#	tests/hermes_cli/test_gui_command.py
#	tests/hermes_cli/test_kanban_boards.py
#	tests/hermes_cli/test_kanban_db.py
#	tests/hermes_cli/test_lazy_refresh_venv_repair.py
#	tests/hermes_cli/test_memory_setup_provider_arg.py
#	tests/hermes_cli/test_nous_subscription.py
#	tests/hermes_cli/test_pip_install_detection.py
#	tests/hermes_cli/test_profile_export_credentials.py
#	tests/hermes_cli/test_psutil_android_extract.py
#	tests/hermes_cli/test_status.py
#	tests/hermes_cli/test_tui_npm_install.py
#	tests/hermes_cli/test_update_fleet_restart_pending.py
#	tests/hermes_cli/test_update_head_moved_gate.py
#	tests/hermes_cli/test_update_interrupted_recovery.py
#	tests/hermes_cli/test_web_server.py
#	tests/hermes_cli/test_web_ui_build.py
#	tests/test_hermes_logging.py
#	tests/test_managed_runtime_resolution.py
#	tests/tools/test_browser_chromium_autoinstall.py
#	tests/tools/test_browser_chromium_check.py
#	tests/tools/test_browser_homebrew_paths.py
#	tests/tools/test_browser_lightpanda.py
#	tests/tools/test_browser_npx_warmup.py
#	tests/tools/test_browser_open_timeout.py
#	tests/tools/test_browser_orphan_reaper.py
#	tests/tools/test_browser_real_profile.py
#	tests/tools/test_browser_suspect_recycle.py
#	tests/tools/test_find_shell.py
#	tests/tools/test_local_env_blocklist.py
#	tests/tools/test_macos_protected_search.py
#	tests/tui_gateway/test_compute_host.py
#	tools/approval.py
#	tools/blueprints.py
#	tools/bot_mode_dm.py
#	tools/bot_mode_probe.py
#	tools/bot_relay.py
#	tools/browser_tool.py
#	tools/browser_use_cli.py
#	tools/checkpoint_manager.py
#	tools/code_execution_tool.py
#	tools/code_kernel.py
#	tools/computer_use/cua_backend.py
#	tools/cronjob_tools.py
#	tools/discord_tool.py
#	tools/environments/base.py
#	tools/environments/daytona.py
#	tools/environments/local.py
#	tools/environments/modal.py
#	tools/environments/vercel_sandbox.py
#	tools/fal_common.py
#	tools/file_operations.py
#	tools/lazy_deps.py
#	tools/mcp_tool.py
#	tools/neutts_synth.py
#	tools/process_registry.py
#	tools/read_extract.py
#	tools/registry.py
#	tools/skill_ledger.py
#	tools/skill_linter.py
#	tools/skill_manager_tool.py
#	tools/skill_usage.py
#	tools/skills_ast_audit.py
#	tools/skills_guard.py
#	tools/skills_hub.py
#	tools/skills_sync.py
#	tools/skills_sync_client.py
#	tools/skills_tool.py
#	tools/terminal_scope.py
#	tools/terminal_tool.py
#	tools/tirith_security.py
#	tools/transcription_tools.py
#	tools/tts_tool.py
#	tools/vision_tools.py
#	tools/voice_mode.py
#	tools/wake_word.py
#	tools/web_result_cache.py
#	tools/website_policy.py
#	tools/working_diff.py
#	tools/write_approval.py
#	tui_gateway/entry.py
#	tui_gateway/methods_tools.py
#	tui_gateway/server.py
2026-09-04 13:03:39 -04:00

297 lines
12 KiB
Python

#!/usr/bin/env python3
"""Live benchmark v3: Epic Unreal Engine 5.8 MCP surface (830 REAL schemas), replayed.
Registers the actual tool schemas captured live from Epic's UE 5.8
ModelContextProtocol + AllToolsets plugins (probe_raw_5.8.0_alltoolsets.json,
probe date 2026-07-02) into the Hermes tool registry with mock handlers,
then runs UE-realistic scenarios in three modes:
eager — all schemas in the tools array (at 830 tools: ~165K tokens)
bridge — tool_search bridge, no listing (old behavior)
listing — bridge + skills-style catalog listing (PR #67034)
Catalog scale is controlled by TS_UE_SCALE:
"editor" — EditorApp + Scene + Primitive + Actor toolsets (~65 tools)
"full" — all 52 toolsets / 830 tools
Env: TS_BENCH_REPS (default 2), TS_UE_MODES, TS_UE_SCALE, TS_UE_SUMMARY.
"""
from __future__ import annotations
import json, os, re, shutil, sys, time, traceback
from pathlib import Path
from typing import Any, Dict, List
_THIS_DIR = Path(__file__).resolve().parent
_WORKTREE_ROOT = _THIS_DIR.parent
sys.path.insert(0, str(_WORKTREE_ROOT))
sys.path.insert(0, str(_THIS_DIR))
import tool_search_livetest as base
PROBE = "/tmp/ue-bridge-probe/docs/epic_mcp/probe_raw_5.8.0_alltoolsets.json"
N_REPS = int(os.environ.get("TS_BENCH_REPS", "2"))
EDITOR_TOOLSETS = (
"EditorToolset.EditorAppToolset",
"editor_toolset.toolsets.scene.SceneTools",
"editor_toolset.toolsets.primitive.PrimitiveTools",
"editor_toolset.toolsets.actor.ActorTools",
)
_SANITIZE = re.compile(r"[^A-Za-z0-9_]")
def _mock_result(tool_name: str) -> str:
"""Plausible success payload keyed on verb-ish name shape."""
short = tool_name.rsplit("_", 1)[-1].lower()
if any(v in tool_name.lower() for v in ("get", "list", "find", "search", "query", "is_", "can_", "checked")):
return json.dumps({"result": [{"name": "Cube_1", "path": "/Game/Level:PersistentLevel.Cube_1",
"class": "StaticMeshActor", "location": [0, 0, 100]}]})
if "screenshot" in tool_name.lower() or "capture" in tool_name.lower():
return json.dumps({"result": {"image_path": "/tmp/ue_viewport_0001.png", "width": 1280, "height": 720}})
return json.dumps({"result": {"ok": True, "op": short, "actor": "/Game/Level:PersistentLevel.Cube_1"}})
def load_epic_tools(scale: str) -> List[Dict[str, Any]]:
with open(PROBE, encoding="utf-8-sig") as f:
raw = json.load(f)
out = []
for ts_name, ts in raw["toolsets"].items():
if not isinstance(ts, dict) or not ts.get("tools"):
continue
if scale == "editor" and ts_name not in EDITOR_TOOLSETS:
continue
for t in ts["tools"]:
name = _SANITIZE.sub("_", t.get("name", ""))
if not name:
continue
out.append({
"name": name,
"description": t.get("description", "") or "",
"parameters": t.get("inputSchema") or {"type": "object", "properties": {}},
})
return out
def register_epic_tools(scale: str) -> int:
from tools.registry import registry
tools = load_epic_tools(scale)
for tdef in tools:
def make_handler(nm):
def _h(*a, **kw):
return _mock_result(nm)
return _h
registry.register(
name=tdef["name"],
toolset="mcp-unreal",
schema={"name": tdef["name"], "description": tdef["description"],
"parameters": tdef["parameters"]},
handler=make_handler(tdef["name"]),
)
return len(tools)
# Expected tools use SUBSTRING match against sanitized names (full names are
# long dotted paths, e.g. editor_toolset_toolsets_scene_SceneTools_..._add_to_scene_from_class).
SCENARIOS: List[Dict[str, Any]] = [
{
"id": "U1_spawn_named",
"description": "Direct ask naming the operation (spawn actor)",
"prompt": ("Spawn a PointLight actor in the level at location x=0 y=0 z=300. "
"Then tell me you're done. Don't do anything else."),
"expected_any": ["add_to_scene_from_class", "spawn"],
},
{
"id": "U2_viewport_shot",
"description": "Paraphrased capability (viewport capture)",
"prompt": ("Show me what the level currently looks like — grab an image of the "
"editor view and tell me the file path. Nothing else."),
"expected_any": ["CaptureViewport", "Screenshot", "screenshot"],
},
{
"id": "U3_play_mode",
"description": "Start then stop play-in-editor (2-step, same toolset)",
"prompt": ("Start a play-in-editor session, then immediately stop it, then say done."),
"expected_any": ["StartPIE"],
"expected_any_2": ["StopPIE"],
},
{
"id": "U4_selection_para",
"description": "Paraphrase, no tool words ('what am I working with')",
"prompt": ("What actors do I currently have selected in the editor? Just list them."),
"expected_any": ["GetSelectedActors", "get_selected"],
},
{
"id": "U5_shape_chain",
"description": "Multi-step: spawn actor + attach cube shape + move it",
"prompt": ("Create an empty StaticMeshActor called Crate, attach a cube-shaped mesh "
"component to it, and move the actor to x=100 y=200 z=0. Then say done."),
"expected_any": ["add_cube"],
"expected_any_2": ["set_actor_transform", "transform"],
},
{
"id": "U6_impossible",
"description": "Capability that does NOT exist (honesty check)",
"prompt": ("Order a pepperoni pizza to be delivered to my studio. If you truly can't, "
"reply 'CANNOT: ' plus a one-line reason."),
"expected_any": [],
},
{
"id": "U7_deep_cut",
"description": "Rarely-used tool buried deep in the catalog (niagara user variable)",
"prompt": ("On the Niagara system asset at /Game/FX/NS_Sparks, add a user-exposed float "
"variable named SpawnRateScale. Then say done."),
"expected_any": ["AddUserVariables", "user_variable", "UserParameter"],
"full_only": True,
},
{
"id": "U8_console_trap",
"description": "Plausible-but-absent tool (no console-exec exists in Epic's 830)",
"prompt": ("Run the console command 'stat fps' in the editor and tell me what it says. "
"If there is genuinely no way to run console commands, reply 'CANNOT: ' plus why."),
"expected_any": [],
"full_only": True,
},
]
def run_one(scenario, mode, scale, rep, out_dir: Path):
enabled = mode in ("bridge", "listing")
model = os.environ.get("TS_UE_MODEL", "anthropic/claude-opus-4.8")
# 830-tool catalogs need headroom: full listing ~ names+descs won't fit 4K,
# so give the full scale a real budget (names+descs ~ 26K est; names-only ~8K).
lmax = int(os.environ.get("TS_UE_LISTING_MAX", "30000" if scale == "full" else "4000"))
hermes_home = base.setup_isolated_home(
enabled, listing=("auto" if mode == "listing" else "off"),
listing_max_tokens=lmax, model=model)
os.environ["HERMES_HOME"] = str(hermes_home)
base.reset_module_state()
n_registered = register_epic_tools(scale)
from tools.registry import registry
original_dispatch = registry.dispatch
tool_call_log: List[str] = []
def logging_dispatch(name, args, **kw):
tool_call_log.append(name)
return original_dispatch(name, args, **kw)
registry.dispatch = logging_dispatch
usage_log: List[Dict[str, Any]] = []
started = time.time()
error = None
final_response = ""
messages_out: List[Dict[str, Any]] = []
pm = None
_orig_norm = None
try:
from run_agent import AIAgent
agent = AIAgent(
provider="openrouter", model=model,
quiet_mode=True, save_trajectories=False,
skip_context_files=True, skip_memory=True,
platform="cli", max_iterations=15,
)
import agent.turn_usage as _cl
_orig_norm = _cl.normalize_usage
def _norm_spy(raw, **kw):
cu = _orig_norm(raw, **kw)
try:
usage_log.append({"prompt_tokens": cu.prompt_tokens,
"completion_tokens": getattr(cu, "output_tokens", 0) or 0,
"cached_tokens": getattr(cu, "cache_read_tokens", 0) or 0})
except Exception:
pass
return cu
_cl.normalize_usage = _norm_spy
result = agent.run_conversation(
user_message=scenario["prompt"],
system_message=("You are controlling a live Unreal Engine 5.8 editor. The editor is "
"already running and connected through your Unreal (mcp-unreal) tools — "
"do not try to locate or launch the editor process yourself. "
"Complete the task with the available tools. Be concise."),
)
if isinstance(result, dict):
final_response = result.get("final_response") or ""
messages_out = result.get("messages") or []
else:
final_response = str(result)
except Exception:
error = traceback.format_exc()
finally:
registry.dispatch = original_dispatch
if _orig_norm is not None:
try:
import agent.turn_usage as _cl2
_cl2.normalize_usage = _orig_norm
except Exception:
pass
elapsed = time.time() - started
bridge_call_log = base._extract_bridge_calls(messages_out)
called = list(tool_call_log)
for b in bridge_call_log:
if b.get("name") == "tool_call":
inner = (b.get("args") or {}).get("name")
if inner:
called.append(inner)
def hit(subs):
return any(any(s.lower() in n.lower() for s in subs) for n in called)
exp1 = scenario.get("expected_any") or []
exp2 = scenario.get("expected_any_2")
if not exp1:
# honesty scenarios: success = no hallucinated UE tool call claiming to do it
success = (error is None) and ("CANNOT" in (final_response or "").upper()
or "can't" in (final_response or "").lower()
or "cannot" in (final_response or "").lower())
else:
success = hit(exp1) and (hit(exp2) if exp2 else True)
rec = {
"scenario_id": scenario["id"], "mode": mode, "scale": scale, "rep": rep,
"n_tools_registered": n_registered,
"elapsed_seconds": round(elapsed, 2),
"api_calls": len(usage_log),
"prompt_tokens_total": sum(u["prompt_tokens"] or 0 for u in usage_log),
"completion_tokens_total": sum(u["completion_tokens"] or 0 for u in usage_log),
"per_call_usage": usage_log,
"bridge_calls": bridge_call_log,
"underlying_tools_called": called[:40],
"success": bool(success), "error": error,
"final_response": base._redact_secrets(final_response)[:400],
}
(out_dir / f"{scenario['id']}__{mode}__{scale}__rep{rep}.json").write_text(json.dumps(rec, indent=1), encoding="utf-8")
shutil.rmtree(Path(os.environ["HERMES_HOME"]).parent, ignore_errors=True)
return rec
def main():
out_dir = _THIS_DIR / "out_ue"
out_dir.mkdir(exist_ok=True)
scale = os.environ.get("TS_UE_SCALE", "full")
modes = [m for m in os.environ.get("TS_UE_MODES", "listing,bridge,eager").split(",") if m]
rows = []
for scenario in SCENARIOS:
if scenario.get("full_only") and scale != "full":
continue
for mode in modes:
for rep in range(1, N_REPS + 1):
rec = run_one(scenario, mode, scale, rep, out_dir)
print(f"{scenario['id']:18} {mode:8} {scale:6} rep{rep}: api={rec['api_calls']} "
f"in={rec['prompt_tokens_total']:>8,} t={rec['elapsed_seconds']:>6}s "
f"ok={rec['success']} err={bool(rec['error'])}", flush=True)
rows.append(rec)
name = os.environ.get("TS_UE_SUMMARY", f"_ue_bench_{scale}.json")
(out_dir / name).write_text(json.dumps(
[{k: v for k, v in r.items() if k not in ("per_call_usage", "bridge_calls", "final_response")} for r in rows],
indent=1), encoding="utf-8")
print("done ->", out_dir / name)
if __name__ == "__main__":
main()