diff --git a/.gitignore b/.gitignore index 5b4716c9e8..5d35cd3667 100644 --- a/.gitignore +++ b/.gitignore @@ -190,8 +190,8 @@ docs/superpowers/* /.install_method # Tool Search live-test harness output — non-deterministic model transcripts, -# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. -scripts/out/ +# regenerated by evals/tool_search/tool_search_livetest*.py. Never an artifact of the repo. +evals/tool_search/out*/ # Per-release changelog drafts. These exist only transiently during a release # cut (passed to `gh release create --notes-file`); the GitHub Release itself diff --git a/evals/browser_use/README.md b/evals/browser_use/README.md index 61150ba225..35b48dee09 100644 --- a/evals/browser_use/README.md +++ b/evals/browser_use/README.md @@ -21,7 +21,7 @@ web tasks. rating aggregation, JS/delayed render, login chain, cross-category compare). - **Resume-safe.** Completed cells in `results/*.jsonl` are skipped on rerun - (same pattern as `scripts/toolperf_abeval`). + (same pattern as `evals/toolperf_abeval`). - **Backend matrix.** `orchestrate.py` drives a local headless-Chrome CDP; `orchestrate_cloud.py --backend nous-cloud|browserbase` provisions a real cloud browser per cell through the same provider plumbing the product uses. diff --git a/scripts/benchmark_browser_eval.py b/evals/browser_use/benchmark_browser_eval.py similarity index 98% rename from scripts/benchmark_browser_eval.py rename to evals/browser_use/benchmark_browser_eval.py index 6d87f73cc8..d2cfc14075 100644 --- a/scripts/benchmark_browser_eval.py +++ b/evals/browser_use/benchmark_browser_eval.py @@ -4,7 +4,7 @@ Runs both paths against the same live Chrome and prints a comparison table. Not a pytest — a script you run manually for the PR description. Usage: - .venv/bin/python scripts/benchmark_browser_eval.py [--iterations N] + .venv/bin/python evals/browser_use/benchmark_browser_eval.py [--iterations N] """ from __future__ import annotations diff --git a/evals/browser_use/orchestrate.py b/evals/browser_use/orchestrate.py index 678708430e..b78f9c0a60 100644 --- a/evals/browser_use/orchestrate.py +++ b/evals/browser_use/orchestrate.py @@ -1,7 +1,7 @@ """Local-CDP battery orchestrator: tasks x arms x models x reps. Resume-safe: completed cells in results.jsonl are skipped, so a killed -battery continues where it left off (same pattern as scripts/toolperf_abeval). +battery continues where it left off (same pattern as evals/toolperf_abeval). Usage: # start a headless Chrome first: diff --git a/scripts/LIVETEST_README.md b/evals/tool_search/README.md similarity index 87% rename from scripts/LIVETEST_README.md rename to evals/tool_search/README.md index 332d5509b9..bfa5d1a3d3 100644 --- a/scripts/LIVETEST_README.md +++ b/evals/tool_search/README.md @@ -2,14 +2,14 @@ Runs five scenarios against a real model (Claude Haiku 4.5 via OpenRouter) to verify that the bridge tools work end-to-end. Records transcripts in -`scripts/out/`. +`evals/tool_search/out/`. ## Running ```bash cd -python3 scripts/tool_search_livetest.py # runs all 5 scenarios x 2 modes -python3 scripts/analyze_livetest.py # side-by-side report +python3 evals/tool_search/tool_search_livetest.py # runs all 5 scenarios x 2 modes +python3 evals/tool_search/analyze_livetest.py # side-by-side report ``` Requires `OPENROUTER_API_KEY` set or present in `~/.hermes/.env`. @@ -34,7 +34,7 @@ A/B baseline. The harness records: ## Output structure ``` -scripts/out/ +evals/tool_search/out/ __enabled.json # tool_search ON __disabled.json # tool_search OFF _summary.json # one-line summary across all runs diff --git a/scripts/analyze_livetest.py b/evals/tool_search/analyze_livetest.py similarity index 100% rename from scripts/analyze_livetest.py rename to evals/tool_search/analyze_livetest.py diff --git a/scripts/tool_search_livetest.py b/evals/tool_search/tool_search_livetest.py similarity index 99% rename from scripts/tool_search_livetest.py rename to evals/tool_search/tool_search_livetest.py index 8cd775dddf..13157015e9 100644 --- a/scripts/tool_search_livetest.py +++ b/evals/tool_search/tool_search_livetest.py @@ -37,7 +37,7 @@ ORIGINAL_HOME = os.environ.get("HERMES_HOME") ORIGINAL_AUTH = Path.home() / ".hermes" / "auth.json" _THIS_DIR = Path(__file__).resolve().parent -_WORKTREE_ROOT = _THIS_DIR.parent +_WORKTREE_ROOT = _THIS_DIR.parents[1] sys.path.insert(0, str(_WORKTREE_ROOT)) # --------------------------------------------------------------------------- diff --git a/scripts/tool_search_livetest2.py b/evals/tool_search/tool_search_livetest2.py similarity index 99% rename from scripts/tool_search_livetest2.py rename to evals/tool_search/tool_search_livetest2.py index eb9d797133..94cf5ea896 100644 --- a/scripts/tool_search_livetest2.py +++ b/evals/tool_search/tool_search_livetest2.py @@ -16,7 +16,7 @@ from pathlib import Path from typing import Any, Dict, List _THIS_DIR = Path(__file__).resolve().parent -_WORKTREE_ROOT = _THIS_DIR.parent +_WORKTREE_ROOT = _THIS_DIR.parents[1] sys.path.insert(0, str(_WORKTREE_ROOT)) sys.path.insert(0, str(_THIS_DIR)) diff --git a/scripts/tool_search_livetest_ue.py b/evals/tool_search/tool_search_livetest_ue.py similarity index 99% rename from scripts/tool_search_livetest_ue.py rename to evals/tool_search/tool_search_livetest_ue.py index 5f9f54e29b..65870c83e5 100644 --- a/scripts/tool_search_livetest_ue.py +++ b/evals/tool_search/tool_search_livetest_ue.py @@ -23,7 +23,7 @@ from pathlib import Path from typing import Any, Dict, List _THIS_DIR = Path(__file__).resolve().parent -_WORKTREE_ROOT = _THIS_DIR.parent +_WORKTREE_ROOT = _THIS_DIR.parents[1] sys.path.insert(0, str(_WORKTREE_ROOT)) sys.path.insert(0, str(_THIS_DIR)) diff --git a/scripts/tool_search_livetest_ue_disc.py b/evals/tool_search/tool_search_livetest_ue_disc.py similarity index 99% rename from scripts/tool_search_livetest_ue_disc.py rename to evals/tool_search/tool_search_livetest_ue_disc.py index 6a89a86274..ba13c00ad4 100644 --- a/scripts/tool_search_livetest_ue_disc.py +++ b/evals/tool_search/tool_search_livetest_ue_disc.py @@ -24,7 +24,7 @@ from pathlib import Path from typing import Any, Dict, List _THIS_DIR = Path(__file__).resolve().parent -_WORKTREE_ROOT = _THIS_DIR.parent +_WORKTREE_ROOT = _THIS_DIR.parents[1] sys.path.insert(0, str(_WORKTREE_ROOT)) sys.path.insert(0, str(_THIS_DIR)) diff --git a/scripts/tool_search_livetest_ue_hard.py b/evals/tool_search/tool_search_livetest_ue_hard.py similarity index 99% rename from scripts/tool_search_livetest_ue_hard.py rename to evals/tool_search/tool_search_livetest_ue_hard.py index cb930aa277..7d45b09a0d 100644 --- a/scripts/tool_search_livetest_ue_hard.py +++ b/evals/tool_search/tool_search_livetest_ue_hard.py @@ -27,7 +27,7 @@ from pathlib import Path from typing import Any, Dict, List _THIS_DIR = Path(__file__).resolve().parent -_WORKTREE_ROOT = _THIS_DIR.parent +_WORKTREE_ROOT = _THIS_DIR.parents[1] sys.path.insert(0, str(_WORKTREE_ROOT)) sys.path.insert(0, str(_THIS_DIR)) diff --git a/scripts/toolperf_abeval/README.md b/evals/toolperf_abeval/README.md similarity index 99% rename from scripts/toolperf_abeval/README.md rename to evals/toolperf_abeval/README.md index 5a781d0b32..1f8fee214e 100644 --- a/scripts/toolperf_abeval/README.md +++ b/evals/toolperf_abeval/README.md @@ -55,7 +55,7 @@ in real production traffic. ## Run ```bash -cd scripts/toolperf_abeval +cd evals/toolperf_abeval export ABEVAL_ROOT=/tmp/abeval-workspace # results + sandboxes land here export ABEVAL_HOME=/tmp/abeval-home ./run_all.sh /tmp/abeval-baseline /path/to/fixes-tree 3 \ diff --git a/scripts/toolperf_abeval/ab_eval.py b/evals/toolperf_abeval/ab_eval.py similarity index 100% rename from scripts/toolperf_abeval/ab_eval.py rename to evals/toolperf_abeval/ab_eval.py diff --git a/scripts/toolperf_abeval/run_all.sh b/evals/toolperf_abeval/run_all.sh similarity index 100% rename from scripts/toolperf_abeval/run_all.sh rename to evals/toolperf_abeval/run_all.sh