Activation reaches plugin discovery before the application dependencies exist. Give PM its own locked Python project and runtime so it can install or repair the application without importing that dependency tree. Keep PM outside the application workspace. A shared uv workspace resolves the application graph and cannot provide this isolation. Route mutations through an isolated worker and preserve transaction callbacks, cancellation, custom package registrations, and correlated receipts. Use the same runtime builder for source installs and packaged payloads. Keep offline wheelhouse support in that builder. Nix builds the independent PM lock as a separate derivation. Refuse lazy-disabled bootstrap before installing tools or dependencies. Move first-party YAML readers and writers to ruamel. Keep the application lock's transitive PyYAML requirements for third-party packages. Verification: - Focused canonical Python suite: 177 passed, 1 host-gated skip. - Electron backend probes: 12 passed. Electron typecheck passed. - Both uv locks, scoped lint, Bash syntax, and whitespace checks passed. - Cold activation, corrupt-app repair, offline staging, and relocation ran. - Built and exercised the Nix PM runtime and standalone YAML merge script. Six broader caller test files retain the same 24 failing test IDs as an archive of HEAD. The existing real-home guard blocks those tests before they can exercise the affected paths. No full-suite pass is claimed. Native Windows signing and full Bionic package execution remain unverified.
200 lines
6.8 KiB
Python
200 lines
6.8 KiB
Python
"""One benchmark cell: task x arm x model x rep.
|
|
|
|
Usage:
|
|
python3 single_run.py <arm> <task_key> <model> <rep>
|
|
|
|
Arms:
|
|
base - built-in ``browser_*`` toolset (twelve tools), pinned tree $BUBENCH_BASE_TREE
|
|
pr - Browser Use CLI mode (single ``browser_exec`` tool), pinned tree $BUBENCH_PR_TREE
|
|
prns - same as pr but with the schema description stripped to the header only
|
|
(isolates the value of the helpers digest in the tool description)
|
|
|
|
Environment:
|
|
BUBENCH_ROOT workspace dir (default: dir containing this script)
|
|
BUBENCH_BASE_TREE checkout used for the ``base`` arm (e.g. a merge-base worktree)
|
|
BUBENCH_PR_TREE checkout used for the ``pr``/``prns`` arms
|
|
BUBENCH_TASKS tasks json (default: $BUBENCH_ROOT/tasks/hard.json)
|
|
BENCH_CDP_URL CDP endpoint both arms drive (default http://127.0.0.1:9333)
|
|
OPENROUTER_API_KEY provider credential for the runs
|
|
|
|
The run gets a throwaway HERMES_HOME so no local config leaks in, and the
|
|
web-fetch credential env vars are stripped so every arm must actually drive
|
|
the browser (no web_extract shortcuts).
|
|
|
|
Prints one line: ``RESULT_JSON:{...}`` consumed by orchestrate.py.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import tempfile
|
|
import time
|
|
|
|
ARM, TASK_KEY, MODEL, REP = sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4]
|
|
|
|
ROOT = os.environ.get("BUBENCH_ROOT", os.path.dirname(os.path.abspath(__file__)))
|
|
BASE_TREE = os.environ["BUBENCH_BASE_TREE"]
|
|
PR_TREE = os.environ["BUBENCH_PR_TREE"]
|
|
WT = {"base": BASE_TREE, "pr": PR_TREE, "prns": PR_TREE}[ARM]
|
|
|
|
TASKS_PATH = os.environ.get("BUBENCH_TASKS", os.path.join(ROOT, "tasks", "hard.json"))
|
|
TASKS = json.load(open(TASKS_PATH, encoding="utf-8"))
|
|
task = TASKS[TASK_KEY]
|
|
|
|
home = tempfile.mkdtemp(prefix=f"buhome-{ARM}-")
|
|
hh = os.path.join(home, ".hermes")
|
|
os.makedirs(os.path.join(hh, "logs"), exist_ok=True)
|
|
cdp = os.environ.get("BENCH_CDP_URL", "http://127.0.0.1:9333")
|
|
browser_cfg = (
|
|
{"cloud_provider": "local", "cdp_url": cdp}
|
|
if ARM == "base"
|
|
else {"backend": "browser-use"}
|
|
)
|
|
cfg = {
|
|
"model": {"provider": "openrouter", "default": MODEL},
|
|
"browser": browser_cfg,
|
|
"display": {"quiet": True},
|
|
}
|
|
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
|
import hermes_yaml as yaml
|
|
|
|
with open(os.path.join(hh, "config.yaml"), "w", encoding="utf-8") as f:
|
|
yaml.safe_dump(cfg, f)
|
|
os.environ["HERMES_HOME"] = hh
|
|
# Strip web-fetch shortcuts: every arm must drive the browser.
|
|
os.environ.pop("BROWSER_USE_API_KEY", None)
|
|
for k in ("FIRECRAWL_API_KEY", "NOUS_API_KEY", "TAVILY_API_KEY", "SERPER_API_KEY"):
|
|
os.environ.pop(k, None)
|
|
os.environ["BU_CDP_URL"] = cdp
|
|
os.environ["PATH"] = (
|
|
os.path.expanduser("~/.local/bin") + os.pathsep + os.environ.get("PATH", "")
|
|
)
|
|
|
|
sys.path.insert(0, WT)
|
|
import logging
|
|
|
|
logging.disable(logging.CRITICAL)
|
|
|
|
import run_agent # noqa: E402
|
|
|
|
_loaded = os.path.normcase(os.path.normpath(run_agent.__file__))
|
|
_want = os.path.normcase(os.path.normpath(WT))
|
|
assert _loaded.startswith(_want), f"wrong tree: {run_agent.__file__}"
|
|
|
|
if ARM == "prns":
|
|
# Strip the helpers digest from the schema: header-only description.
|
|
import tools.browser_use_cli as bu # noqa: E402
|
|
|
|
bu._skill_text_fetched = True
|
|
bu._skill_text_cache = None
|
|
bu.BROWSER_EXEC_SCHEMA["description"] = bu._description_header()
|
|
|
|
from run_agent import AIAgent # noqa: E402
|
|
|
|
# Provider resolution: default openrouter (original battery), but allow the
|
|
# Nous-subscription path on boxes without an OpenRouter key. Credentials are
|
|
# resolved through the product's own auth state, never printed.
|
|
_or_key = os.environ.get("OPENROUTER_API_KEY", "").strip()
|
|
if _or_key:
|
|
_agent_auth = dict(
|
|
base_url="https://openrouter.ai/api/v1",
|
|
api_key=_or_key,
|
|
provider="openrouter",
|
|
)
|
|
else:
|
|
# Resolved by the orchestrator BEFORE HERMES_HOME is redirected to the
|
|
# throwaway home (auth state lives in the real profile). Never printed.
|
|
_tok = os.environ.get("BUBENCH_NOUS_TOKEN", "").strip()
|
|
if not _tok:
|
|
raise SystemExit("no OPENROUTER_API_KEY and no Nous auth available")
|
|
_agent_auth = dict(
|
|
base_url=os.environ.get("BUBENCH_NOUS_BASE_URL", "https://inference-api.nousresearch.com/v1"),
|
|
api_key=_tok,
|
|
provider="nous",
|
|
)
|
|
|
|
agent = AIAgent(
|
|
**_agent_auth,
|
|
model=MODEL,
|
|
max_iterations=30,
|
|
quiet_mode=True,
|
|
skip_context_files=True,
|
|
skip_memory=True,
|
|
# NB: "terminal" must be present for the pr arms — since #81958's terminal
|
|
# gate, browser_exec is stripped from sessions whose toolsets exclude
|
|
# terminal. Both arms get the same toolsets for parity; audit
|
|
# tool_call_names in the results for terminal-tool bypasses (curl etc.).
|
|
enabled_toolsets=["browser", "terminal"],
|
|
save_trajectories=False,
|
|
)
|
|
|
|
schema_desc_len = 0
|
|
try:
|
|
from model_tools import get_tool_definitions
|
|
|
|
for t in get_tool_definitions(agent.enabled_toolsets):
|
|
if t["function"]["name"].startswith("browser"):
|
|
schema_desc_len += len(json.dumps(t["function"]))
|
|
except Exception:
|
|
pass
|
|
|
|
t0 = time.time()
|
|
error = None
|
|
final = ""
|
|
messages = []
|
|
try:
|
|
result = agent.run_conversation(task["prompt"])
|
|
final = (
|
|
(result.get("final_response") or "")
|
|
if isinstance(result, dict)
|
|
else str(result)
|
|
)
|
|
messages = result.get("messages", []) if isinstance(result, dict) else []
|
|
except Exception as e: # noqa: BLE001
|
|
error = f"{type(e).__name__}: {e}"
|
|
messages = getattr(agent, "messages", []) or []
|
|
wall = time.time() - t0
|
|
|
|
tool_calls = []
|
|
for m in messages:
|
|
if isinstance(m, dict) and m.get("role") == "assistant":
|
|
for tc in m.get("tool_calls") or []:
|
|
fn = (
|
|
(tc.get("function") or {}).get("name") if isinstance(tc, dict) else None
|
|
)
|
|
if fn:
|
|
tool_calls.append(fn)
|
|
|
|
|
|
def _ok(text: str) -> bool:
|
|
if task.get("oracle_all"):
|
|
return all(
|
|
re.search(re.escape(x), text, re.IGNORECASE) for x in task["oracle_all"]
|
|
)
|
|
return any(
|
|
re.search(re.escape(x), text, re.IGNORECASE) for x in task.get("oracle_any", [])
|
|
)
|
|
|
|
|
|
out = {
|
|
"arm": ARM,
|
|
"task": TASK_KEY,
|
|
"model": MODEL,
|
|
"rep": int(REP),
|
|
"ok": bool(final) and _ok(final) and error is None,
|
|
"wall_s": round(wall, 1),
|
|
"prompt_tokens": getattr(agent, "session_prompt_tokens", 0),
|
|
"completion_tokens": getattr(agent, "session_completion_tokens", 0),
|
|
"total_tokens": getattr(agent, "session_total_tokens", 0),
|
|
"api_calls": len([
|
|
m for m in messages if isinstance(m, dict) and m.get("role") == "assistant"
|
|
]),
|
|
"tool_calls": len(tool_calls),
|
|
"tool_call_names": tool_calls,
|
|
"browser_schema_chars": schema_desc_len,
|
|
"error": error,
|
|
"final_snippet": (final or "")[-400:],
|
|
}
|
|
print("RESULT_JSON:" + json.dumps(out, ensure_ascii=False))
|