diff --git a/scripts/analyze_livetest.py b/scripts/analyze_livetest.py index 7e55b5c7c7..6be6b7ad92 100644 --- a/scripts/analyze_livetest.py +++ b/scripts/analyze_livetest.py @@ -43,7 +43,7 @@ def fmt_bridge_seq(calls): if isinstance(qs, list): q = "; ".join(str(x) for x in qs) else: # legacy single-query transcripts - q = str(args.get("query", "?")) + q = str(args["query"] if "query" in args else "?") parts.append(f"search('{q[:30]}')") elif c["name"] == "tool_describe": args = c.get("args") or {} diff --git a/scripts/tool_search_livetest.py b/scripts/tool_search_livetest.py index 1df22ff8c7..fa28406030 100644 --- a/scripts/tool_search_livetest.py +++ b/scripts/tool_search_livetest.py @@ -287,7 +287,7 @@ def setup_isolated_home(enabled: bool, listing: str = "off", "enabled": "on" if enabled else "off", "threshold_pct": 10, "search_default_limit": 5, - "max_search_limit": 20, + "max_search_limit": 25, "listing": listing, "listing_max_tokens": listing_max_tokens, }, diff --git a/scripts/tool_search_livetest_ue_disc.py b/scripts/tool_search_livetest_ue_disc.py index f3d0bb33f3..615a662326 100644 --- a/scripts/tool_search_livetest_ue_disc.py +++ b/scripts/tool_search_livetest_ue_disc.py @@ -115,6 +115,15 @@ def score_survey(resp: str, truth: Dict[str, bool]) -> bool: return True +def _bridge_query_text(call: Dict[str, Any]) -> str: + """Render current multi-query calls and legacy saved transcript calls.""" + args = call.get("args") or {} + queries = args.get("queries") + if isinstance(queries, list): + return "; ".join(str(query) for query in queries) + return str(args["query"] if "query" in args else "?") + + def run_one(scenario, mode, rep, out_dir: Path): model = os.environ.get("TS_UE_MODEL", "anthropic/claude-opus-4.8") lmax = int(os.environ.get("TS_UE_LISTING_MAX", "30000")) @@ -203,7 +212,11 @@ def run_one(scenario, mode, rep, out_dir: Path): "prompt_tokens_total": sum(u["prompt_tokens"] or 0 for u in usage_log), "ue_calls": [c[-60:] for c in ue_calls][:15], "write_calls": [c[-60:] for c in write_calls][:10], - "bridge_queries": [(b.get("args") or {}).get("query") for b in bridge_call_log if b["name"] == "tool_search"][:10], + "bridge_queries": [ + _bridge_query_text(call) + for call in bridge_call_log + if call["name"] == "tool_search" + ][:10], "success": bool(success), "error": error, "final_response": base._redact_secrets(final_response)[:400], } diff --git a/scripts/tool_search_livetest_ue_hard.py b/scripts/tool_search_livetest_ue_hard.py index 24ad8f2d1f..c2f8950975 100644 --- a/scripts/tool_search_livetest_ue_hard.py +++ b/scripts/tool_search_livetest_ue_hard.py @@ -36,6 +36,22 @@ from tool_search_livetest_ue import load_epic_tools, _SANITIZE # reuse loader N_REPS = int(os.environ.get("TS_BENCH_REPS", "2")) + +def _bridge_call_value(call: Dict[str, Any]) -> Any: + """Summarize current batch arguments with legacy transcript fallbacks.""" + args = call.get("args") or {} + if call["name"] == "tool_search": + queries = args.get("queries") + if isinstance(queries, list): + return "; ".join(str(query) for query in queries) + return args["query"] if "query" in args else None + if call["name"] == "tool_describe": + names = args.get("names") + if isinstance(names, list): + return ", ".join(str(name) for name in names) + return args.get("name") + + # --------------------------------------------------------------------------- # Type-aware mock world # --------------------------------------------------------------------------- @@ -276,7 +292,9 @@ def run_one(scenario, mode, rep, out_dir: Path): "first_correct": first_correct, "final_correct": final_correct, "wrong_calls": wrong_calls, "success": bool(success), "ue_calls": [c["name"][-70:] for c in ue_calls][:20], - "bridge_calls": [(b["name"], (b.get("args") or {}).get("query") or (b.get("args") or {}).get("name")) for b in bridge_call_log][:20], + "bridge_calls": [ + (call["name"], _bridge_call_value(call)) for call in bridge_call_log + ][:20], "error": error, "final_response": base._redact_secrets(final_response)[:300], }