fix(livetest): render multi-query bridge calls

This commit is contained in:
alt-glitch
2026-08-23 13:12:45 +05:30
committed by Teknium
parent 3b065745ca
commit a667ab60d9
4 changed files with 35 additions and 4 deletions

View File

@@ -43,7 +43,7 @@ def fmt_bridge_seq(calls):
if isinstance(qs, list):
q = "; ".join(str(x) for x in qs)
else: # legacy single-query transcripts
q = str(args.get("query", "?"))
q = str(args["query"] if "query" in args else "?")
parts.append(f"search('{q[:30]}')")
elif c["name"] == "tool_describe":
args = c.get("args") or {}

View File

@@ -287,7 +287,7 @@ def setup_isolated_home(enabled: bool, listing: str = "off",
"enabled": "on" if enabled else "off",
"threshold_pct": 10,
"search_default_limit": 5,
"max_search_limit": 20,
"max_search_limit": 25,
"listing": listing,
"listing_max_tokens": listing_max_tokens,
},

View File

@@ -115,6 +115,15 @@ def score_survey(resp: str, truth: Dict[str, bool]) -> bool:
return True
def _bridge_query_text(call: Dict[str, Any]) -> str:
"""Render current multi-query calls and legacy saved transcript calls."""
args = call.get("args") or {}
queries = args.get("queries")
if isinstance(queries, list):
return "; ".join(str(query) for query in queries)
return str(args["query"] if "query" in args else "?")
def run_one(scenario, mode, rep, out_dir: Path):
model = os.environ.get("TS_UE_MODEL", "anthropic/claude-opus-4.8")
lmax = int(os.environ.get("TS_UE_LISTING_MAX", "30000"))
@@ -203,7 +212,11 @@ def run_one(scenario, mode, rep, out_dir: Path):
"prompt_tokens_total": sum(u["prompt_tokens"] or 0 for u in usage_log),
"ue_calls": [c[-60:] for c in ue_calls][:15],
"write_calls": [c[-60:] for c in write_calls][:10],
"bridge_queries": [(b.get("args") or {}).get("query") for b in bridge_call_log if b["name"] == "tool_search"][:10],
"bridge_queries": [
_bridge_query_text(call)
for call in bridge_call_log
if call["name"] == "tool_search"
][:10],
"success": bool(success), "error": error,
"final_response": base._redact_secrets(final_response)[:400],
}

View File

@@ -36,6 +36,22 @@ from tool_search_livetest_ue import load_epic_tools, _SANITIZE # reuse loader
N_REPS = int(os.environ.get("TS_BENCH_REPS", "2"))
def _bridge_call_value(call: Dict[str, Any]) -> Any:
"""Summarize current batch arguments with legacy transcript fallbacks."""
args = call.get("args") or {}
if call["name"] == "tool_search":
queries = args.get("queries")
if isinstance(queries, list):
return "; ".join(str(query) for query in queries)
return args["query"] if "query" in args else None
if call["name"] == "tool_describe":
names = args.get("names")
if isinstance(names, list):
return ", ".join(str(name) for name in names)
return args.get("name")
# ---------------------------------------------------------------------------
# Type-aware mock world
# ---------------------------------------------------------------------------
@@ -276,7 +292,9 @@ def run_one(scenario, mode, rep, out_dir: Path):
"first_correct": first_correct, "final_correct": final_correct,
"wrong_calls": wrong_calls, "success": bool(success),
"ue_calls": [c["name"][-70:] for c in ue_calls][:20],
"bridge_calls": [(b["name"], (b.get("args") or {}).get("query") or (b.get("args") or {}).get("name")) for b in bridge_call_log][:20],
"bridge_calls": [
(call["name"], _bridge_call_value(call)) for call in bridge_call_log
][:20],
"error": error,
"final_response": base._redact_secrets(final_response)[:300],
}