fix(execute_code): report in-script tool errors the script ignored
A script that calls hermes_tools.write_file/patch and never prints the
returned {"error": ...} got status=success with empty output, so the
model believed the write happened. In a cross-agent bench this cost 3 of
18 Opus runs 4-6 extra turns (write_file refused by the read-before-write
guard, then re-probing and re-emitting whole files). The result now
carries tool_errors for that cell, and the in-script write_file doc says
existing files must be read first.
This commit is contained in:
@@ -428,6 +428,22 @@ class TestKernelOwnershipAndLifecycle(unittest.TestCase):
|
||||
self.assertIsNotNone(proc.returncode)
|
||||
|
||||
|
||||
class TestInScriptToolErrors(unittest.TestCase):
|
||||
def test_ignored_helper_error_is_reported_for_that_cell_only(self):
|
||||
"""A script that drops a helper's {"error": ...} return must not read as a clean success."""
|
||||
def _handle(tool_name, tool_args, task_id=None):
|
||||
if tool_name == "write_file":
|
||||
return json.dumps({"error": "Refusing to overwrite a.py: never read"})
|
||||
return json.dumps({"ok": True})
|
||||
|
||||
with _kernel_config(), patch("model_tools.handle_function_call", new=_handle):
|
||||
failed = _run("import hermes_tools\nhermes_tools.write_file('a.py', 'x')\nprint('done')\n")
|
||||
clean = _run("import hermes_tools\nhermes_tools.web_search(query='q')\n")
|
||||
self.assertEqual(failed["status"], "success", failed)
|
||||
self.assertEqual(failed["tool_errors"], [{"tool": "write_file", "error": "Refusing to overwrite a.py: never read"}])
|
||||
self.assertNotIn("tool_errors", clean)
|
||||
|
||||
|
||||
class TestPerCellRpcAuthority(unittest.TestCase):
|
||||
"""Interpreter state persists across cells; RPC authority must not."""
|
||||
|
||||
|
||||
Reference in New Issue
Block a user