From 9bf2c3c9fa546b2cea44c7717171f62f10f4a27e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 26 Sep 2026 22:40:55 +0530 Subject: [PATCH] fix(code-kernel): evict the remote kernel when a cell request fails to ship The cell ship now fails closed (RuntimeError). _execute_remote catches it and falls back to per-call execution, but the kernel stayed registered with cell_seq bumped. The next call would then reuse a kernel whose state silently missed this cell, with no state_lost flag. Discard (kill) it the way the timeout path does, so the next call starts a fresh kernel. The request never reached the runner (atomic publish), so the fallback still runs the code exactly once. --- tools/code_kernel_remote.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/tools/code_kernel_remote.py b/tools/code_kernel_remote.py index 709536309d..2e86b476ab 100644 --- a/tools/code_kernel_remote.py +++ b/tools/code_kernel_remote.py @@ -380,6 +380,13 @@ def _run_attached_cell(kernel: RemoteKernel, key: Tuple, code: str, *, env, task cell_status, cell_payload = "no-result", {} try: cell_status, cell_payload = _run_remote_cell(kernel, code, timeout) + except Exception: + # The request never reached the runner (the atomic ship failed), so the + # caller's per-call fallback runs the code exactly once. Kill the + # kernel as the timeout path does: leaving it registered would let the + # next call reuse a kernel whose state silently missed this cell. + _REGISTRY.discard(key, kernel) + raise finally: stop_event.set() rpc_thread.join(timeout=5)