fix(serve): SSH-isolated idle-exit keeps the backend alive while a cron job runs
turn_in_flight read only the dashboard session table; an in-process cron run never registers there, so the watchdog reported "no running turn" and exited mid-job (tool calls then failed with "cannot schedule new futures after interpreter shutdown", the execution was marked unknown, the slot lost). The probe now also consults cron.scheduler.get_running_job_ids — the ledger the gateway shutdown drain already uses. Addresses #107485
This commit is contained in:
@@ -95,15 +95,22 @@ _probe_failure_logged = False
|
||||
|
||||
|
||||
def turn_in_flight() -> Optional[bool]:
|
||||
"""True/False from the gateway's running-session table; None when it cannot be read. The table
|
||||
lives on ``tui_gateway.server`` (the voice mixin's helper is bound into that namespace). None
|
||||
keeps the backend alive forever, so the cause is logged once — a silent never-exits would be
|
||||
the original bug with a new face."""
|
||||
"""True/False from the gateway's running-session table OR the in-process cron scheduler; None
|
||||
when neither can be read. The session table lives on ``tui_gateway.server`` (the voice mixin's
|
||||
helper is bound into that namespace). Cron runs live outside that table
|
||||
(``cron.scheduler.get_running_job_ids``, the same ledger the gateway shutdown drain reads):
|
||||
without it a daily job mid-run reported "no turn" and the exit killed it (#107485). None keeps
|
||||
the backend alive forever, so the cause is logged once — a silent never-exits would be the
|
||||
original bug with a new face."""
|
||||
global _probe_failure_logged
|
||||
try:
|
||||
import tui_gateway.server as gateway
|
||||
with gateway._sessions_lock:
|
||||
return any(s.get("running") for s in gateway._sessions.values())
|
||||
running = any(s.get("running") for s in gateway._sessions.values())
|
||||
if running:
|
||||
return True
|
||||
from cron.scheduler import get_running_job_ids
|
||||
return bool(get_running_job_ids())
|
||||
except Exception:
|
||||
if not _probe_failure_logged:
|
||||
_probe_failure_logged = True
|
||||
|
||||
@@ -78,3 +78,19 @@ def test_watchdog_sets_should_exit_and_only_arms_for_ssh_isolated_backends(monke
|
||||
assert isinstance(ws_mod.app.state.ssh_isolated_clients, IdleClientTracker)
|
||||
finally:
|
||||
ws_mod.app.state._state.pop("ssh_isolated_clients", None) # process-global app: never leak the tracker
|
||||
|
||||
|
||||
def test_turn_probe_counts_in_flight_cron_execution():
|
||||
"""#107485: a cron job mid-run must keep the SSH-isolated backend alive; the run lives outside
|
||||
the dashboard session table, in the scheduler's running-job ledger."""
|
||||
import cron.scheduler as scheduler
|
||||
from hermes_cli.web_server_idle_exit import turn_in_flight
|
||||
|
||||
assert turn_in_flight() is False
|
||||
with scheduler._running_lock:
|
||||
scheduler._running_job_ids.add("idle-exit-probe-job")
|
||||
try:
|
||||
assert turn_in_flight() is True
|
||||
finally:
|
||||
with scheduler._running_lock:
|
||||
scheduler._running_job_ids.discard("idle-exit-probe-job")
|
||||
|
||||
Reference in New Issue
Block a user