fix(cron): acknowledged worker removes its own stderr capture; comments match captured stderr

The `<execution_id>.stderr` capture was unlinked only by the launching gateway
(pre-ack exit branch and the `handoff_files` finally). In the restart-safe
topology the worker deliberately outlives a gateway restart, so a gateway that
died mid-run left one orphan file per surviving run under
cron/external-workers/. After the ack the gateway never reads the capture
(it only serves the pre-ack death report), so the worker now unlinks the
sibling derived from its ack path in the post-run finally; the pre-ack paths
are unchanged so a worker dying before its ack still gets its stderr reported.

Two comments still described stderr as DEVNULL (the systemd-run failure branch
and the `__main__` setup_logging motivation); reword them to the captured-stderr
behaviour while keeping the WHY.
This commit is contained in:
teknium1
2026-09-16 12:41:35 -07:00
committed by Teknium
parent e094e25b26
commit f857a4ed99
2 changed files with 17 additions and 4 deletions

View File

@@ -3314,7 +3314,8 @@ def _launch_external_cron_worker(job: dict) -> bool:
stderr_tail = external_worker_stderr_tail(stderr_path)
stderr_path.unlink(missing_ok=True)
if dispatch.mode == "scoped" and scoped_spawn_lost_user_bus(worker_env):
# systemd-run itself failed (stderr is DEVNULL): name the cause, not the exit code.
# systemd-run itself failed before any worker ran, so the captured stderr
# holds nothing useful: name the cause, not the exit code.
raise RuntimeError(
"restart-safe systemd scope could not be created: the user D-Bus session at "
f"/run/user/{os.getuid()}/bus disappeared after the gateway started. On a " # windows-footgun: ok — scoped dispatch exists only on Linux
@@ -3417,6 +3418,11 @@ def _run_external_worker_payload(payload_path: Path, ack_path: Path) -> bool:
os.environ.pop("_HERMES_CRON_EXTERNAL_WORKER", None)
else:
os.environ["_HERMES_CRON_EXTERNAL_WORKER"] = old_external_execution
# Post-ack the gateway never reads the stderr capture (it only
# serves the pre-ack death report) and may not outlive this run
# in the restart-safe topology, so the worker removes its own.
with contextlib.suppress(OSError):
ack_path.with_suffix(".stderr").unlink(missing_ok=True)
finally:
reset_secret_scope(secret_token)
set_multiplex_active(previous_multiplex)
@@ -3937,8 +3943,10 @@ if __name__ == "__main__":
parser.add_argument("--external-worker-file", type=Path, required=True)
parser.add_argument("--ack-file", type=Path, required=True)
args = parser.parse_args()
# The gateway spawns this worker with stdout/stderr on DEVNULL; without
# a handler every adoption/ack failure below would be invisible.
# The gateway spawns this worker with stdout on DEVNULL and stderr on a
# capture file it only reads back if we die before the ack; without a
# log handler every adoption/ack failure below would otherwise be
# invisible to the persistent log.
try:
from hermes_logging import setup_logging

View File

@@ -153,7 +153,9 @@ def test_external_worker_adopts_execution_and_runs_payload_once(
import cron.scheduler as scheduler
payload = tmp_path / "payload.json"
ack = tmp_path / "ready.json"
ack = tmp_path / "exec-1.ready"
stderr_capture = tmp_path / "exec-1.stderr"
stderr_capture.write_text("", encoding="utf-8")
payload.write_text(
json.dumps({
"job": {"id": "job-1", "execution_id": "exec-1"},
@@ -187,6 +189,9 @@ def test_external_worker_adopts_execution_and_runs_payload_once(
assert observed_homes == [expected_home, expected_home]
assert ack.exists()
assert not payload.exists()
# Post-ack the worker owns the stderr capture: a gateway that restarted mid-run
# would otherwise leave one orphan per surviving run.
assert not stderr_capture.exists()
def test_external_worker_refuses_to_run_without_durable_ownership(