test(e2e/handoff): gate the dashboard respawn on #124938 when the runner's systemd unit owns the cgroup

On CI every job runs inside hosted-compute-agent.service, so the hand-started
dashboard sits in that unit's cgroup and hermes update restarts the unit
instead of respawning the dashboard (both columns). The verdict names the unit
the update restarted, the N-1 column accepts #124778 or #124938, and the
failure report shows the sandbox's cgroup. Reproduced locally by running the
suite inside a transient systemd user service.
This commit is contained in:
teknium1
2026-09-27 00:55:03 -07:00
committed by Teknium
parent c02f6b0f5c
commit fb2dded3d1

View File

@@ -9,6 +9,7 @@ settle) is paid once per column, not once per property.
from __future__ import annotations
import contextlib
import datetime as dt
import json
import os
@@ -39,9 +40,17 @@ _IMPORT_ERR = re.compile(r"ModuleNotFoundError|ImportError|No module named|canno
# Open issues behind a property, per column (a fix PR deletes its entry). Each pattern matches only the
# verdict text its bug produces; any other failure of the same property propagates.
# A dashboard started from a shell that runs under a systemd service (every GitHub-hosted job runs in
# hosted-compute-agent.service) sits in that unit's cgroup; the update restarts the unit instead of
# respawning the dashboard. The N-1 column keeps this gate until a release carrying the fix is N-1.
_FOREIGN_UNIT_GATE = (r"the update restarted the systemd unit \S+ the dashboard was started under",
"gated on #124938: the update restarts the systemd unit whose cgroup a manual dashboard "
"was started in instead of respawning the dashboard")
DASHBOARD_GATES = {
"n1": (r"the respawned dashboard died parsing its launcher",
"gated on #124778: the update respawns a manual dashboard on a launcher the PM takeover made a shell shim"),
"head": [_FOREIGN_UNIT_GATE],
"n1": [(r"the respawned dashboard died parsing its launcher",
"gated on #124778: the update respawns a manual dashboard on a launcher the PM takeover made a shell "
"shim"), _FOREIGN_UNIT_GATE],
}
# The respawned dashboard can die after the updater already booked it "restarted": the receipt says
# success and the update exits 0 with the dashboard dark (racy: an earlier death is "unaccounted", exit 1).
@@ -52,6 +61,18 @@ CRON_GATES = {
"gated on #113293: a cron job due during an update fires into the swap window, fails, and loses its slot"),
}
# ``hermes update``'s own line for each systemd unit it restarted (or tried to) after stopping a dashboard.
_UNIT_RESTART = re.compile(r"✓ restarted systemd service (\S+\.service)|⚠ (\S+\.service): systemctl restart returned")
@contextlib.contextmanager
def _gated(entries):
"""``known_failure`` over several open issues: the one whose pattern the failure matches xfails it."""
with contextlib.ExitStack() as stack:
for pattern, reason in entries:
stack.enter_context(known_failure(pattern, reason))
yield
class Model:
"""The fake provider's brain. Chat turns get ``REPLY``; the cron job's prompt is counted; each
@@ -356,7 +377,10 @@ def run(column: str, root: Path) -> SimpleNamespace:
# Workers spawned after the update: every run of the new card, and every re-run of the in-flight one.
o.post_update_worker_logs = "\n".join(_run_segments(card_log[o.card2]) + _run_segments(card_log[o.card1])[1:])
inst.logs.append(restart_log)
o.diag = (inst.diagnostics(o.up) + f"\n--- gateway.pid history ---\n{watch.render()}"
# bwrap keeps the caller's cgroup: the sandbox's processes (the dashboard too) live in ours.
o.cgroup = Path("/proc/self/cgroup").read_text(errors="replace").strip() if sys.platform == "linux" else ""
o.diag = (inst.diagnostics(o.up) + f"\n--- cgroup of the sandbox ---\n{o.cgroup}"
f"\n--- gateway.pid history ---\n{watch.render()}"
f"\n--- cron job ---\n{json.dumps(o.cron_job_after, indent=1)[-2000:]}"
f"\n--- kanban ---\ncards={o.cards} provider kanban_complete calls={o.completes}\n"
f"peak live workers per card={o.worker_peak} (at, pids)={o.worker_evidence}\nruns={json.dumps(o.runs)}"
@@ -423,7 +447,7 @@ class HandoffProperties:
o = fleet
out = o.up.stdout + o.up.stderr
assert "TypeError" not in out, f"the update tripped over the dashboard\n{o.diag}"
with known_gate(DASHBOARD_GATES, o.column):
with _gated(DASHBOARD_GATES.get(o.column, ())):
assert o.dash_new is not None, f"{dashboard_verdict(o)}\n{o.diag}"
assert len(o.dash_roots) == 1, f"expected exactly one dashboard: {o.dash_roots}\n{o.diag}"
assert o.dash_listener == o.dash_new, f"the dashboard on port {o.dash_port} churned\n{o.diag}"
@@ -465,6 +489,12 @@ def dashboard_verdict(o) -> str:
if re.search(r"^SyntaxError", o.dash_restarts, re.M) and "restarted:" in o.up.stdout:
return (f"the respawned dashboard died parsing its launcher: the update replayed the pre-update argv and "
f"Python read a shell script (SyntaxError in logs/dashboard-restart.log); port {o.dash_port} is dark")
# The sandbox runs no Hermes unit, so any unit the dashboard stop restarts is the one the test runner
# (and so the hand-started dashboard) happens to live in.
unit = next((a or b for a, b in _UNIT_RESTART.findall(o.up.stdout)), None)
if unit and not o.dash_restarts:
return (f"the update restarted the systemd unit {unit} the dashboard was started under (cgroup "
f"{o.cgroup}) instead of respawning the dashboard; port {o.dash_port} is dark")
return f"no new dashboard serves port {o.dash_port} (old pid {o.dash_old}, listener now {o.dash_listener})"