Files
hermes-agent/tests/cron/test_run_one_job.py
teknium1 aedc6ccc3a test: purge low-value tests, lane py05 (339 removed)
Change-detectors, tautologies, source-reading tests, redundant duplicates,
mock-echo tests and dead/unrunnable tests. Per-test rationale in the lane
ledger (category + reason for every removal).
2026-09-23 03:15:26 -07:00

417 lines
15 KiB
Python

"""Characterization + unit tests for the `run_one_job` shared helper (Phase 4A).
`tick`'s per-job body (`_process_job`) is the execute → save → deliver → mark
sequence that fires ONE due job. Phase 4A extracts it into a module-level
`run_one_job(job, *, adapters=None, loop=None, verbose=False)` so the external
Chronos provider's `fire_due` can reuse the IDENTICAL body — no duplicated
correctness.
The first test characterizes the sequence as driven through `tick()` (proving
the extraction didn't change `tick`'s behavior); the rest unit-test the
extracted helper directly.
"""
import pytest
import cron.scheduler as s
def _patch_pipeline(monkeypatch, *, success=True, output="out", final="final response",
error=None, silent_marker_in=None):
"""Patch the job pipeline primitives and record the call order."""
calls = []
def fake_run_job(job, *, defer_agent_teardown=None, **kw):
calls.append(("run_job", job["id"]))
fr = final if silent_marker_in is None else silent_marker_in
return (success, output, fr, error)
def fake_save(jid, out):
calls.append(("save", jid))
return f"/tmp/{jid}.txt"
def fake_deliver(job, content, adapters=None, loop=None, **kwargs):
calls.append(("deliver", job["id"]))
return None
def fake_mark(jid, ok, err=None, delivery_error=None, **_kw):
calls.append(("mark", jid, ok))
monkeypatch.setattr(s, "run_job", fake_run_job)
monkeypatch.setattr(s, "save_job_output", fake_save)
monkeypatch.setattr(s, "_deliver_result", fake_deliver)
monkeypatch.setattr(s, "mark_job_run", fake_mark)
return calls
def test_tick_skips_job_when_durable_fire_claim_is_lost(monkeypatch):
"""A manual/external fire that wins the shared CAS must exclude ticker."""
calls = _patch_pipeline(monkeypatch)
monkeypatch.setattr(s, "get_due_jobs", lambda: [{"id": "j1", "name": "t"}])
monkeypatch.setattr(s, "claim_job_for_fire", lambda _job_id: False)
assert s.tick(verbose=False, sync=True) == 0
assert calls == []
def test_run_one_job_success_sequence(monkeypatch):
"""The extracted helper runs the same execute→save→deliver→mark sequence
for a successful job."""
calls = _patch_pipeline(monkeypatch)
ok = s.run_one_job({"id": "j2", "name": "t"})
assert ok is True
assert [c[0] for c in calls] == ["run_job", "save", "deliver", "mark"]
assert calls[-1] == ("mark", "j2", True)
def test_run_one_job_agent_declared_failure_uses_failure_bookkeeping(monkeypatch):
"""A delegated-child failure reported by the agent is not a healthy cron run."""
calls = _patch_pipeline(
monkeypatch,
final="[CRON_FAILURE]\nThe delegated child could not finish the report.",
)
ok = s.run_one_job({"id": "declared-failure", "name": "delegate", "deliver": "telegram"})
assert ok is True
assert [call[0] for call in calls] == ["run_job", "save", "deliver", "mark"]
assert calls[-1] == ("mark", "declared-failure", False)
def test_run_one_job_agent_declared_failure_is_delivered_verbatim(monkeypatch):
"""The agent's own evidence reaches the operator as written, not re-diagnosed by the
provider-error heuristics (a child that "timed out" is not a model-service timeout)."""
delivered = []
evidence = "The export subagent timed out after 30 minutes waiting on the database."
_patch_pipeline(monkeypatch, final=f"[CRON_FAILURE]\n{evidence}")
monkeypatch.setattr(
s, "_deliver_result", lambda job, content, **kw: delivered.append(content))
s.run_one_job({"id": "verbatim", "name": "nightly export", "deliver": "telegram"})
assert len(delivered) == 1
assert evidence.rstrip(".") in delivered[0]
assert "model service" not in delivered[0]
def test_run_one_job_marker_mentioned_in_report_stays_successful(monkeypatch):
"""Only the exact first line is control text; quoted markers remain report content."""
calls = _patch_pipeline(
monkeypatch,
final="The child documentation says [CRON_FAILURE], but this run recovered.",
)
s.run_one_job({"id": "quoted-marker", "name": "delegate", "deliver": "telegram"})
assert calls[-1] == ("mark", "quoted-marker", True)
def test_run_one_job_exception_delivers_failure_alert(monkeypatch):
"""An exception escaping the run body must not become a silent error row."""
delivered = []
marked = []
finished = []
monkeypatch.setattr(
s, "create_execution", lambda *_a, **_kw: {"id": "exec-j3"}
)
monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True)
monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: {})
monkeypatch.setattr(
s,
"run_job",
lambda *_a, **_kw: (_ for _ in ()).throw(
RuntimeError("Gemini HTTP 503 (UNAVAILABLE)")
),
)
monkeypatch.setattr(
s,
"_deliver_result",
lambda job, content, **_kw: delivered.append((job["id"], content)) or None,
)
monkeypatch.setattr(
s,
"mark_job_run",
lambda *args, **kwargs: marked.append((args, kwargs)),
)
monkeypatch.setattr(
s,
"finish_execution",
lambda *args, **kwargs: finished.append((args, kwargs)),
)
ok = s.run_one_job({"id": "j3", "name": "morning", "deliver": "telegram"})
assert ok is False
assert len(delivered) == 1 and delivered[0][0] == "j3"
# The notice carries the classifier verdict's gloss from the copy table (whatever its wording),
# never the raw HTTP code as the lead, plus a retry command.
from cron.scheduler_failure_copy import _provider_failure_cause, classify_cron_failure_reason
gloss = _provider_failure_cause(classify_cron_failure_reason("Gemini HTTP 503 (UNAVAILABLE)"))
assert gloss and gloss in delivered[0][1]
assert not delivered[0][1].lstrip("⚠️ ").startswith("Gemini HTTP 503")
assert "hermes cron run j3" in delivered[0][1]
assert marked == [
(("j3", False, "Gemini HTTP 503 (UNAVAILABLE)"), {"delivery_error": None})
]
assert finished == [
(
("exec-j3",),
{
"success": False,
"error": "Gemini HTTP 503 (UNAVAILABLE)",
"delivery_outcome": "delivered",
},
)
]
def test_run_one_job_exception_records_failure_alert_delivery_error(monkeypatch):
"""A failed fallback alert must populate last_delivery_error."""
marked = []
monkeypatch.setattr(
s, "create_execution", lambda *_a, **_kw: {"id": "exec-j4"}
)
monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True)
monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: {})
monkeypatch.setattr(
s,
"run_job",
lambda *_a, **_kw: (_ for _ in ()).throw(RuntimeError("provider failed")),
)
monkeypatch.setattr(s, "_deliver_result", lambda *_a, **_kw: "send failed: 502")
monkeypatch.setattr(
s,
"mark_job_run",
lambda *args, **kwargs: marked.append((args, kwargs)),
)
monkeypatch.setattr(s, "finish_execution", lambda *_a, **_kw: None)
assert s.run_one_job({"id": "j4", "deliver": "telegram"}) is False
assert marked == [
(("j4", False, "provider failed"), {"delivery_error": "send failed: 502"})
]
def _patch_escaped_failure(monkeypatch, delivered, *, exec_id, err):
"""Make run_job raise, and capture what the escape handler delivers."""
monkeypatch.setattr(s, "create_execution", lambda *_a, **_kw: {"id": exec_id})
monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True)
monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: {})
monkeypatch.setattr(
s,
"run_job",
lambda *_a, **_kw: (_ for _ in ()).throw(RuntimeError(err)),
)
monkeypatch.setattr(
s,
"_deliver_result",
lambda job, content, **_kw: delivered.append(content) or None,
)
monkeypatch.setattr(s, "mark_job_run", lambda *_a, **_kw: None)
monkeypatch.setattr(s, "finish_execution", lambda *_a, **_kw: None)
# Deterministic threshold: default 3, independent of the host config.
monkeypatch.setattr(s, "load_config", lambda: {})
def test_escaped_failure_delivery_carries_the_streak_nudge(monkeypatch):
"""A repeatedly-failing job must be nudged even when it fails at the
scheduler layer (#88655).
``mark_job_run`` increments ``failure_streak`` for an escaped failure just
as it does for an agent failure, so the counter climbs either way. But the
nudge that spends it was only composed on the normal delivery path, so a
job that raises before the run body on every tick - a bad import from a
half-applied update, a provider client that cannot construct - alerts
forever and is never told it should be reviewed or paused. Nothing else
surfaces the streak in chat.
"""
delivered = []
_patch_escaped_failure(
monkeypatch, delivered, exec_id="exec-j5", err="cannot import name X"
)
ok = s.run_one_job(
{
"id": "j5",
"name": "scout",
"deliver": "telegram",
"schedule": {"kind": "interval"},
"failure_streak": 2, # + this run = 3 = default threshold
}
)
assert ok is False
assert len(delivered) == 1
assert "cannot import name X" in delivered[0]
assert "hermes cron pause scout" in delivered[0]
def test_escaped_failure_delivery_stays_quiet_below_the_threshold(monkeypatch):
"""The nudge is appended, not always-on: a first failure reads as before."""
delivered = []
_patch_escaped_failure(
monkeypatch, delivered, exec_id="exec-j6", err="provider failed"
)
ok = s.run_one_job(
{
"id": "j6",
"name": "scout",
"deliver": "telegram",
"schedule": {"kind": "interval"},
"failure_streak": 0,
}
)
assert ok is False
assert len(delivered) == 1
assert "provider failed" in delivered[0]
assert "hermes cron pause scout" not in delivered[0]
def test_run_one_job_exception_after_delivery_does_not_redeliver(monkeypatch):
"""Once delivery has been attempted, the outer handler must not send again."""
delivered = []
mark_calls = []
monkeypatch.setattr(
s, "create_execution", lambda *_a, **_kw: {"id": "exec-j5"}
)
monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True)
monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: {})
monkeypatch.setattr(
s,
"run_job",
lambda *_a, **_kw: (True, "out", "final response", None),
)
monkeypatch.setattr(s, "save_job_output", lambda jid, out: f"/tmp/{jid}.txt")
monkeypatch.setattr(
s,
"_deliver_result",
lambda job, content, **_kw: delivered.append((job["id"], content)) or None,
)
def fake_mark(*args, **kwargs):
mark_calls.append((args, kwargs))
if len(mark_calls) == 1:
raise RuntimeError("bookkeeping boom")
monkeypatch.setattr(s, "mark_job_run", fake_mark)
monkeypatch.setattr(s, "finish_execution", lambda *_a, **_kw: None)
ok = s.run_one_job({"id": "j5", "name": "once", "deliver": "telegram"})
assert ok is False
assert delivered == [("j5", "final response")]
assert mark_calls[0] == (("j5", True, None), {"delivery_error": None})
assert mark_calls[1] == (
("j5", False, "bookkeeping boom"),
{"delivery_error": None},
)
def test_run_one_job_keyboard_interrupt_skips_delivery_and_reraises(monkeypatch):
"""Hard interrupts must not attempt failure delivery; they re-raise."""
delivered = []
marked = []
finished = []
monkeypatch.setattr(
s, "create_execution", lambda *_a, **_kw: {"id": "exec-j6"}
)
monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True)
monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: {})
monkeypatch.setattr(
s,
"run_job",
lambda *_a, **_kw: (_ for _ in ()).throw(KeyboardInterrupt()),
)
monkeypatch.setattr(
s,
"_deliver_result",
lambda job, content, **_kw: delivered.append((job["id"], content)) or None,
)
monkeypatch.setattr(
s,
"mark_job_run",
lambda *args, **kwargs: marked.append((args, kwargs)),
)
monkeypatch.setattr(
s,
"finish_execution",
lambda *args, **kwargs: finished.append((args, kwargs)),
)
with pytest.raises(KeyboardInterrupt):
s.run_one_job({"id": "j6", "name": "interrupt", "deliver": "telegram"})
assert delivered == []
assert marked == [(("j6", False, "KeyboardInterrupt"), {})]
assert finished == [
(
("exec-j6",),
{
"success": False,
"error": "KeyboardInterrupt",
"delivery_outcome": "suppressed",
},
)
]
def test_run_one_job_installs_secret_scope_under_multiplex(monkeypatch, tmp_path):
"""Regression: under profile isolation (multiplex active), run_one_job must
keep one profile secret scope active through execution and delivery so
credential reads do not fail closed or fall through to another profile,
then tear the scope down after the complete job lifecycle.
Behavior contract: the same scope is present during run_job and
_deliver_result, and no scope remains after run_one_job returns.
"""
from agent import secret_scope as ss
# Point cron's home resolution at a profile whose .env carries a secret.
(tmp_path / ".env").write_text("OPENROUTER_BASE_URL=https://openrouter.ai/api/v1\n")
monkeypatch.setattr(s, "_get_hermes_home", lambda: tmp_path)
scope_during_run = {}
scope_during_delivery = {}
def fake_run_job(job, *, defer_agent_teardown=None, **kw):
# This is where resolve_runtime_provider() would read a secret. Prove a
# scope is installed and the profile's secret resolves without raising.
scope_during_run["scope"] = ss.current_secret_scope()
scope_during_run["base_url"] = ss.get_secret("OPENROUTER_BASE_URL")
return (True, "out", "final", None)
def fake_deliver(*args, **kwargs):
scope_during_delivery["scope"] = ss.current_secret_scope()
scope_during_delivery["base_url"] = ss.get_secret("OPENROUTER_BASE_URL")
return None
monkeypatch.setattr(s, "run_job", fake_run_job)
monkeypatch.setattr(s, "save_job_output", lambda jid, out: f"/tmp/{jid}.txt")
monkeypatch.setattr(s, "_deliver_result", fake_deliver)
monkeypatch.setattr(s, "mark_job_run", lambda *a, **k: None)
ss.set_multiplex_active(True)
try:
ok = s.run_one_job({"id": "j7", "name": "t"})
finally:
ss.set_multiplex_active(False)
assert ok is True
# The same profile scope covered both execution and delivery.
assert scope_during_run["scope"] is not None
assert scope_during_run["base_url"] == "https://openrouter.ai/api/v1"
assert scope_during_delivery["scope"] == scope_during_run["scope"]
assert scope_during_delivery["base_url"] == "https://openrouter.ai/api/v1"
# And it was torn down after the full lifecycle returned (no leak).
assert ss.current_secret_scope() is None