fix(cron): record agent-declared cron failures
This commit is contained in:
@@ -445,6 +445,23 @@ from cron.executions import (
|
||||
# Response marker that suppresses delivery (output is still saved locally for audit).
|
||||
SILENT_MARKER = "[SILENT]"
|
||||
|
||||
# Agent-declared failure marker for cron runs. Unlike SILENT, it is deliberately strict so a
|
||||
# report that merely quotes the token cannot turn a healthy run into a failed one.
|
||||
CRON_FAILURE_MARKER = "[CRON_FAILURE]"
|
||||
|
||||
|
||||
def _cron_failure_marker_error(text: str) -> Optional[str]:
|
||||
"""Return failure evidence when an agent response declares a cron failure.
|
||||
|
||||
Only the exact, standalone first line is control text. The caller keeps the complete response
|
||||
in the saved run output while routing this evidence through normal failure bookkeeping.
|
||||
"""
|
||||
lines = (text or "").splitlines()
|
||||
if not lines or lines[0].rstrip() != CRON_FAILURE_MARKER:
|
||||
return None
|
||||
evidence = "\n".join(lines[1:]).strip()
|
||||
return evidence or "Cron agent reported failure."
|
||||
|
||||
|
||||
def _is_cron_silence_response(text: str) -> bool:
|
||||
"""True when a cron final response should suppress delivery: ``[SILENT]`` (or SILENT /
|
||||
@@ -2926,6 +2943,14 @@ def _run_one_job_body(
|
||||
_record_fire_ownership_lost(job["id"], fire_owner, execution_id)
|
||||
return True
|
||||
|
||||
# An agent can finish its own turn after a delegated child has failed. Let it explicitly
|
||||
# declare that semantic failure so the existing failure path updates status, streaks,
|
||||
# ledger, and notification routing instead of recording a false healthy result.
|
||||
if success and not job.get("no_agent"):
|
||||
marker_error = _cron_failure_marker_error(final_response)
|
||||
if marker_error is not None:
|
||||
success, error = False, marker_error
|
||||
|
||||
# Agent is still live through delivery; wrap ALL of save/compose/deliver in try/finally so a
|
||||
# raise anywhere still tears the deferred agent down.
|
||||
d = _RunDelivery(job=job, success=success, error=error)
|
||||
|
||||
@@ -201,6 +201,9 @@ _CRON_HINT = (
|
||||
"rephrase it, whatever language the rest of your answer uses. "
|
||||
"Never combine [SILENT] with content — either report your "
|
||||
"findings normally, or say [SILENT] and nothing more. "
|
||||
"FAILURE: If a delegated child fails and this cron run must be "
|
||||
"recorded as failed, put [CRON_FAILURE] on the first line by itself, "
|
||||
"then explain the child failure on following lines. "
|
||||
"RECURSION: This is a run of an EXISTING scheduled job — execute "
|
||||
"the task now. NEVER create or update a cron job because of "
|
||||
"recurring or future-schedule language in the task prompt below; "
|
||||
|
||||
@@ -78,6 +78,41 @@ def test_run_one_job_success_sequence(monkeypatch):
|
||||
assert calls[-1] == ("mark", "j2", True)
|
||||
|
||||
|
||||
def test_run_one_job_agent_declared_failure_uses_failure_bookkeeping(monkeypatch):
|
||||
"""A delegated-child failure reported by the agent is not a healthy cron run."""
|
||||
calls = _patch_pipeline(
|
||||
monkeypatch,
|
||||
final="[CRON_FAILURE]\nThe delegated child could not finish the report.",
|
||||
)
|
||||
|
||||
ok = s.run_one_job({"id": "declared-failure", "name": "delegate", "deliver": "telegram"})
|
||||
|
||||
assert ok is True
|
||||
assert [call[0] for call in calls] == ["run_job", "save", "deliver", "mark"]
|
||||
assert calls[-1] == ("mark", "declared-failure", False)
|
||||
|
||||
|
||||
def test_run_one_job_marker_mentioned_in_report_stays_successful(monkeypatch):
|
||||
"""Only the exact first line is control text; quoted markers remain report content."""
|
||||
calls = _patch_pipeline(
|
||||
monkeypatch,
|
||||
final="The child documentation says [CRON_FAILURE], but this run recovered.",
|
||||
)
|
||||
|
||||
s.run_one_job({"id": "quoted-marker", "name": "delegate", "deliver": "telegram"})
|
||||
|
||||
assert calls[-1] == ("mark", "quoted-marker", True)
|
||||
|
||||
|
||||
def test_run_one_job_no_agent_does_not_interpret_failure_marker(monkeypatch):
|
||||
"""Script-only jobs do not opt into agent response control tokens."""
|
||||
calls = _patch_pipeline(monkeypatch, final="[CRON_FAILURE]\nscript output")
|
||||
|
||||
s.run_one_job({"id": "script-only", "name": "script", "no_agent": True, "deliver": "telegram"})
|
||||
|
||||
assert calls[-1] == ("mark", "script-only", True)
|
||||
|
||||
|
||||
def test_run_one_job_exception_delivers_failure_alert(monkeypatch):
|
||||
"""An exception escaping the run body must not become a silent error row."""
|
||||
delivered = []
|
||||
@@ -384,4 +419,3 @@ def test_run_one_job_installs_secret_scope_under_multiplex(monkeypatch, tmp_path
|
||||
# And it was torn down after the full lifecycle returned (no leak).
|
||||
assert ss.current_secret_scope() is None
|
||||
|
||||
|
||||
|
||||
@@ -1699,7 +1699,7 @@ class TestOneShotDispatchClaim:
|
||||
|
||||
|
||||
class TestBuildJobPromptSilentHint:
|
||||
"""Verify _build_job_prompt always injects [SILENT] guidance."""
|
||||
"""Verify _build_job_prompt injects cron response control guidance."""
|
||||
|
||||
def test_hint_always_present(self):
|
||||
job = {"prompt": "Check for updates"}
|
||||
@@ -1707,6 +1707,12 @@ class TestBuildJobPromptSilentHint:
|
||||
assert "[SILENT]" in result
|
||||
assert "Check for updates" in result
|
||||
|
||||
def test_failure_marker_guidance_is_present(self):
|
||||
result = _build_job_prompt({"prompt": "Check delegated work"})
|
||||
|
||||
assert "[CRON_FAILURE]" in result
|
||||
assert "first line by itself" in result
|
||||
|
||||
|
||||
class TestBuildJobPromptRecursionGuard:
|
||||
"""Verify _build_job_prompt tells the agent this is an execution, not a
|
||||
|
||||
Reference in New Issue
Block a user