fix(cron): show run history for script-only jobs via per-fire output docs
Script-only (no_agent) jobs deliberately skip SessionDB, so their run history
was permanently empty — the desktop showed "No runs" next to hundreds of
completed fires. When no session rows exist, the runs endpoint now falls back to
the job's output docs under cron/output/<job_id>/, one row per fire, with the
latest status prefixed; with no surviving docs but a recorded last_run_at, a
single metadata row is surfaced instead. Rows mirror /api/sessions shape with
source='cron_output' and a cron_output: id prefix that cannot collide with
SessionDB cron_{job_id}_* session ids.
Filename timestamps are read back with hermes_time.get_timezone() — the same
configured zone save_job_output writes them with — not a fixed-offset snapshot
of today's local offset (wrong by hours when HERMES_TIMEZONE differs from the
server zone, and by an hour across DST).
Fixes #42433 (run-history facet)
Salvaged from #61403 by @LeonSGP43 (fallback design, metadata-only row, tests),
reworked per its review: timezone handling uses the configured zone and the
tests pin started_at under a configured-zone mismatch.
Co-authored-by: LeonSGP43 <cine.dreamer.one@gmail.com>
This commit is contained in:
@@ -7,7 +7,9 @@ late-binding seam so ``monkeypatch.setattr(web_server_cron, ...)`` keeps working
|
||||
|
||||
import asyncio
|
||||
import functools
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
@@ -21,6 +23,8 @@ from hermes_cli.web_server_cron import (
|
||||
)
|
||||
from hermes_cli.web_models import AutomationBlueprintInstantiate, CronJobCreate, CronJobUpdate
|
||||
from hermes_cli.web_routers._common import log as _log
|
||||
from hermes_time import get_timezone as _get_timezone
|
||||
from hermes_constants import get_hermes_home as _get_hermes_home
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
@@ -149,6 +153,175 @@ def _get_cron_job_sync(job_id: str, profile: Optional[str] = None):
|
||||
return _found(_call_cron_for_profile(_job_profile(job_id, profile), "get_job", job_id))
|
||||
|
||||
|
||||
_CRON_OUTPUT_FILENAME_FORMAT = "%Y-%m-%d_%H-%M-%S"
|
||||
|
||||
|
||||
def _cron_output_runs_dir(profile: Optional[str], job_id: str) -> Path:
|
||||
"""Output docs live under the job's home — resolve it even without a hint."""
|
||||
if profile:
|
||||
try:
|
||||
_, profile_home = _cron_profile_home(profile)
|
||||
except Exception:
|
||||
profile_home = _get_hermes_home()
|
||||
else:
|
||||
profile_home = _get_hermes_home()
|
||||
return Path(profile_home) / "cron" / "output" / job_id
|
||||
|
||||
|
||||
def _cron_output_run_timestamp(path: Path) -> Optional[float]:
|
||||
"""Epoch seconds for an output filename's wall time.
|
||||
|
||||
save_job_output writes the stem with hermes_time.now() — the configured
|
||||
IANA timezone (HERMES_TIMEZONE / config), falling back to server-local.
|
||||
Read it back with that same ZONE, not a fixed offset snapshot of today's
|
||||
local offset: a snapshot is wrong when the configured zone differs from
|
||||
the server's, and off by an hour for a historical file across a DST
|
||||
transition. With no configured zone, astimezone() on the naive datetime
|
||||
resolves the offset in effect on the filename's own date.
|
||||
"""
|
||||
try:
|
||||
naive = datetime.strptime(path.stem, _CRON_OUTPUT_FILENAME_FORMAT)
|
||||
except ValueError:
|
||||
return None
|
||||
tz = _get_timezone()
|
||||
if tz is not None:
|
||||
return naive.replace(tzinfo=tz).timestamp()
|
||||
return naive.astimezone().timestamp()
|
||||
|
||||
|
||||
def _cron_output_run_preview(path: Path, max_chars: int = 180) -> str:
|
||||
try:
|
||||
raw = path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return ""
|
||||
preview = re.sub(r"\s+", " ", raw).strip()
|
||||
if len(preview) <= max_chars:
|
||||
return preview
|
||||
return preview[: max_chars - 1].rstrip() + "…"
|
||||
|
||||
|
||||
def _cron_job_last_run_timestamp(job: Optional[Dict[str, Any]]) -> Optional[float]:
|
||||
if not isinstance(job, dict):
|
||||
return None
|
||||
raw = job.get("last_run_at")
|
||||
if isinstance(raw, (int, float)):
|
||||
return float(raw)
|
||||
if isinstance(raw, str):
|
||||
text = raw.strip()
|
||||
if not text:
|
||||
return None
|
||||
try:
|
||||
return datetime.fromisoformat(text.replace("Z", "+00:00")).timestamp()
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _cron_output_status_label(job: Optional[Dict[str, Any]]) -> str:
|
||||
if not isinstance(job, dict):
|
||||
return ""
|
||||
status = str(job.get("last_status") or "").strip()
|
||||
if not status:
|
||||
return ""
|
||||
return status.replace("_", " ").upper()
|
||||
|
||||
|
||||
def _list_cron_output_runs(
|
||||
job: Optional[Dict[str, Any]],
|
||||
canonical_job_id: str,
|
||||
profile: Optional[str],
|
||||
limit: int,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""SessionDB-less run history for jobs that never create agent sessions.
|
||||
|
||||
Script-only (no_agent) jobs deliberately skip SessionDB (cron/scheduler.run_job),
|
||||
so their completed runs are invisible to the run-history endpoint. Their
|
||||
output docs — one .md per fire under cron/output/<job_id>/ — are the only
|
||||
per-run record. Rows mirror /api/sessions shape with source='cron_output'
|
||||
so the frontend reuses SessionInfo; ids use a cron_output: prefix that can
|
||||
never collide with SessionDB cron_{job_id}_* session ids.
|
||||
"""
|
||||
output_dir = _cron_output_runs_dir(profile, canonical_job_id)
|
||||
try:
|
||||
files = sorted(
|
||||
(path for path in output_dir.glob("*.md") if path.is_file()),
|
||||
key=lambda path: path.name,
|
||||
reverse=True,
|
||||
)
|
||||
except OSError:
|
||||
files = []
|
||||
|
||||
latest_ts = _cron_job_last_run_timestamp(job)
|
||||
latest_status = _cron_output_status_label(job)
|
||||
runs: List[Dict[str, Any]] = []
|
||||
|
||||
for index, path in enumerate(files[:limit]):
|
||||
started_at = _cron_output_run_timestamp(path)
|
||||
if started_at is None:
|
||||
try:
|
||||
started_at = path.stat().st_mtime
|
||||
except OSError:
|
||||
started_at = 0.0
|
||||
preview = _cron_output_run_preview(path)
|
||||
title = preview or "Script-only run"
|
||||
if (
|
||||
index == 0
|
||||
and latest_status
|
||||
and (latest_ts is None or abs(latest_ts - started_at) <= 120)
|
||||
):
|
||||
title = f"{latest_status} · {title}"
|
||||
runs.append({
|
||||
"id": f"cron_output:{canonical_job_id}:{path.stem}",
|
||||
"title": title,
|
||||
"preview": preview or None,
|
||||
"source": "cron_output",
|
||||
"started_at": started_at,
|
||||
"last_active": started_at,
|
||||
"ended_at": started_at,
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"message_count": 0,
|
||||
"tool_call_count": 0,
|
||||
"model": None,
|
||||
"cwd": None,
|
||||
"archived": False,
|
||||
"is_active": False,
|
||||
})
|
||||
|
||||
if runs:
|
||||
return runs
|
||||
|
||||
# No output docs survived (pruned or never written) but the job HAS run:
|
||||
# surface one metadata-only row from last_run_at/last_status instead of the
|
||||
# bare "No runs" the issue reports.
|
||||
if latest_ts is None:
|
||||
return []
|
||||
|
||||
preview = ""
|
||||
if isinstance(job, dict):
|
||||
preview = str(job.get("last_error") or "").strip()
|
||||
title = _cron_output_status_label(job) or "Script-only run"
|
||||
if preview:
|
||||
title = f"{title} · {preview}"
|
||||
return [{
|
||||
"id": f"cron_output:{canonical_job_id}:latest",
|
||||
"title": title,
|
||||
"preview": preview or None,
|
||||
"source": "cron_output",
|
||||
"started_at": latest_ts,
|
||||
"last_active": latest_ts,
|
||||
"ended_at": latest_ts,
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"message_count": 0,
|
||||
"tool_call_count": 0,
|
||||
"model": None,
|
||||
"cwd": None,
|
||||
"archived": False,
|
||||
"is_active": False,
|
||||
}]
|
||||
|
||||
|
||||
def _list_cron_job_runs_sync(job_id: str, profile: Optional[str] = None, limit: int = 20):
|
||||
"""Run sessions produced by a cron job, newest first.
|
||||
|
||||
@@ -157,10 +330,13 @@ def _list_cron_job_runs_sync(job_id: str, profile: Optional[str] = None, limit:
|
||||
this job. Same row shape as ``/api/sessions`` so the frontend reuses
|
||||
SessionInfo. Backed by ``SessionDB.list_cron_job_runs`` — a bounded id-range
|
||||
scan, so cost scales with the requested window, not total cron history.
|
||||
Script-only (no_agent) jobs never write sessions: their history falls back
|
||||
to per-fire output docs (``_list_cron_output_runs``).
|
||||
"""
|
||||
selected = _job_owner_profile(job_id, profile)
|
||||
# job_id may be a human name; resolve to the canonical id used in run-session ids.
|
||||
canonical = job_id
|
||||
job = None
|
||||
if selected:
|
||||
job = _call_cron_for_profile(selected, "get_job", job_id)
|
||||
if job and job.get("id"):
|
||||
@@ -174,6 +350,8 @@ def _list_cron_job_runs_sync(job_id: str, profile: Optional[str] = None, limit:
|
||||
db = _open_session_db_for_profile(selected, read_only=True)
|
||||
try:
|
||||
runs = db.list_cron_job_runs(canonical, limit=limit_n, offset=0)
|
||||
if not runs:
|
||||
return {"runs": _list_cron_output_runs(job, canonical, selected, limit_n), "limit": limit_n}
|
||||
now = time.time()
|
||||
for s in runs:
|
||||
s["is_active"] = s.get("ended_at") is None and (now - s.get("last_active", s.get("started_at", 0))) < 300
|
||||
|
||||
@@ -3,6 +3,8 @@
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
import json
|
||||
import threading
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from fastapi import HTTPException
|
||||
@@ -1228,3 +1230,174 @@ class TestCronJobsCrossProfileDedup:
|
||||
prompts = {j.get("prompt") for j in result}
|
||||
|
||||
assert prompts == {"legacy job in default", "legacy job in worker"}
|
||||
|
||||
|
||||
class TestCronRunHistoryFallback:
|
||||
"""Script-only (no_agent) jobs deliberately skip SessionDB, so their run
|
||||
history was permanently empty ("No runs" despite hundreds of completed
|
||||
fires — #42433). When no session rows exist, the runs endpoint falls back
|
||||
to the job's per-fire output docs under cron/output/<job_id>/."""
|
||||
|
||||
@staticmethod
|
||||
def _script_only_job(job_id):
|
||||
return {"id": job_id, "no_agent": True, "script": "sync.sh", "prompt": ""}
|
||||
|
||||
@pytest.fixture()
|
||||
def no_session_rows(self, monkeypatch):
|
||||
class _EmptyDB:
|
||||
def list_cron_job_runs(self, job_id, limit=20, offset=0):
|
||||
return []
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_open_session_db_for_profile", lambda profile, *, read_only: _EmptyDB()
|
||||
)
|
||||
|
||||
def _write_output_doc(self, monkeypatch, home, job_id, filename, text):
|
||||
output_dir = Path(home) / "cron" / "output" / job_id
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
(output_dir / filename).write_text(text, encoding="utf-8")
|
||||
|
||||
def test_falls_back_to_output_docs_when_no_session_runs_exist(
|
||||
self, isolated_profiles, monkeypatch, no_session_rows
|
||||
):
|
||||
job_id = "job-script-only"
|
||||
home = isolated_profiles["default"]
|
||||
self._write_output_doc(monkeypatch, home, job_id, "2026-07-08_09-00-00.md", "older output\n")
|
||||
self._write_output_doc(monkeypatch, home, job_id, "2026-07-08_09-05-00.md", "latest output\n")
|
||||
|
||||
monkeypatch.setattr(_rt_cron, "_job_owner_profile", lambda _job_id, _profile: "default")
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_call_cron_for_profile",
|
||||
lambda _profile, cmd, *_args, **_kwargs: {
|
||||
**self._script_only_job(job_id), "last_status": "ok",
|
||||
} if cmd == "get_job" else None,
|
||||
)
|
||||
|
||||
result = _rt_cron._list_cron_job_runs_sync(job_id, limit=2)
|
||||
|
||||
runs = result["runs"]
|
||||
assert result["limit"] == 2
|
||||
assert [run["id"] for run in runs] == [
|
||||
f"cron_output:{job_id}:2026-07-08_09-05-00",
|
||||
f"cron_output:{job_id}:2026-07-08_09-00-00",
|
||||
]
|
||||
assert runs[0]["source"] == "cron_output"
|
||||
assert runs[0]["title"].startswith("OK · latest output")
|
||||
assert runs[1]["title"] == "older output"
|
||||
assert all(run["is_active"] is False for run in runs)
|
||||
|
||||
def test_output_doc_timestamps_honor_configured_timezone(
|
||||
self, isolated_profiles, monkeypatch, no_session_rows
|
||||
):
|
||||
"""The filename stem is written by save_job_output via hermes_time.now()
|
||||
— the configured zone. Reading it back must use that zone, not the
|
||||
server's (review of #61403: a fixed-offset snapshot was off by hours
|
||||
when HERMES_TIMEZONE differs from the server zone)."""
|
||||
from zoneinfo import ZoneInfo
|
||||
from hermes_time import _tz_cache
|
||||
|
||||
job_id = "job-tz"
|
||||
home = isolated_profiles["default"]
|
||||
# Wall time 2026-07-10_00-14-10 in Asia/Taipei (+08:00) = 2026-07-09 16:14:10 UTC.
|
||||
self._write_output_doc(monkeypatch, home, job_id, "2026-07-10_00-14-10.md", "tz output\n")
|
||||
|
||||
monkeypatch.setenv("HERMES_TIMEZONE", "Asia/Taipei")
|
||||
_tz_cache.clear()
|
||||
|
||||
monkeypatch.setattr(_rt_cron, "_job_owner_profile", lambda _job_id, _profile: "default")
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_call_cron_for_profile",
|
||||
lambda _profile, cmd, *_args, **_kwargs: {
|
||||
**self._script_only_job(job_id), "last_status": "ok",
|
||||
} if cmd == "get_job" else None,
|
||||
)
|
||||
|
||||
try:
|
||||
result = _rt_cron._list_cron_job_runs_sync(job_id, limit=5)
|
||||
finally:
|
||||
monkeypatch.delenv("HERMES_TIMEZONE", raising=False)
|
||||
_tz_cache.clear()
|
||||
|
||||
started = result["runs"][0]["started_at"]
|
||||
naive = datetime.strptime("2026-07-10_00-14-10", "%Y-%m-%d_%H-%M-%S")
|
||||
expected = naive.replace(tzinfo=ZoneInfo("Asia/Taipei")).timestamp()
|
||||
assert abs(started - expected) < 1, (
|
||||
f"Filename wall time misread: got epoch {started}, expected {expected}"
|
||||
)
|
||||
|
||||
def test_keeps_session_runs_when_db_history_exists(
|
||||
self, isolated_profiles, monkeypatch
|
||||
):
|
||||
job_id = "job-with-sessions"
|
||||
home = isolated_profiles["default"]
|
||||
self._write_output_doc(monkeypatch, home, job_id, "2026-07-08_09-05-00.md", "fallback output\n")
|
||||
|
||||
class _FakeDB:
|
||||
def list_cron_job_runs(self, canonical, limit=20, offset=0):
|
||||
return [{
|
||||
"id": f"cron_{canonical}_1", "source": "cron",
|
||||
"started_at": 123.0, "last_active": 125.0, "ended_at": 126.0, "archived": False,
|
||||
}]
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_open_session_db_for_profile", lambda profile, *, read_only: _FakeDB()
|
||||
)
|
||||
monkeypatch.setattr(_rt_cron, "_job_owner_profile", lambda _job_id, _profile: "default")
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_call_cron_for_profile",
|
||||
lambda _profile, cmd, *_args, **_kwargs: {"id": job_id} if cmd == "get_job" else None,
|
||||
)
|
||||
monkeypatch.setattr(_rt_cron.time, "time", lambda: 200.0)
|
||||
|
||||
result = _rt_cron._list_cron_job_runs_sync(job_id, limit=5)
|
||||
|
||||
assert [run["id"] for run in result["runs"]] == [f"cron_{job_id}_1"]
|
||||
assert result["runs"][0]["source"] == "cron"
|
||||
assert result["runs"][0]["is_active"] is False
|
||||
|
||||
def test_surfaces_latest_run_when_only_job_metadata_exists(
|
||||
self, isolated_profiles, monkeypatch, no_session_rows
|
||||
):
|
||||
"""No output docs survived pruning, but the job HAS completed runs:
|
||||
one metadata row from last_run_at/last_status beats a bare 'No runs'."""
|
||||
job_id = "job-last-run-only"
|
||||
monkeypatch.setattr(_rt_cron, "_job_owner_profile", lambda _job_id, _profile: "default")
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_call_cron_for_profile",
|
||||
lambda _profile, cmd, *_args, **_kwargs: {
|
||||
**self._script_only_job(job_id),
|
||||
"last_status": "ok",
|
||||
"last_run_at": "2026-07-08T09:05:00+00:00",
|
||||
} if cmd == "get_job" else None,
|
||||
)
|
||||
|
||||
result = _rt_cron._list_cron_job_runs_sync(job_id, limit=5)
|
||||
|
||||
runs = result["runs"]
|
||||
assert len(runs) == 1
|
||||
assert runs[0]["source"] == "cron_output"
|
||||
assert runs[0]["id"] == f"cron_output:{job_id}:latest"
|
||||
assert runs[0]["title"].startswith("OK")
|
||||
assert runs[0]["started_at"] == datetime.fromisoformat("2026-07-08T09:05:00+00:00").timestamp()
|
||||
|
||||
def test_agent_job_with_no_history_stays_empty(
|
||||
self, isolated_profiles, monkeypatch, no_session_rows
|
||||
):
|
||||
"""The fallback must not invent history for an agent job that simply
|
||||
hasn't run yet: no sessions, no output docs, no last_run_at → []."""
|
||||
job_id = "agent-job-never-run"
|
||||
monkeypatch.setattr(_rt_cron, "_job_owner_profile", lambda _job_id, _profile: "default")
|
||||
monkeypatch.setattr(
|
||||
_rt_cron, "_call_cron_for_profile",
|
||||
lambda _profile, cmd, *_args, **_kwargs: {"id": job_id} if cmd == "get_job" else None,
|
||||
)
|
||||
|
||||
result = _rt_cron._list_cron_job_runs_sync(job_id, limit=5)
|
||||
|
||||
assert result["runs"] == []
|
||||
|
||||
Reference in New Issue
Block a user