diff --git a/cron/jobs.py b/cron/jobs.py index bdad991631..5a749440e3 100644 --- a/cron/jobs.py +++ b/cron/jobs.py @@ -2055,6 +2055,11 @@ def update_job(job_id: str, updates: Dict[str, Any]) -> Optional[Dict[str, Any]] if "schedule" in updates: _apply_schedule_update(updated, updates, job_id) + # next_run_at now follows the new schedule; a stale quota_hold_until would only shield + # the record from the stale-error re-arm while no longer describing where it is + # parked. The next fire re-parks (with a fresh notice) if the window is still closed. + from cron.quota_hold import clear_state as _clear_quota_hold + _clear_quota_hold(updated) if {"schedule", "next_run_at", "enabled", "state"}.intersection(updates): # An explicit schedule/lifecycle rewrite supersedes any occurrence the dispatcher # left unclaimed — pause/resume/edit must not resurrect a slot from before the edit. diff --git a/tests/cron/test_quota_hold.py b/tests/cron/test_quota_hold.py index 804bd22b5c..73fbe91edd 100644 --- a/tests/cron/test_quota_hold.py +++ b/tests/cron/test_quota_hold.py @@ -105,3 +105,11 @@ def test_quota_hold_parks_past_window_survives_stale_rearm_and_clears_on_model_r j = get_job(job_id) assert qh.STATE_KEY not in j assert datetime.fromisoformat(j["next_run_at"]) - now < timedelta(hours=1) + + # Editing the schedule recomputes next_run_at from the new cadence; the stale marker must + # not linger on a record that is no longer parked where it says. + assert mark_job_run(job_id, False, QUOTA_MSG, quota_hold_seconds=123518) + assert qh.STATE_KEY in get_job(job_id) + j = update_job(job_id, {"schedule": "every 15m"}) + assert qh.STATE_KEY not in j + assert datetime.fromisoformat(j["next_run_at"]) - now < timedelta(hours=1) diff --git a/website/docs/user-guide/features/cron.md b/website/docs/user-guide/features/cron.md index bf16096d15..982c289a47 100644 --- a/website/docs/user-guide/features/cron.md +++ b/website/docs/user-guide/features/cron.md @@ -449,11 +449,13 @@ cron: ### Holding a job through a closed provider usage window -The mirror case: the provider says exactly how long it will stay closed. A -subscription provider whose usage limit is exhausted rejects every request -with a 429 and a `retry after s` hint (often many hours). When the whole -fallback chain is unavailable, re-firing a sub-hourly job into that window is -guaranteed to fail identically on every tick — and to alert every time. +The mirror case: the provider says exactly how long it will stay closed. When +the scheduler resolves a subscription provider (currently the OpenAI Codex +usage probe) and the provider reports its usage limit exhausted with a +`retry after s` hint (often many hours), and the whole fallback chain is +unavailable, re-firing a sub-hourly job into that window is guaranteed to fail +identically on every tick — and to alert every time. A 429 the model API +returns mid-run is not held this way; it is retried on the normal cadence. Instead, the scheduler **parks the job**: the one failure alert says the window is closed and that the job is held, `next_run_at` moves to the first