fix(cron): record failure for BaseException escapes and leave a diagnostic when removing wedged one-shots

Fixes #73973.

A finite one-shot whose dispatch was claimed (claim_dispatch increments
repeat.completed BEFORE execution) but whose run died before mark_job_run
was left permanently wedged: completed==times, last_run_at null, state
'scheduled'. The run-claim TTL blocked re-dispatch, and once it expired
the dispatch-limit guard silently removed the job with no output and no
error.

Two complementary fixes:

- cron/scheduler.py run_one_job: the outer handler now catches
  BaseException, not just Exception. The inner run_job handler re-raises
  CancelledError/KeyboardInterrupt/SystemExit after agent teardown, and
  none of those are Exception subclasses, so the outer 'except Exception'
  missed them and mark_job_run(False) was never called. Failures are now
  recorded first (mark_job_run + finish_execution, each independently
  guarded), then non-Exception BaseExceptions are re-raised to preserve
  teardown semantics. Plain Exceptions keep the existing behavior
  (recorded, return False, no re-raise). Empty str(e) (bare
  CancelledError) falls back to the exception class name.

- cron/jobs.py: when either removal site (claim_dispatch or the
  get_due_jobs dispatch-limit guard) drops a one-shot whose claimed run
  never completed (last_run_at null), _write_wedged_oneshot_diagnostic
  now writes an operator-visible .md into cron/output/<job_id>/ instead
  of vanishing silently. Best-effort: diagnostics can never break the
  removal. No diagnostic when last_run_at is set (normal completion
  race).
This commit is contained in:
kshitijk4poor 2026-07-29 19:32:09 +05:00 • committed by kshitij
parent 41a07f5b84
commit cff9728587
4 changed files with 184 additions and 5 deletions

View file

@ -1909,6 +1909,57 @@ class TestClaimDispatch:
assert due == []
assert load_jobs() == [] # cleaned up
def test_get_due_jobs_stale_removal_writes_diagnostic(self, tmp_cron_dir, monkeypatch):
"""#73973: when the due-scan removes a wedged one-shot (dispatch claimed,
run never completed), it must leave an operator-visible diagnostic file
in the job's output dir instead of vanishing silently."""
import cron.jobs as jobs_mod
from cron.jobs import _hermes_now, _oneshot_run_claim_ttl_seconds
monkeypatch.delenv("HERMES_CRON_TIMEOUT", raising=False)
ttl = _oneshot_run_claim_ttl_seconds()
stale = (_hermes_now() - timedelta(seconds=ttl + 300)).isoformat()
save_jobs([{
"id": "wedged1",
"name": "wedged one-shot",
"enabled": True,
"schedule": {"kind": "once", "run_at": stale},
"repeat": {"times": 1, "completed": 1},
"run_claim": {"at": stale, "by": "dead-process:123"},
"next_run_at": stale,
}])
assert get_due_jobs() == []
assert load_jobs() == []
out_dir = jobs_mod._job_output_dir("wedged1")
files = list(out_dir.glob("*.md"))
assert files, "expected a diagnostic file in the output dir"
text = files[0].read_text(encoding="utf-8")
assert "removed without producing output" in text
assert "wedged1" in text
def test_claim_dispatch_stale_removal_writes_diagnostic(self, tmp_cron_dir):
"""#73973: the claim_dispatch removal path for an already-maxed one-shot
with no completed run also leaves a diagnostic."""
import cron.jobs as jobs_mod
save_jobs([self._oneshot(times=1, completed=1)])
assert claim_dispatch("os1") is False
assert load_jobs() == []
out_dir = jobs_mod._job_output_dir("os1")
files = list(out_dir.glob("*.md"))
assert files, "expected a diagnostic file in the output dir"
assert "removed without producing output" in files[0].read_text(encoding="utf-8")
def test_no_diagnostic_when_run_completed(self, tmp_cron_dir):
"""A one-shot that DID complete a run (last_run_at set) is a normal
completion race, not a wedge — no diagnostic file is written."""
import cron.jobs as jobs_mod
job = self._oneshot(times=1, completed=1)
job["last_run_at"] = "2026-01-01T00:05:00+00:00"
save_jobs([job])
assert claim_dispatch("os1") is False
assert load_jobs() == []
out_dir = jobs_mod._job_output_dir("os1")
assert not out_dir.exists() or not list(out_dir.glob("*.md"))
def test_bad_schedule_does_not_crash_or_block_sibling_jobs(self, tmp_cron_dir):
"""Regression for a job with non-dict 'schedule' (null / string / etc.

View file

@ -119,6 +119,57 @@ def test_run_one_job_exception_marks_failure(monkeypatch):
assert marks == [("j6", False)]
def test_run_one_job_base_exception_records_failure_then_reraises(monkeypatch):
"""#73973: a BaseException escaping run_job (CancelledError re-raised by the
inner teardown handler, KeyboardInterrupt, SystemExit) must still record the
failure via mark_job_run — otherwise a claim_dispatch()-consumed one-shot is
left wedged with completed==times but last_run_at never written. The
BaseException itself is re-raised after recording so shutdown semantics are
preserved."""
import asyncio
import pytest
for exc in (asyncio.CancelledError(), KeyboardInterrupt(), SystemExit(1)):
def boom(job, *, defer_agent_teardown=None, _exc=exc):
raise _exc
monkeypatch.setattr(s, "run_job", boom)
marks = []
monkeypatch.setattr(
s, "mark_job_run",
lambda jid, ok, err=None, delivery_error=None: marks.append((jid, ok, err)),
)
with pytest.raises(type(exc)):
s.run_one_job({"id": "jbase", "name": "t"})
assert marks and marks[0][0] == "jbase" and marks[0][1] is False, (
f"{type(exc).__name__}: failure was not recorded"
)
# Empty str(exc) (e.g. bare CancelledError) falls back to the class name.
assert marks[0][2], f"{type(exc).__name__}: error text must be non-empty"
def test_run_one_job_plain_exception_still_swallowed(monkeypatch):
"""The BaseException widening must not change plain-Exception behavior:
recorded, returns False, NOT re-raised."""
def boom(job, *, defer_agent_teardown=None):
raise ValueError("plain failure")
monkeypatch.setattr(s, "run_job", boom)
marks = []
monkeypatch.setattr(
s, "mark_job_run",
lambda jid, ok, err=None, delivery_error=None: marks.append((jid, ok)),
)
ok = s.run_one_job({"id": "jplain", "name": "t"})
assert ok is False
assert marks == [("jplain", False)]
def test_run_one_job_installs_secret_scope_under_multiplex(monkeypatch, tmp_path):
"""Regression: under profile isolation (multiplex active), run_one_job must
execute run_job inside a profile secret scope so credential reads