mirror of
https://github.com/HKUDS/nanobot.git
synced 2026-09-04 10:11:46 +03:00
fix(cron): prevent replay after persistence failure
This commit is contained in:
@@ -1,6 +1,7 @@
|
||||
import asyncio
|
||||
import json
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -960,8 +961,8 @@ def test_stale_instance_remove_preserves_external_add(tmp_path) -> None:
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_save_store_failure_does_not_kill_scheduler(tmp_path, monkeypatch):
|
||||
"""A persistence failure in _on_timer must not stop future ticks."""
|
||||
async def test_save_store_failure_retries_without_replaying_job(tmp_path, monkeypatch):
|
||||
"""A failed post-run save must be retried before jobs can execute again."""
|
||||
store_path = tmp_path / "cron" / "jobs.json"
|
||||
calls: list[str] = []
|
||||
arm_calls: list[str] = []
|
||||
@@ -990,25 +991,51 @@ async def test_save_store_failure_does_not_kill_scheduler(tmp_path, monkeypatch)
|
||||
service._save_store()
|
||||
arm_calls.clear()
|
||||
|
||||
# Simulate a disk-write failure on the next tick.
|
||||
def failing_save() -> None:
|
||||
raise OSError("disk full")
|
||||
real_atomic_write = service._atomic_write
|
||||
save_attempts = 0
|
||||
writes_fail = True
|
||||
|
||||
monkeypatch.setattr(service, "_save_store", failing_save)
|
||||
def flaky_atomic_write(path: Path, content: str) -> None:
|
||||
nonlocal save_attempts, writes_fail
|
||||
save_attempts += 1
|
||||
if writes_fail:
|
||||
raise OSError("disk full")
|
||||
real_atomic_write(path, content)
|
||||
|
||||
monkeypatch.setattr(service, "_atomic_write", flaky_atomic_write)
|
||||
await service._on_timer()
|
||||
|
||||
# The scheduler must still be re-armed after the failed save...
|
||||
# The failed tick stays alive and retains the advanced in-memory state,
|
||||
# including when a public read would normally reload from disk.
|
||||
assert arm_calls == ["arm"], "scheduler must re-arm after a failed tick"
|
||||
assert service._active_executions == 0
|
||||
# ...the first tick already ran the due job...
|
||||
assert calls == [job.id]
|
||||
# ...and a later, healthy tick must still run the job again.
|
||||
monkeypatch.setattr(service, "_save_store", CronService._save_store.__get__(service))
|
||||
job.state.next_run_at_ms = max(1, int(time.time() * 1000) - 1_000)
|
||||
assert service._store_dirty is True
|
||||
loaded = service.get_job(job.id)
|
||||
assert loaded is not None
|
||||
assert loaded.state.last_run_at_ms is not None
|
||||
|
||||
# Manual execution is a second side-effecting entrypoint. It must also
|
||||
# refuse to run until the previous result can be made durable.
|
||||
with pytest.raises(OSError, match="disk full"):
|
||||
await service.run_job(job.id, force=True)
|
||||
assert calls == [job.id]
|
||||
|
||||
# The next healthy tick is reserved for persisting the dirty snapshot. It
|
||||
# must not reload the stale due record or execute the side effect twice.
|
||||
writes_fail = False
|
||||
await service._on_timer()
|
||||
|
||||
assert calls == [job.id, job.id]
|
||||
assert arm_calls == ["arm", "arm"]
|
||||
assert calls == [job.id]
|
||||
assert arm_calls == ["arm", "arm", "arm"]
|
||||
assert save_attempts == 3
|
||||
assert service._store_dirty is False
|
||||
|
||||
persisted = CronService(store_path).get_job(job.id)
|
||||
assert persisted is not None
|
||||
assert persisted.state.last_run_at_ms is not None
|
||||
assert persisted.state.next_run_at_ms is not None
|
||||
assert persisted.state.next_run_at_ms > persisted.state.last_run_at_ms
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
Reference in New Issue
Block a user