"""Schedule models and file store: revisions, lifecycle, history (T07).""" from __future__ import annotations from datetime import UTC, datetime from pathlib import Path import pytest from pydantic import ValidationError from wf_scheduling.models import ( CronTrigger, OccurrenceRecord, PendingCandidate, Schedule, ) from wf_scheduling.store import ( FileScheduleStore, ScheduleExistsError, StaleScheduleRevisionError, ) def _schedule(sid: str = "sched-1", **over: object) -> Schedule: now = datetime(2026, 9, 8, 12, 0, tzinfo=UTC) base: dict[str, object] = { "id": sid, "deployment_id": "dep-1", "trigger": {"kind": "cron", "expression": "0 * * * *", "timezone": "UTC"}, "input_bindings": [], "created_at": now.isoformat(), "updated_at": now.isoformat(), } base.update(over) return Schedule.model_validate(base) def test_schedule_defaults_match_settled_policies() -> None: sched = _schedule() assert sched.overlap == "skip" assert sched.misfire == "skip" assert sched.lateness_allowance_s == 60.0 assert sched.revision == 1 assert not sched.paused and not sched.deleted and not sched.exhausted assert sched.enabled def test_cron_trigger_and_lateness_validation() -> None: assert ( CronTrigger.model_validate( {"kind": "cron", "expression": "* * * * *", "timezone": "UTC"} ).timezone == "UTC" ) with pytest.raises(ValidationError): _schedule(lateness_allowance_s=-1.0) with pytest.raises(ValidationError): _schedule(lateness_allowance_s=float("inf")) with pytest.raises(ValidationError): Schedule.model_validate( { "id": "s", "deployment_id": "d", "trigger": {"kind": "oneshot", "at": "2026-09-08T12:00:00"}, "input_bindings": [], "created_at": "2026-09-08T12:00:00+00:00", "updated_at": "2026-09-08T12:00:00+00:00", } ) def test_parallel_requires_positive_max_active() -> None: with pytest.raises(ValidationError): _schedule(overlap="parallel", max_active_runs=0) def test_create_get_list_and_id_never_reused(tmp_path: Path) -> None: store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) store.create_schedule(_schedule("b")) assert [s.id for s in store.list_schedules()] == ["a", "b"] with pytest.raises(ScheduleExistsError): store.create_schedule(_schedule("a")) # Delete clears future admission but keeps history; id stays taken. doomed = store.get_schedule("a") doomed.deleted = True store.save_schedule(doomed) assert [s.id for s in store.list_schedules()] == ["b"] assert [s.id for s in store.list_schedules(include_deleted=True)] == ["a", "b"] with pytest.raises(ScheduleExistsError): store.create_schedule(_schedule("a")) def test_stale_revision_edits_rejected(tmp_path: Path) -> None: store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) current = store.get_schedule("a") updated = current.model_copy(update={"revision": 2, "paused": True}) store.update_schedule(updated, expected_revision=1) assert store.get_schedule("a").revision == 2 with pytest.raises(StaleScheduleRevisionError): store.update_schedule(updated, expected_revision=1) def test_pause_and_disable_exclude_interval_no_backfill(tmp_path: Path) -> None: store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) now = datetime(2026, 9, 8, 12, 30, tzinfo=UTC) sched = store.get_schedule("a") # Pause clears the pending candidate and advances the watermark; resume # starts from the next future instant (poll semantics tested in T08). store.save_candidate( PendingCandidate( schedule_id="a", intended_at=datetime(2026, 9, 8, 12, 0, tzinfo=UTC), revision=sched.revision, ), schedule_id="a", ) paused = sched.model_copy(update={"paused": True, "updated_at": now}) store.save_schedule(paused) store.save_candidate(None, schedule_id="a") store.save_consumed("a", now) assert store.get_candidate("a") is None assert store.get_consumed("a") == now def test_candidate_is_at_most_one(tmp_path: Path) -> None: store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) first = PendingCandidate( schedule_id="a", intended_at=datetime(2026, 9, 8, 12, 0, tzinfo=UTC), revision=1, ) second = PendingCandidate( schedule_id="a", intended_at=datetime(2026, 9, 8, 13, 0, tzinfo=UTC), revision=1, ) store.save_candidate(first, schedule_id="a") store.save_candidate(second, schedule_id="a") candidate = store.get_candidate("a") assert candidate is not None assert candidate.intended_at == datetime(2026, 9, 8, 13, 0, tzinfo=UTC) def test_history_pagination_over_resolved_utc(tmp_path: Path) -> None: store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) for hour in (12, 13, 14): store.append_history( OccurrenceRecord( schedule_id="a", occurrence_id=f"a|2026-09-08T{hour:02d}:00:00+00:00", kind="admitted", resolved_at=datetime(2026, 9, 8, hour, 0, tzinfo=UTC), run_id=f"run-{hour}", revision=1, created_at=datetime(2026, 9, 8, hour, 0, tzinfo=UTC), ) ) page = store.list_occurrences("a", limit=2) assert page["total"] == 3 next_cursor = page["next_cursor"] assert isinstance(next_cursor, str) assert "|" in next_cursor assert len(page["occurrences"]) == 2 # type: ignore[arg-type] second = store.list_occurrences( "a", cursor=next_cursor, limit=2, # type: ignore[arg-type] ) assert second["next_cursor"] is None assert len(second["occurrences"]) == 1 # type: ignore[arg-type] # Legacy integer offsets remain accepted. legacy = store.list_occurrences("a", cursor="2", limit=2) assert len(legacy["occurrences"]) == 1 # type: ignore[arg-type] def test_inspection_payload_carries_contract_fields(tmp_path: Path) -> None: store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) store.append_history( OccurrenceRecord( schedule_id="a", occurrence_id="a|2026-09-08T12:00:00+00:00", kind="skipped-overlap", resolved_at=datetime(2026, 9, 8, 12, 0, tzinfo=UTC), revision=2, reason="active=['run-1']", admitted_at=None, started_at=None, created_at=datetime(2026, 9, 8, 12, 0, tzinfo=UTC), ) ) page = store.list_occurrences("a", limit=10) row = page["occurrences"][0] # type: ignore[index] assert row["resolved_at"] is not None assert row["revision"] == 2 assert row["reason"] == "active=['run-1']" def _tied_entry( kind: str, checkpoint_id: str | None, created_minute: int, ) -> OccurrenceRecord: """One history entry for the shared 12:00 occurrence of schedule a.""" return OccurrenceRecord( schedule_id="a", occurrence_id="a|2026-09-08T12:00:00+00:00", kind=kind, # type: ignore[arg-type] resolved_at=datetime(2026, 9, 8, 12, 0, tzinfo=UTC), run_id="run-1", revision=1, checkpoint_id=checkpoint_id, created_at=datetime(2026, 9, 8, 12, created_minute, tzinfo=UTC), ) def _traverse(store: FileScheduleStore, limit: int) -> list[dict]: """Walk every page to the end, returning all rows in visit order.""" rows: list[dict] = [] cursor: str | None = None for _ in range(1000): page = store.list_occurrences("a", cursor=cursor, limit=limit) rows.extend(page["occurrences"]) # type: ignore[arg-type] cursor = page["next_cursor"] # type: ignore[assignment] if cursor is None: assert page["total"] == len(rows) return rows raise AssertionError(f"pagination did not terminate; visited {len(rows)} rows") def test_history_pagination_visits_every_tied_entry_once(tmp_path: Path) -> None: """Admission + repeated interruptions + completion share one cursor tie. Every stored entry must appear exactly once across small-page traversal (B4): the cursor carries the entry ordinal, not just the shared ``(resolved_at, occurrence_id)`` tie. """ store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) store.append_history(_tied_entry("admitted", None, 0)) store.append_history(_tied_entry("interrupted", "run-1.000001", 5)) store.append_history(_tied_entry("interrupted", "run-1.000002", 9)) store.append_history(_tied_entry("completed", "run-1.000003", 14)) rows = _traverse(store, limit=1) assert [(row["kind"], row["checkpoint_id"]) for row in rows] == [ ("admitted", None), ("interrupted", "run-1.000001"), ("interrupted", "run-1.000002"), ("completed", "run-1.000003"), ] first = store.list_occurrences("a", limit=1) assert first["next_cursor"] is not None assert len(str(first["next_cursor"]).split("|")) >= 3 def test_history_pagination_legacy_rows_keep_file_order(tmp_path: Path) -> None: """Rows persisted before the entry ordinal order by file position. Real persisted data without ``seq`` must still traverse exactly once; pre-ordinal two-part cursors stay accepted and resume after the tied group exactly as they did before (no duplicates, no crash). """ import json store = FileScheduleStore(tmp_path) store.create_schedule(_schedule("a")) history_path = tmp_path / "schedules" / "a" / "history.json" history_path.parent.mkdir(parents=True, exist_ok=True) legacy = [ _tied_entry("admitted", None, 0).model_dump(mode="json"), _tied_entry("interrupted", "run-1.000001", 5).model_dump(mode="json"), ] for item in legacy: del item["seq"] history_path.write_text(json.dumps(legacy), encoding="utf-8") rows = _traverse(store, limit=1) assert [row["kind"] for row in rows] == ["admitted", "interrupted"] legacy_cursor = "2026-09-08T12:00:00+00:00|a|2026-09-08T12:00:00+00:00" resumed = store.list_occurrences("a", cursor=legacy_cursor, limit=1) assert resumed["occurrences"] == [] assert resumed["next_cursor"] is None with pytest.raises(ValueError): store.list_occurrences("a", cursor="a|b|c|d", limit=1)