Files
lda-wf/tests/scheduling/test_poll.py
T

657 lines
23 KiB
Python

"""Scheduler poll loop against real file stores (T08)."""
from __future__ import annotations
from datetime import UTC, datetime, timedelta
from pathlib import Path
from typing import Any, cast
from tests.scheduling.controlled import (
DictDeployments,
ScriptedDispatcher,
fixture_environment,
)
from wf_artifacts.runs.models import StoredRunStatus
from wf_artifacts.runs.store import FileRunStore
from wf_scheduling.calendar import OneShotSource
from wf_scheduling.models import Schedule
from wf_scheduling.ownership import SchedulerOwnership
from wf_scheduling.poll import SCAN_CAP, Scheduler
from wf_scheduling.prepare import SchedulePreparer
from wf_scheduling.store import FileScheduleStore
UTC_TZ = UTC
def ts(y: int, mo: int, d: int, h: int = 0, mi: int = 0, s: int = 0) -> datetime:
return datetime(y, mo, d, h, mi, s, tzinfo=UTC_TZ)
class PeriodicSource:
def __init__(self, period: timedelta, start: datetime) -> None:
self.period = period
self.start = start
self.next_calls = 0
self.prev_calls = 0
def next_after(self, instant: datetime) -> datetime | None:
self.next_calls += 1
if instant < self.start:
return self.start
n = (instant - self.start) // self.period + 1
return self.start + n * self.period
def prev_before(self, instant: datetime) -> datetime | None:
self.prev_calls += 1
if instant <= self.start:
return None
n = (instant - self.start - timedelta(microseconds=1)) // self.period
return self.start + n * self.period
def _sched_model(sid: str, start_hint: str = "cron", **kw: Any) -> Schedule:
now = ts(2026, 9, 8, 12, 0)
base: dict[str, Any] = {
"id": sid,
"deployment_id": "dep-1",
"trigger": {"kind": "cron", "expression": "0 * * * *", "timezone": "UTC"},
"input_bindings": [],
"created_at": now.isoformat(),
"updated_at": now.isoformat(),
}
base.update(kw)
return Schedule.model_validate(base)
def _harness(
tmp_path: Path,
*,
capacity: int = 4,
script: dict | None = None,
deployments: dict[str, dict] | None = None,
) -> tuple[Scheduler, FileScheduleStore, FileRunStore, dict]:
sched_store = FileScheduleStore(tmp_path / "sched")
run_store = FileRunStore(tmp_path / "runs")
sources: dict = {}
preparer = SchedulePreparer(
DictDeployments(
deployments
if deployments is not None
else {"dep-1": {"rev": 1, "required": []}}
),
fixture_environment,
)
sched = Scheduler(
schedule_store=sched_store,
run_store=run_store,
sources=sources,
capacity=capacity,
preparer=preparer,
dispatcher=ScriptedDispatcher(script),
ownership=SchedulerOwnership(tmp_path, owner="test").acquire(),
)
return sched, sched_store, run_store, sources
def _add(
sched: Scheduler,
store: FileScheduleStore,
sources: dict,
sid: str,
src: Any,
consumed: datetime,
**kw: Any,
) -> Any:
model = _sched_model(sid, **kw)
store.create_schedule(model)
store.save_consumed(sid, consumed)
sources[sid] = src
return store.get_schedule(sid)
def _history(store: FileScheduleStore, sid: str) -> list[dict]:
page = store.list_occurrences(sid, limit=100)
return page["occurrences"] # type: ignore[return-value]
def test_overlap_skip_blocks_and_late_drops() -> None:
import tempfile
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
sched, store, runs, sources = _harness(root, capacity=4, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
_add(
sched,
store,
sources,
"a",
PeriodicSource(timedelta(minutes=10), t0),
t0 - timedelta(minutes=10),
)
sched.poll(t0)
assert len(runs.list_runs()) == 1
sched.poll(t0 + timedelta(minutes=10))
kinds = [(r["kind"], r["resolved_at"]) for r in _history(store, "a")]
assert any(k == "skipped-overlap" and "12:10" in str(v) for k, v in kinds)
sched.poll(t0 + timedelta(minutes=30))
kinds = [(r["kind"], r["resolved_at"]) for r in _history(store, "a")]
assert any(k == "skipped-misfire" for k, _ in kinds)
sched.ownership.release()
def test_latest_coalesces_to_one_candidate_and_no_double_admit() -> None:
import tempfile
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
sched, store, runs, sources = _harness(root, capacity=0)
_add(
sched,
store,
sources,
"h",
PeriodicSource(timedelta(hours=1), ts(2026, 9, 8, 9, 0)),
ts(2026, 9, 8, 9, 0),
misfire="latest",
)
sched.poll(ts(2026, 9, 8, 12, 20))
cand = store.get_candidate("h")
assert cand is not None and cand.intended_at == ts(2026, 9, 8, 12, 0)
assert [r for r in _history(store, "h") if r["kind"] == "admitted"] == []
# Capacity returns at 13:00 while 13:00 is also due: exactly one
# admission for 13:00, 12:00 superseded, never both.
sched.capacity = 4
cast(ScriptedDispatcher, sched.dispatcher).script = {"*": "complete"}
sched.poll(ts(2026, 9, 8, 13, 0))
admitted = [r for r in _history(store, "h") if r["kind"] == "admitted"]
assert len(admitted) == 1
assert datetime.fromisoformat(admitted[0]["resolved_at"]) == ts(
2026, 9, 8, 13, 0
)
assert store.get_candidate("h") is None
sched.ownership.release()
def test_parallel_limits_and_interrupted_slots() -> None:
import tempfile
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
sched, store, runs, sources = _harness(root, capacity=4, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
_add(
sched,
store,
sources,
"p",
PeriodicSource(timedelta(minutes=5), t0),
t0 - timedelta(minutes=5),
overlap="parallel",
max_active_runs=2,
)
sched.poll(t0)
assert sched._task_load() == 1
first = runs.list_runs()[0]
# A waiting interruption holds no task slot: production dispatch
# clears the executing mark when persisting the stopped result, so
# the simulated interruption clears it too.
runs.save_run(first.model_copy(update={"status": StoredRunStatus.INTERRUPTED}))
runs.clear_executing(first.id)
assert sched._task_load() == 0
sched.poll(t0 + timedelta(minutes=5))
assert len(runs.list_runs()) == 2
sched.poll(t0 + timedelta(minutes=10))
assert any(r["kind"] == "skipped-overlap" for r in _history(store, "p"))
# Lowering the limit never cancels; blocks new admission instead.
sched_model = store.get_schedule("p")
sched_model.max_active_runs = 1
store.save_schedule(sched_model)
second = [r for r in runs.list_runs() if r.id != first.id][0]
runs.save_run(second.model_copy(update={"status": StoredRunStatus.INTERRUPTED}))
runs.clear_executing(second.id)
sched.poll(t0 + timedelta(minutes=15))
assert any(
r["kind"] == "skipped-overlap" and "12:15" in str(r["resolved_at"])
for r in _history(store, "p")
)
sched.ownership.release()
def test_pause_is_not_downtime() -> None:
import tempfile
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
sched, store, runs, sources = _harness(
root, capacity=4, script={"*": "complete"}
)
_add(
sched,
store,
sources,
"a",
PeriodicSource(timedelta(hours=1), ts(2026, 9, 8, 9, 0)),
ts(2026, 9, 8, 9, 0),
misfire="latest",
)
sched.poll(ts(2026, 9, 8, 9, 0))
model = store.get_schedule("a")
model.paused = True
store.save_schedule(model)
sched.poll(ts(2026, 9, 8, 10, 30))
assert store.get_candidate("a") is None
sched.resume_schedule("a", ts(2026, 9, 8, 12, 30))
sched.poll(ts(2026, 9, 8, 12, 30))
admitted = [
r["resolved_at"] for r in _history(store, "a") if r["kind"] == "admitted"
]
assert "2026-09-08T10:00:00+00:00" not in admitted
assert "2026-09-08T11:00:00+00:00" not in admitted
sched.ownership.release()
def test_long_downtime_is_bounded(tmp_path: Path) -> None:
sched, store, runs, sources = _harness(
tmp_path, capacity=4, script={"*": "complete"}
)
src = PeriodicSource(timedelta(minutes=1), ts(2023, 9, 8, 12, 0))
_add(sched, store, sources, "m", src, ts(2023, 9, 8, 12, 0), misfire="latest")
sched.poll(ts(2026, 9, 8, 12, 0, 0))
assert src.next_calls + src.prev_calls <= SCAN_CAP + 2
admitted = [r for r in _history(store, "m") if r["kind"] == "admitted"]
assert len(admitted) == 1
# Latest-eligible <= now: a poll exactly at a due instant selects that
# instant (F3), not its exclusive predecessor.
assert datetime.fromisoformat(admitted[0]["resolved_at"]) == ts(2026, 9, 8, 12, 0)
assert (
len([r for r in _history(store, "m") if r["kind"] == "interval-summary"]) == 1
)
# A second poll at the same instant admits nothing more.
sched.poll(ts(2026, 9, 8, 12, 0, 0))
assert len([r for r in _history(store, "m") if r["kind"] == "admitted"]) == 1
sched.ownership.release()
def test_fairness_slow_schedule_not_starved(tmp_path: Path) -> None:
sched, store, runs, sources = _harness(tmp_path, capacity=1, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
_add(
sched,
store,
sources,
"fast",
PeriodicSource(timedelta(minutes=1), t0),
t0 - timedelta(minutes=1),
)
_add(sched, store, sources, "slow", OneShotSource(t0), t0 - timedelta(hours=1))
sched.poll(t0)
assert any(
r["kind"] == "admitted" and r["schedule_id"] == "fast"
for r in _history(store, "fast")
)
for run in runs.list_runs():
runs.save_run(run.model_copy(update={"status": StoredRunStatus.COMPLETED}))
runs.clear_executing(run.id)
sched.poll(t0 + timedelta(seconds=30))
assert any(r["kind"] == "admitted" for r in _history(store, "slow"))
sched.ownership.release()
def test_capacity_wait_then_expire_for_skip(tmp_path: Path) -> None:
sched, store, runs, sources = _harness(tmp_path, capacity=0)
t0 = ts(2026, 9, 8, 12, 0)
_add(sched, store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1))
assert sched.poll(t0) == {"a": "admit:held-undecided"}
assert store.get_candidate("a") is None
sched.capacity = 1
cast(ScriptedDispatcher, sched.dispatcher).script = {"*": "complete"}
sched.poll(t0 + timedelta(seconds=30))
assert len([r for r in _history(store, "a") if r["kind"] == "admitted"]) == 1
sched.ownership.release()
def test_poll_one_rereads_pause_before_deciding(tmp_path: Path) -> None:
sched, store, runs, sources = _harness(tmp_path, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
_add(sched, store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1))
stale = store.get_schedule("a")
# Same-process administration lands after the tick listed schedules:
# the decision must observe the pause, not the stale snapshot.
live = store.get_schedule("a")
live.paused = True
store.save_schedule(live)
assert sched._poll_one(stale, t0 + timedelta(seconds=1)) == "paused"
assert runs.list_runs() == []
assert runs.list_admissions() == []
sched.ownership.release()
def test_poll_one_uses_fresh_definition_after_edit(tmp_path: Path) -> None:
sched, store, runs, sources = _harness(tmp_path, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
_add(
sched,
store,
sources,
"a",
OneShotSource(t0),
t0 - timedelta(hours=1),
input_bindings=[
{"target": "team", "expression": {"kind": "literal", "value": "old"}}
],
)
stale = store.get_schedule("a")
live = store.get_schedule("a")
live.revision = 2
from wf_core.models.input_bindings import ScheduleInputBinding
live.input_bindings = [
ScheduleInputBinding.model_validate(
{"target": "team", "expression": {"kind": "literal", "value": "new"}}
)
]
store.save_schedule(live)
result = sched._poll_one(stale, t0 + timedelta(seconds=1))
assert result.startswith("admit:run-")
run_id = result.split(":", 1)[1]
admission = runs.get_admission(run_id)
assert admission.resolved_input["team"] == "new"
assert admission.schedule_revision == 2
sched.ownership.release()
def test_manual_and_other_schedule_runs_are_overlap_independent(
tmp_path: Path,
) -> None:
from wf_api.run_lifecycle import (
materialize_admitted_view,
persist_admission,
)
sched, store, runs, sources = _harness(tmp_path, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
# An active run of another schedule occupies only its own slot.
_add(sched, store, sources, "b", OneShotSource(t0), t0 - timedelta(hours=1))
assert sched.poll(t0)["b"].startswith("admit:run-")
# A manual run (no schedule owner) participates in no overlap check.
manual_id = runs.allocate_run_id()
manual = persist_admission(
store=runs,
run_id=manual_id,
environment=fixture_environment(object()),
resolved_input={},
max_steps=None,
)
materialize_admitted_view(store=runs, admission=manual)
# A due one-shot admits despite both unrelated active runs.
_add(sched, store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1))
result = sched.poll(t0 + timedelta(seconds=1))
assert result["a"].startswith("admit:run-")
assert result["b"] == "exhausted"
sched.ownership.release()
def test_manual_admission_does_not_consume_scheduler_capacity(tmp_path: Path) -> None:
"""Manual runs bypass scheduler capacity; scheduled work still dispatches."""
from wf_api.run_lifecycle import materialize_admitted_view, persist_admission
sched, store, runs, sources = _harness(tmp_path, capacity=1, script={"*": "hang"})
t0 = ts(2026, 9, 8, 12, 0)
manual = persist_admission(
store=runs,
run_id=runs.allocate_run_id(),
environment=fixture_environment(object()),
resolved_input={},
max_steps=None,
)
materialize_admitted_view(store=runs, admission=manual)
assert sched._task_load() == 0
_add(sched, store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1))
assert sched.poll(t0)["a"].startswith("admit:run-")
sched.ownership.release()
def test_existing_admission_rebuilds_missing_view_and_pending_marker(
tmp_path: Path,
) -> None:
"""A history tear after admission remains dispatchable on the next poll."""
class FailHistoryOnce(FileScheduleStore):
def __init__(self, root: Path) -> None:
super().__init__(root)
self.armed = True
def append_history(self, record: Any) -> None:
if self.armed:
self.armed = False
raise OSError("injected history failure")
super().append_history(record)
sched_store = FailHistoryOnce(tmp_path / "sched")
run_store = FileRunStore(tmp_path / "runs")
sources: dict[str, Any] = {}
t0 = ts(2026, 9, 8, 12, 0)
sched = Scheduler(
schedule_store=sched_store,
run_store=run_store,
sources=sources,
capacity=1,
preparer=SchedulePreparer(
DictDeployments({"dep-1": {"rev": 1, "required": []}}),
fixture_environment,
),
dispatcher=ScriptedDispatcher({"*": "complete"}),
ownership=SchedulerOwnership(tmp_path, owner="test").acquire(),
)
try:
_add(
sched, sched_store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1)
)
try:
sched.poll(t0)
raise AssertionError("history failure must propagate")
except OSError as exc:
assert "injected history failure" in str(exc)
[admission] = run_store.list_admissions()
assert run_store.list_runs() == []
assert not run_store.is_pending_dispatch(admission.id)
result = sched.poll(t0 + timedelta(seconds=1))
assert result["a"] == f"admit:{admission.id}"
assert run_store.get_run(admission.id).status.value == "admitted"
assert run_store.is_pending_dispatch(admission.id)
sched.poll(t0 + timedelta(seconds=2))
assert run_store.get_run(admission.id).status.value == "completed"
assert not run_store.is_pending_dispatch(admission.id)
finally:
sched.ownership.release()
def test_oneshot_terminal_bookkeeping_is_idempotent_after_torn_write(
tmp_path: Path,
) -> None:
"""A torn exhausted transition does not duplicate history or retain a candidate."""
class FailScheduleOnce(FileScheduleStore):
def __init__(self, root: Path) -> None:
super().__init__(root)
self.armed = False
def save_schedule(self, schedule: Any) -> None:
if self.armed:
self.armed = False
raise OSError("injected schedule failure")
super().save_schedule(schedule)
sched_store = FailScheduleOnce(tmp_path / "sched")
run_store = FileRunStore(tmp_path / "runs")
sources: dict[str, Any] = {}
t0 = ts(2026, 9, 8, 12, 0)
sched = Scheduler(
schedule_store=sched_store,
run_store=run_store,
sources=sources,
capacity=1,
preparer=SchedulePreparer(
DictDeployments({"dep-1": {"rev": 1, "required": []}}),
fixture_environment,
),
dispatcher=ScriptedDispatcher({"*": "complete"}),
ownership=SchedulerOwnership(tmp_path, owner="test").acquire(),
)
try:
_add(
sched, sched_store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1)
)
from wf_scheduling.models import PendingCandidate
sched_store.save_candidate(
PendingCandidate(schedule_id="a", intended_at=t0, revision=1),
schedule_id="a",
)
sched_store.armed = True
try:
sched.poll(t0 + timedelta(minutes=5))
raise AssertionError("schedule failure must propagate")
except OSError as exc:
assert "injected schedule failure" in str(exc)
assert (
len([r for r in _history(sched_store, "a") if r["kind"] == "exhausted"])
== 1
)
assert sched_store.get_candidate("a") is not None
assert sched.poll(t0 + timedelta(minutes=6))["a"] == "exhausted"
assert (
len([r for r in _history(sched_store, "a") if r["kind"] == "exhausted"])
== 1
)
assert sched_store.get_candidate("a") is None
finally:
sched.ownership.release()
def test_corrupt_unattributed_view_isolated_from_healthy_schedules(
tmp_path: Path,
) -> None:
"""A view without an admission fails closed without blocking siblings."""
from wf_artifacts import ResumeReadiness, WorkflowRunRecord
from wf_artifacts.runs.models import StoredRunStatus
sched, store, runs, sources = _harness(tmp_path, script={"*": "complete"})
t0 = ts(2026, 9, 8, 12, 0)
_add(sched, store, sources, "a", OneShotSource(t0), t0 - timedelta(hours=1))
_add(sched, store, sources, "b", OneShotSource(t0), t0 - timedelta(hours=1))
corrupt_id = runs.allocate_run_id()
runs.save_run(
WorkflowRunRecord(
id=corrupt_id,
status=StoredRunStatus.ADMITTED,
resume_readiness=ResumeReadiness.NOT_APPLICABLE,
environment=fixture_environment(object()),
latest_checkpoint_id=None,
created_at=t0,
updated_at=t0,
)
)
results = sched.poll(t0)
assert results["a"].startswith("admit:run-")
assert results["b"].startswith("admit:run-")
assert runs.get_run(corrupt_id).status.value == "failed"
assert all(
runs.get_run(run.id).status.value == "completed"
for run in runs.list_runs()
if run.id != corrupt_id
)
sched.ownership.release()
def test_record_resumed_stopped_result_is_idempotent_and_attributed(
tmp_path: Path,
) -> None:
"""Live resumed results record exactly once with schedule attribution (B3)."""
from tests.scheduling.controlled import fixture_environment
from wf_api.run_lifecycle import persist_admission
sched, store, runs, sources = _harness(tmp_path, script={"*": "interrupt"})
t0 = ts(2026, 9, 8, 12, 0)
_add(sched, store, sources, "a", OneShotSource(t0), t0 - timedelta(seconds=1))
assert sched.poll(t0)["a"].startswith("admit:run-")
[run] = runs.list_runs()
assert run.status.value == "interrupted"
assert len([r for r in _history(store, "a") if r["kind"] == "interrupted"]) == 1
now = t0 + timedelta(minutes=1)
resumed_ckpt = f"{run.id}.000002"
assert (
sched.record_resumed_stopped_result(
run.id,
status_value="completed",
checkpoint_id=resumed_ckpt,
now=now,
)
is True
)
# Repeats (poll retries, restart recovery) never duplicate the entry.
assert (
sched.record_resumed_stopped_result(
run.id,
status_value="completed",
checkpoint_id=resumed_ckpt,
now=now,
)
is False
)
completed = [r for r in _history(store, "a") if r["kind"] == "completed"]
assert len(completed) == 1
assert completed[0]["run_id"] == run.id
assert completed[0]["checkpoint_id"] == resumed_ckpt
assert completed[0]["revision"] == 1
assert completed[0]["reason"] == "resumed"
# A resumed re-interruption is a new result (new checkpoint), not a dup.
assert (
sched.record_resumed_stopped_result(
run.id,
status_value="interrupted",
checkpoint_id=f"{run.id}.000003",
now=now,
)
is True
)
assert len([r for r in _history(store, "a") if r["kind"] == "interrupted"]) == 2
# Unknown runs, manual runs, and bad statuses record nothing.
assert (
sched.record_resumed_stopped_result(
"run-999999",
status_value="completed",
checkpoint_id=None,
now=now,
)
is False
)
assert (
sched.record_resumed_stopped_result(
run.id, status_value="bogus", checkpoint_id=None, now=now
)
is False
)
manual = persist_admission(
store=runs,
run_id=runs.allocate_run_id(),
environment=fixture_environment(object()),
resolved_input={},
max_steps=None,
)
assert (
sched.record_resumed_stopped_result(
manual.id, status_value="completed", checkpoint_id=None, now=now
)
is False
)
sched.ownership.release()