Files
lda-wf/src/wf_scheduling/poll.py
T

1079 lines
45 KiB
Python

"""Poll loop: overlap, misfire, candidates, fairness, capacity (T08).
Implements the scheduling state rules (first probed as a reference model,
retired to docs/historical now that tests/scheduling/ pins them) against
real file stores. Calendar iteration uses the canonical
:class:`wf_scheduling.calendar.OccurrenceSource`
(``next_after``/``prev_before`` only, never enumeration); latest-missed
catch-up is one bounded ``prev_before`` query (F1). Overlap decisions
precede capacity checks; terminal skips never reappear; ``latest`` retains
at most one candidate.
Admission preparation and execution arrive as typed collaborators
(:mod:`wf_scheduling.prepare`, :mod:`wf_scheduling.dispatch`): this module
contains no fixture input, fixture environments, or canned results.
"""
from __future__ import annotations
from dataclasses import dataclass
from datetime import datetime, timezone
from typing import Any, cast
from wf_scheduling.calendar import (
CronSource,
InvalidScheduleDefinitionError,
OccurrenceSource,
OneShotSource,
)
from wf_scheduling.dispatch import RunDispatcher, StillRunning
from wf_scheduling.history import (
FileScheduleHistoryRecorder,
HistoryEntry,
HistoryRecorder,
)
from wf_scheduling.models import OccurrenceKind, PendingCandidate
from wf_scheduling.ownership import (
SchedulerOwnership,
SecondOwnerError,
describe_unsupported_layout,
)
from wf_scheduling.prepare import InvocationPreparer, PreparationRejected
from wf_scheduling.store import schedule_store_transaction
UTC = timezone.utc
SCAN_CAP = 100
EPOCH = datetime(2020, 1, 1, tzinfo=UTC)
class BlockedSchedule(ValueError):
"""A schedule is blocked by a corrupt record and fails closed."""
@dataclass(frozen=True, slots=True)
class _AdmissionOutcome:
"""Result of one atomic admission and any dispatch required afterward."""
result: str | None
dispatch_run_id: str | None = None
def source_for_trigger(trigger: Any) -> OccurrenceSource:
"""Build a calendar source from a schedule trigger model."""
kind = trigger.kind if hasattr(trigger, "kind") else trigger.get("kind")
if kind == "cron":
expression = (
trigger.expression
if hasattr(trigger, "expression")
else trigger["expression"]
)
zone = (
trigger.timezone
if hasattr(trigger, "timezone")
else trigger.get("timezone", "UTC")
)
return CronSource(expression, zone)
if kind == "oneshot":
at = trigger.at if hasattr(trigger, "at") else trigger["at"]
if isinstance(at, str):
at = datetime.fromisoformat(at)
return OneShotSource(at)
raise ValueError(f"unknown trigger kind {kind!r}")
def _is_oneshot(source: OccurrenceSource) -> bool:
return isinstance(source, OneShotSource)
def _save_consumed_max(store: Any, sched_id: str, instant: datetime) -> None:
"""Advance the consumed watermark without ever moving it backwards.
Admission decides instants out of order across retries (a held
candidate admitted after the watermark already advanced past it); a
backwards write would resurrect the intervening instants on restart.
"""
existing = store.get_consumed(sched_id)
if existing is None or instant > existing:
store.save_consumed(sched_id, instant)
def _latest_eligible(source: OccurrenceSource, now: datetime) -> datetime | None:
"""Select the latest eligible occurrence ``<= now`` with bounded queries.
``prev_before`` is exclusive, so a poll exactly at a due instant would
miss it. ``prev_before(now)`` is the greatest occurrence strictly before
``now``; at most one occurrence (``now`` itself) can lie in between, so
a single bounded ``next_after`` probe closes the gap without any custom
calendar math.
"""
latest = source.prev_before(now)
if latest is None:
return None
forward = source.next_after(latest)
if forward is not None and forward <= now:
latest = forward
return latest
class Scheduler:
"""File-store scheduler core with injected clock and collaborators.
Invocation preparation (:class:`wf_scheduling.prepare.InvocationPreparer`)
and execution (:class:`wf_scheduling.dispatch.RunDispatcher`) are typed
collaborators: production code here never fabricates input, environments,
or outcomes. Stopped results are persisted through the shared
``wf_api.run_lifecycle`` boundary.
"""
def __init__(
self,
*,
schedule_store: Any,
run_store: Any,
sources: dict[str, OccurrenceSource],
capacity: int,
preparer: InvocationPreparer,
dispatcher: RunDispatcher,
ownership: SchedulerOwnership,
history: HistoryRecorder | None = None,
) -> None:
self.schedule_store = schedule_store
self.run_store = run_store
self.sources = sources
self.capacity = capacity
self.preparer = preparer
self.dispatcher = dispatcher
self.ownership = ownership
self.history: HistoryRecorder = (
history
if history is not None
else FileScheduleHistoryRecorder(schedule_store)
)
self._poll_cursor = 0
self._source_definitions: dict[str, Any] = {}
def set_source_definitions(self, definitions: dict[str, Any]) -> None:
"""Record the trigger definitions used to build managed sources.
The service refreshes this map at tick start. A schedule edit can
commit between that refresh and a schedule's fresh read; the poller
then rebuilds only that schedule's source before resolving an instant.
Direct Scheduler tests without this service-owned map keep their
injected source collaborators unchanged.
"""
self._source_definitions = dict(definitions)
def _require_ownership(self) -> None:
"""Reject schedule mutation/dispatch without proven live ownership.
Runs before any store write or dispatcher side effect: the held
lock must cover the actual schedule and run store composition, not
merely be held on some unrelated directory. Without covering
ownership this process cannot prove exclusive ownership, so polling
or administering schedules would risk double admission.
"""
ownership = self.ownership
sched_root = getattr(self.schedule_store, "root", None)
runs_root = getattr(self.run_store, "root", None)
unsupported = describe_unsupported_layout(sched_root, runs_root)
if unsupported is not None:
raise SecondOwnerError(unsupported)
if ownership is None or not ownership.covers(sched_root, runs_root):
raise SecondOwnerError(
"scheduler ownership must cover the schedule and run stores "
"before polling or mutating schedules"
)
# -- helpers ------------------------------------------------------
@staticmethod
def _status_value(run: Any) -> Any:
status = getattr(run, "status", None)
return getattr(status, "value", status)
def _active(self, sched_id: str) -> list[Any]:
active: list[Any] = []
for run in self.run_store.list_runs():
if self._status_value(run) not in ("admitted", "interrupted"):
continue
direct = getattr(run, "sched_id", None)
if direct is not None:
if direct == sched_id:
active.append(run)
continue
if self._run_sched(run) == sched_id:
active.append(run)
return active
def _run_sched(self, run: Any) -> str | None:
try:
admission = self.run_store.get_admission(run.id)
except KeyError:
return None
return admission.schedule_id
def _fail_unattributed_active_runs(self, now: datetime) -> None:
"""Fail corrupt active views without assigning blame to a sibling.
A run view without an admission has no schedule identity and cannot
safely participate in overlap or capacity accounting. It is failed
closed in place, while healthy schedules continue their own poll.
"""
from wf_scheduling.recovery import (
CORRUPT_VIEW_REASON,
_fail_run,
clear_executing,
clear_pending,
)
for run in self.run_store.list_runs():
if self._status_value(run) not in ("admitted", "interrupted"):
continue
try:
self.run_store.get_admission(run.id)
except KeyError:
_fail_run(self.run_store, run, CORRUPT_VIEW_REASON, now, self.history)
clear_pending(self.run_store, run.id)
clear_executing(self.run_store, run.id)
def _task_load(self) -> int:
from wf_scheduling.recovery import _is_pending, is_executing
count = 0
for run in self.run_store.list_runs():
# Manual API runs share the run store but do not consume the
# scheduler's bounded dispatch capacity. An admission with no
# schedule owner is the durable discriminator for that path.
try:
admission = self.run_store.get_admission(run.id)
except KeyError:
continue
if admission.schedule_id is None:
continue
if self._status_value(run) != "admitted":
# A mid-resume scheduled run is interrupted but holds the
# durable executing mark: it occupies a live execution
# slot exactly like a dispatched run (B2), so the shared
# capacity gate must count it.
if is_executing(self.run_store, run.id) and not _is_pending(
self.run_store, run.id
):
count += 1
continue
if _is_pending(self.run_store, run.id):
continue
count += 1
return count
def _record(
self,
*,
kind: OccurrenceKind,
sched_id: str,
intended: datetime | None = None,
run_id: str | None = None,
reason: str = "",
interval: tuple[datetime, datetime] | None = None,
count: int = 0,
revision: int | None = None,
now: datetime | None = None,
admitted_at: datetime | None = None,
started_at: datetime | None = None,
checkpoint_id: str | None = None,
) -> None:
created = now if now is not None else datetime.now(UTC)
self.history.record(
HistoryEntry(
schedule_id=sched_id,
kind=kind,
resolved_at=intended,
run_id=run_id,
revision=revision,
reason=reason,
admitted_at=admitted_at,
started_at=started_at,
checkpoint_id=checkpoint_id,
interval_start=interval[0] if interval else None,
interval_end=interval[1] if interval else None,
interval_count=count,
created_at=created,
)
)
def _existing_occurrence_run(self, sched_id: str, intended: datetime) -> str | None:
"""Return the run already owning this occurrence, if any.
Identity is ``(schedule_id, resolved UTC instant)`` from the durable
admission record — the admission persist is the decision point, so
admissions are scanned rather than views (a crashed admission may
not have a view yet). Manual runs carry no scheduled instant and
never match.
"""
for admission in self.run_store.list_admissions():
if admission.schedule_id == sched_id and admission.scheduled_at == intended:
# Admission is the occurrence authority. If a later write
# failed before the view/pending marker, rebuild only this
# run so the next pending sweep can dispatch it; never admit
# the same occurrence again.
try:
self.run_store.get_run(admission.id)
except KeyError:
from wf_api.run_lifecycle import materialize_admitted_view
materialize_admitted_view(store=self.run_store, admission=admission)
self.run_store.mark_pending_dispatch(admission.id)
return admission.id
return None
def _admit(self, sched: Any, intended: datetime, now: datetime) -> str | None:
"""Admit atomically, then dispatch a new run after the transaction.
The durable admission and all schedule-side markers must commit as one
compound transition. Registration can wait on the event loop, though,
so it deliberately happens after that transaction releases; otherwise
an event-loop admin mutation can wait on a worker that is waiting for
the event loop to register its task.
"""
with schedule_store_transaction(self.schedule_store):
outcome = self._admit_locked(sched, intended, now)
if outcome.dispatch_run_id is not None:
self._execute_guarded(outcome.dispatch_run_id, now)
return outcome.result
def _admit_locked(
self, sched: Any, intended: datetime, now: datetime
) -> _AdmissionOutcome:
# The poller's listing is only a fairness snapshot. Re-read under the
# admission transaction so an edit committed before this point wins;
# an edit after this point waits and affects a later occurrence.
try:
current = self.schedule_store.get_schedule(sched.id)
except KeyError:
return _AdmissionOutcome(None)
if current.revision != sched.revision:
return _AdmissionOutcome("schedule-changed")
sched = current
if getattr(sched, "blocked_reason", None):
raise BlockedSchedule(getattr(sched, "blocked_reason"))
if not sched.enabled or sched.deleted or sched.paused:
return _AdmissionOutcome(None)
# Fail closed before any new admission when a corrupt active view
# exists for this schedule.
for run in self._active(sched.id):
try:
self.run_store.get_admission(run.id)
except KeyError:
reason = f"corrupt run view without admission: {run.id}"
sched.blocked_reason = reason
self.schedule_store.save_schedule(sched)
raise BlockedSchedule(reason) from None
existing = self._existing_occurrence_run(sched.id, intended)
if existing is not None:
# The occurrence already owns a run (crash between the admission
# persist and the watermark/history writes, or a lost watermark
# write): an occurrence is immutable and never replayed. Advance
# the watermark, reconcile a missing admitted entry exactly once,
# and return the owner without dispatching (the pending sweep
# owns dispatch).
cand = self.schedule_store.get_candidate(sched.id)
if cand is not None and cand.intended_at == intended:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
_save_consumed_max(self.schedule_store, sched.id, intended)
if not self.history.has_terminal(sched.id, existing, "admitted", None):
self._record(
kind="admitted",
sched_id=sched.id,
intended=intended,
run_id=existing,
reason=f"rev={sched.revision}",
revision=sched.revision,
now=now,
admitted_at=now,
)
return _AdmissionOutcome(existing)
prepared = self.preparer.prepare(sched=sched, intended=intended, now=now)
if isinstance(prepared, PreparationRejected):
self._record(
kind="preflight-rejected",
sched_id=sched.id,
intended=intended,
reason=prepared.reason,
revision=sched.revision,
now=now,
)
cand = self.schedule_store.get_candidate(sched.id)
if cand is not None and cand.intended_at == intended:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
_save_consumed_max(self.schedule_store, sched.id, intended)
return _AdmissionOutcome(None)
active = self._active(sched.id)
if sched.overlap == "skip" and active:
self._record(
kind="skipped-overlap",
sched_id=sched.id,
intended=intended,
reason=f"active={[getattr(r, 'id', None) for r in active]}",
revision=sched.revision,
)
cand = self.schedule_store.get_candidate(sched.id)
if cand is not None and cand.intended_at == intended:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
_save_consumed_max(self.schedule_store, sched.id, intended)
return _AdmissionOutcome(None)
if sched.overlap == "parallel" and len(active) >= sched.max_active_runs:
self._record(
kind="skipped-overlap",
sched_id=sched.id,
intended=intended,
reason=f"max_active={sched.max_active_runs}",
revision=sched.revision,
)
cand = self.schedule_store.get_candidate(sched.id)
if cand is not None and cand.intended_at == intended:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
_save_consumed_max(self.schedule_store, sched.id, intended)
return _AdmissionOutcome(None)
if self._task_load() >= self.capacity:
if sched.misfire == "latest":
self.schedule_store.save_candidate(
PendingCandidate(
schedule_id=sched.id,
intended_at=intended,
revision=sched.revision,
),
schedule_id=sched.id,
)
_save_consumed_max(self.schedule_store, sched.id, intended)
return _AdmissionOutcome("held")
if (now - intended).total_seconds() > sched.lateness_allowance_s:
self._record(
kind="skipped-misfire",
sched_id=sched.id,
intended=intended,
reason="capacity-deadline",
revision=sched.revision,
)
_save_consumed_max(self.schedule_store, sched.id, intended)
return _AdmissionOutcome(None)
return _AdmissionOutcome("held-undecided")
run_id = self.run_store.allocate_run_id()
from wf_api.run_lifecycle import persist_admission
admission = persist_admission(
store=self.run_store,
run_id=run_id,
environment=prepared.environment,
resolved_input=prepared.resolved_input,
max_steps=prepared.max_steps,
scheduled_at=intended,
schedule_id=sched.id,
schedule_revision=sched.revision,
deployment_revision=prepared.deployment_revision,
)
self._record(
kind="admitted",
sched_id=sched.id,
intended=intended,
run_id=run_id,
reason=f"rev={sched.revision}",
revision=sched.revision,
now=now,
admitted_at=now,
)
if _is_oneshot(self.sources.get(sched.id)): # type: ignore[arg-type]
sched.exhausted = True
self.schedule_store.save_schedule(sched)
cand = self.schedule_store.get_candidate(sched.id)
if cand is not None and cand.intended_at == intended:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
_save_consumed_max(self.schedule_store, sched.id, intended)
from wf_api.run_lifecycle import materialize_admitted_view
self.run_store.mark_pending_dispatch(run_id)
materialize_admitted_view(store=self.run_store, admission=admission)
return _AdmissionOutcome(run_id, dispatch_run_id=run_id)
def _execute_guarded(self, run_id: str, now: datetime) -> None:
"""Dispatch one provably undispatched run behind a durable transition.
The undispatched→executing mark is persisted BEFORE the executor is
invoked, and the pending marker is cleared with it. Clearing is
explicit after a stopped result is durably persisted — there is
deliberately no ``finally``: a terminated process must leave the
executing mark so recovery abandons the run instead of retrying it.
"""
from wf_api.run_lifecycle import persist_stopped_run
from wf_artifacts.runs.models import StoredRunStatus
try:
admission = self.run_store.get_admission(run_id)
except KeyError as exc:
raise BlockedSchedule(f"dispatch missing admission: {run_id!r}") from exc
if admission.schedule_id is None:
raise BlockedSchedule(f"dispatch missing schedule owner: {run_id!r}")
self.run_store.mark_executing(run_id)
self.run_store.clear_pending_dispatch(run_id)
result = self.dispatcher.dispatch(admission=admission, now=now)
if isinstance(result, StillRunning):
return
stopped = persist_stopped_run(
store=self.run_store,
environment=admission.environment,
run=result.result,
run_id=run_id,
)
self.run_store.clear_executing(run_id)
kind = cast(
OccurrenceKind,
{
StoredRunStatus.COMPLETED: "completed",
StoredRunStatus.INTERRUPTED: "interrupted",
StoredRunStatus.FAILED: "failed",
}[stopped.status],
)
self._record(
kind=kind,
sched_id=admission.schedule_id,
intended=_admission_intended(self.run_store, run_id),
run_id=run_id,
now=now,
started_at=now,
checkpoint_id=stopped.latest_checkpoint_id,
)
def record_stopped_execution(self, run_id: str, result: Any, now: datetime) -> None:
"""Settle a hanging execution with its late stopped result.
Async-completion seam for dispatchers that returned
:class:`StillRunning`: the executor later produced a genuine stopped
:class:`wf_core.RunState`. Persists through the shared lifecycle
boundary, clears the executing mark, and records terminal history.
Never re-invokes the dispatcher.
An admitted status alone is not enough: pending, never-dispatched
work is also admitted. Settlement requires the durable executing
transition and refuses contradictory pending/executing state (a
crash between the transition writes owns that run now, not this
caller); it also refuses runs that already stopped.
"""
self._require_ownership()
from wf_api.run_lifecycle import persist_stopped_run
from wf_artifacts.runs.models import StoredRunStatus
try:
admission = self.run_store.get_admission(run_id)
except KeyError as exc:
raise BlockedSchedule(f"settle missing admission: {run_id!r}") from exc
if admission.schedule_id is None:
raise BlockedSchedule(f"settle missing schedule owner: {run_id!r}")
try:
record = self.run_store.get_run(run_id)
except KeyError as exc:
raise BlockedSchedule(f"settle missing run view: {run_id!r}") from exc
if self._status_value(record) != "admitted":
raise BlockedSchedule(f"settle non-admitted run: {run_id!r}")
if self.run_store.is_pending_dispatch(run_id):
raise BlockedSchedule(
f"settle contradictory pending/executing run: {run_id!r}"
)
if not self.run_store.is_executing(run_id):
raise BlockedSchedule(f"settle without executing transition: {run_id!r}")
stopped = persist_stopped_run(
store=self.run_store,
environment=admission.environment,
run=result,
run_id=run_id,
)
self.run_store.clear_executing(run_id)
kind = cast(
OccurrenceKind,
{
StoredRunStatus.COMPLETED: "completed",
StoredRunStatus.INTERRUPTED: "interrupted",
StoredRunStatus.FAILED: "failed",
}[stopped.status],
)
self._record(
kind=kind,
sched_id=admission.schedule_id,
intended=_admission_intended(self.run_store, run_id),
run_id=run_id,
now=now,
started_at=now,
checkpoint_id=stopped.latest_checkpoint_id,
)
def record_resumed_stopped_result(
self,
run_id: str,
*,
status_value: str,
checkpoint_id: str | None,
now: datetime,
) -> bool:
"""Record one live resumed stopped result, idempotently.
The run API persists the resumed stopped checkpoint itself; this
only appends the matching occurrence-history entry through the
same ``(run_id, kind, checkpoint_id)`` idempotency as dispatch
and recovery, so repeated polls and restart recovery never
duplicate it. Schedule flags are deliberately not consulted:
pausing or deleting a schedule never suppresses retained run
history. Returns whether an entry was appended.
"""
self._require_ownership()
kind = cast(
OccurrenceKind | None,
{
"completed": "completed",
"interrupted": "interrupted",
"failed": "failed",
}.get(status_value),
)
if kind is None:
return False
try:
admission = self.run_store.get_admission(run_id)
except KeyError:
return False
sched_id = admission.schedule_id
if sched_id is None:
return False
if self.history.has_terminal(sched_id, run_id, kind, checkpoint_id):
return False
self._record(
kind=kind,
sched_id=sched_id,
intended=admission.scheduled_at,
run_id=run_id,
revision=admission.schedule_revision,
reason="resumed",
now=now,
started_at=now,
checkpoint_id=checkpoint_id,
)
return True
# -- polling --------------------------------------------------------
def poll(self, now: datetime) -> dict[str, str]:
self._require_ownership()
self._fail_unattributed_active_runs(now)
self._dispatch_pending(now)
schedules = self.schedule_store.list_schedules(include_deleted=True)
ids = sorted(item.id for item in schedules)
if not ids:
return {}
cursor = self.schedule_store.get_poll_cursor()
start = cursor % len(ids)
order = ids[start:] + ids[:start]
self.schedule_store.save_poll_cursor(cursor + 1)
results: dict[str, str] = {}
by_id = {item.id: item for item in schedules}
for sid in order:
sched = by_id[sid]
if getattr(sched, "blocked_reason", None):
raise BlockedSchedule(getattr(sched, "blocked_reason"))
results[sid] = self._poll_one(sched, now)
return results
def _dispatch_pending(self, now: datetime) -> None:
"""Dispatch recovery-materialized runs through capacity checks.
Recovery NEVER executes: it only completes missing views flagged
pending for this sweep. Each pending marker is validated before it
is trusted: markers without an admission (or without a schedule
owner) fail the run closed instead of dispatching; markers on
stopped runs are stale and cleared without redispatch; unknown
owners and blocked schedules are left untouched. Hanging admitted
runs without a pending marker are never re-executed here.
"""
from wf_scheduling.recovery import (
CORRUPT_PENDING_REASON,
_fail_run,
clear_executing,
clear_pending,
is_executing,
)
from wf_scheduling.recovery import _is_pending as _pending
for run in sorted(self.run_store.list_runs(), key=lambda r: r.id):
if not _pending(self.run_store, run.id):
continue
try:
admission = self.run_store.get_admission(run.id)
except KeyError:
_fail_run(
self.run_store,
run,
CORRUPT_PENDING_REASON,
now,
self.history,
)
clear_pending(self.run_store, run.id)
clear_executing(self.run_store, run.id)
continue
if admission.schedule_id is None:
_fail_run(
self.run_store,
run,
CORRUPT_PENDING_REASON,
now,
self.history,
)
clear_pending(self.run_store, run.id)
clear_executing(self.run_store, run.id)
continue
if self._status_value(run) != "admitted":
# Stale marker on a stopped run: clear it, never redispatch.
# Terminal-history reconciliation is owned by recovery (which
# dedups); the sweep only removes the untrustworthy marker.
clear_pending(self.run_store, run.id)
if is_executing(self.run_store, run.id):
clear_executing(self.run_store, run.id)
continue
try:
sched = self.schedule_store.get_schedule(admission.schedule_id)
except KeyError:
continue
if getattr(sched, "blocked_reason", None):
continue
if self._task_load() >= self.capacity:
continue
self._execute_guarded(run.id, now)
def _poll_one(self, sched: Any, now: datetime) -> str:
# Fresh read per schedule: the tick lists schedules up front while
# same-process administration may commit an edit, pause, resume, or
# delete mid-tick. All admission decisions below (policies,
# revision, flags, and managed trigger source) use this fresh copy,
# never the listing snapshot. A definition edit that races the
# final admission recheck is rejected as schedule-changed, so stale
# terms cannot create a duplicate or replay.
try:
sched = self.schedule_store.get_schedule(sched.id)
except KeyError:
# Soft deletes never remove the file; a vanishing schedule is
# unexpectedly gone — treat it as deleted work, never admit.
return "deleted"
if sched.deleted:
if self.schedule_store.get_candidate(sched.id) is not None:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
return "deleted"
if not sched.enabled:
if self.schedule_store.get_candidate(sched.id) is not None:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
consumed = self.schedule_store.get_consumed(sched.id) or EPOCH
self.schedule_store.save_consumed(sched.id, max(consumed, now))
return "disabled"
if sched.exhausted:
# Exhaustion is durable terminal state. A torn one-shot
# transition may have written the history/flag before its stale
# candidate was cleared; retrying must finish that cleanup.
if self.schedule_store.get_candidate(sched.id) is not None:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
return "exhausted"
if sched.paused:
if self.schedule_store.get_candidate(sched.id) is not None:
self.schedule_store.save_candidate(None, schedule_id=sched.id)
consumed = self.schedule_store.get_consumed(sched.id) or EPOCH
self.schedule_store.save_consumed(sched.id, max(consumed, now))
return "paused"
source_definition = self._source_definitions.get(sched.id)
if source_definition is not None and source_definition != sched.trigger:
try:
src = source_for_trigger(sched.trigger)
except Exception as exc:
raise InvalidScheduleDefinitionError(
f"invalid trigger for schedule {sched.id!r}: {exc}"
) from exc
self.sources[sched.id] = src
self._source_definitions[sched.id] = sched.trigger
else:
try:
src = self.sources[sched.id]
except KeyError as exc:
raise InvalidScheduleDefinitionError(
f"no occurrence source for schedule {sched.id!r}"
) from exc
consumed = self.schedule_store.get_consumed(sched.id) or EPOCH
if consumed > now:
return "clock-rollback-held"
due: list[datetime] = []
cursor = consumed
jumped = False
while True:
nxt = src.next_after(cursor)
if nxt is None or nxt > now:
if (
not due
and _is_oneshot(src)
and nxt is None
and not sched.exhausted
and sched.misfire == "skip"
and getattr(src, "at", None) is not None
and getattr(src, "at") <= now
):
at = getattr(src, "at")
self._record(
kind="exhausted",
sched_id=sched.id,
intended=at,
reason="oneshot-expired-skip",
revision=sched.revision,
)
sched.exhausted = True
self.schedule_store.save_schedule(sched)
self.schedule_store.save_consumed(sched.id, now)
self.schedule_store.save_candidate(None, schedule_id=sched.id)
return "exhausted"
break
due.append(nxt)
cursor = nxt
if len(due) >= SCAN_CAP:
jumped = True
break
if jumped:
if sched.misfire == "latest":
from wf_scheduling.calendar import ScheduleExhaustedError
latest = _latest_eligible(src, now)
if latest is None:
raise ScheduleExhaustedError(
f"latest-missed lookup exhausted for {sched.id!r}"
)
old = self.schedule_store.get_candidate(sched.id)
if old is not None and old.intended_at != latest:
self._record(
kind="superseded",
sched_id=sched.id,
intended=old.intended_at,
reason=f"coalesced-into:{latest.isoformat()}",
revision=sched.revision,
now=now,
)
self.schedule_store.save_candidate(
PendingCandidate(
schedule_id=sched.id,
intended_at=latest,
revision=sched.revision,
),
schedule_id=sched.id,
)
self._record(
kind="interval-summary",
sched_id=sched.id,
reason="coalesced-missed-span",
revision=sched.revision,
interval=(consumed, now),
count=-1,
now=now,
)
self.schedule_store.save_consumed(sched.id, now)
held = self._admit_held_candidate(sched, now)
if held is None:
# Terminal skip inside admission (e.g. overlap) clears the
# candidate; do not misreport it as held.
return "skipped-overlap"
return held
self._record(
kind="interval-summary",
sched_id=sched.id,
reason="skipped-missed-span",
revision=sched.revision,
interval=(consumed, now),
count=-1,
now=now,
)
self.schedule_store.save_consumed(sched.id, now)
return "span-skipped"
if (
sched.misfire == "latest"
and due
and any(
(now - instant).total_seconds() > sched.lateness_allowance_s
for instant in due
)
):
# A bounded catch-up can still contain several missed instants.
# Coalesce them before the per-instant loop so ``latest`` never
# turns a short downtime into a replay burst.
latest = due[-1]
old = self.schedule_store.get_candidate(sched.id)
if old is not None and old.intended_at != latest:
self._record(
kind="superseded",
sched_id=sched.id,
intended=old.intended_at,
reason=f"coalesced-into:{latest.isoformat()}",
revision=sched.revision,
now=now,
)
self.schedule_store.save_candidate(
PendingCandidate(
schedule_id=sched.id,
intended_at=latest,
revision=sched.revision,
),
schedule_id=sched.id,
)
self._record(
kind="interval-summary",
sched_id=sched.id,
reason="coalesced-missed-span",
revision=sched.revision,
interval=(consumed, now),
count=-1,
now=now,
)
self.schedule_store.save_consumed(sched.id, now)
held = self._admit_held_candidate(sched, now)
return "skipped-overlap" if held is None else held
last_result = "idle"
for instant in due:
stored_consumed = self.schedule_store.get_consumed(sched.id) or EPOCH
if instant <= stored_consumed:
continue
age = (now - instant).total_seconds()
if age <= sched.lateness_allowance_s:
old = self.schedule_store.get_candidate(sched.id)
if old is not None and old.intended_at != instant:
self._record(
kind="superseded",
sched_id=sched.id,
intended=old.intended_at,
reason=f"admitted-newer:{instant.isoformat()}",
revision=sched.revision,
)
self.schedule_store.save_candidate(None, schedule_id=sched.id)
result = self._admit(sched, instant, now)
last_result = f"admit:{result}"
elif sched.misfire == "latest":
old = self.schedule_store.get_candidate(sched.id)
if old is not None and old.intended_at != instant:
self._record(
kind="superseded",
sched_id=sched.id,
intended=old.intended_at,
reason=f"coalesced-into:{instant.isoformat()}",
revision=sched.revision,
)
self.schedule_store.save_candidate(
PendingCandidate(
schedule_id=sched.id,
intended_at=instant,
revision=sched.revision,
),
schedule_id=sched.id,
)
self.schedule_store.save_consumed(sched.id, instant)
held = self._admit_held_candidate(sched, now)
if held is None:
last_result = "skipped-overlap"
elif held == "held-undecided":
last_result = "candidate-held"
else:
last_result = held
else:
if _is_oneshot(src) and not sched.exhausted:
self._record(
kind="exhausted",
sched_id=sched.id,
intended=instant,
reason="oneshot-expired-skip",
revision=sched.revision,
)
sched.exhausted = True
self.schedule_store.save_schedule(sched)
self.schedule_store.save_consumed(sched.id, instant)
self.schedule_store.save_candidate(None, schedule_id=sched.id)
last_result = "exhausted"
continue
self._record(
kind="skipped-misfire",
sched_id=sched.id,
intended=instant,
reason=f"age={age:.0f}s",
revision=sched.revision,
)
self.schedule_store.save_consumed(sched.id, instant)
last_result = "skipped-misfire"
if last_result in ("idle",):
cand = self.schedule_store.get_candidate(sched.id)
if cand is not None:
held = self._admit_held_candidate(sched, now)
if held is None:
last_result = "skipped-overlap"
elif held == "held-undecided":
last_result = "candidate-held"
else:
last_result = held
return last_result
def _admit_held_candidate(self, sched: Any, now: datetime) -> str | None:
cand = self.schedule_store.get_candidate(sched.id)
if cand is None or cand.revision != sched.revision:
if cand is not None and cand.revision != sched.revision:
self._record(
kind="superseded",
sched_id=sched.id,
intended=cand.intended_at,
reason="schedule-edit",
revision=sched.revision,
now=now,
)
self.schedule_store.save_candidate(None, schedule_id=sched.id)
return None
return self._admit(sched, cand.intended_at, now)
# -- administration ---------------------------------------------------
def resume_schedule(self, sid: str, now: datetime) -> None:
"""Unpause: resume selects the next future occurrence.
Crash-safe ordering shared with the API surface: the candidate
is cleared and the watermark advances BEFORE the flag flip is
persisted, so a crash can only leave the schedule paused
(retryable) and never an unpaused flag whose span backfills.
"""
self._require_ownership()
self.schedule_store.get_schedule(sid)
self.schedule_store.save_candidate(None, schedule_id=sid)
consumed = self.schedule_store.get_consumed(sid) or EPOCH
self.schedule_store.save_consumed(sid, max(consumed, now))
sched = self.schedule_store.get_schedule(sid)
sched.paused = False
self.schedule_store.save_schedule(sched)
def edit_schedule(self, sid: str, now: datetime) -> None:
"""Definition edit: new revision, discard old candidates, no backfill.
Crash-safe ordering shared with the API surface: old candidates
are discarded (with a superseded row) and the watermark advances
BEFORE the revision bump is persisted, so a crash can only leave
the edit unapplied under the old revision (retryable) and never
a bumped revision that backfills pre-edit instants on restart.
"""
self._require_ownership()
sched = self.schedule_store.get_schedule(sid)
old = self.schedule_store.get_candidate(sid)
if old is not None:
self._record(
kind="superseded",
sched_id=sid,
intended=old.intended_at,
reason="schedule-edit",
revision=sched.revision + 1,
)
self.schedule_store.save_candidate(None, schedule_id=sid)
consumed = self.schedule_store.get_consumed(sid) or EPOCH
self.schedule_store.save_consumed(sid, max(consumed, now))
sched = self.schedule_store.get_schedule(sid)
sched.revision += 1
sched.updated_at = now
self.schedule_store.save_schedule(sched)
def _admission_intended(run_store: Any, run_id: str) -> datetime | None:
try:
return run_store.get_admission(run_id).scheduled_at
except KeyError:
return None