sched: persist executing transition before executor; abandon on crash (F1)

This commit is contained in:
lda
2026-09-08 11:29:09 +07:00 Verified
parent 993ed07fd3
commit 1cf44e3a5c
4 changed files with 592 additions and 24 deletions
+98 -24
View File
@@ -300,38 +300,31 @@ class Scheduler:
self.schedule_store.save_consumed(sched.id, intended)
from wf_api.run_lifecycle import materialize_admitted_view
# Dispatch mark precedes the view so a crash after admission but
# before/during first dispatch stays pending (not abandoned). Cleared
# after dispatch returns regardless of outcome (hang still dispatched).
self.run_store.mark_pending_dispatch(run_id)
materialize_admitted_view(store=self.run_store, admission=admission)
try:
self._dispatch(run_id, now)
finally:
self.run_store.clear_pending_dispatch(run_id)
self._execute_guarded(run_id, now)
return run_id
def _dispatch(self, run_id: str, now: datetime) -> None:
"""Persist one stopped result for an admitted run via the dispatcher.
def _execute_guarded(self, run_id: str, now: datetime) -> None:
"""Dispatch one provably undispatched run behind a durable transition.
The dispatcher returns a genuine stopped :class:`wf_core.RunState`
(or :class:`StillRunning` when the outcome is unknown); persistence
goes through the shared ``wf_api.run_lifecycle`` boundary so
scheduler dispatches share the manual-run torn-write protocol.
The undispatched→executing mark is persisted BEFORE the executor is
invoked, and the pending marker is cleared with it. Clearing is
explicit after a stopped result is durably persisted — there is
deliberately no ``finally``: a terminated process must leave the
executing mark so recovery abandons the run instead of retrying it.
"""
from wf_api.run_lifecycle import persist_stopped_run
from wf_artifacts.runs.models import StoredRunStatus
try:
self.run_store.get_run(run_id)
except KeyError as exc:
raise BlockedSchedule(f"dispatch missing run view: {run_id!r}") from exc
try:
admission = self.run_store.get_admission(run_id)
except KeyError as exc:
raise BlockedSchedule(f"dispatch missing admission: {run_id!r}") from exc
if admission.schedule_id is None:
raise BlockedSchedule(f"dispatch missing schedule owner: {run_id!r}")
self.run_store.mark_executing(run_id)
self.run_store.clear_pending_dispatch(run_id)
result = self.dispatcher.dispatch(admission=admission, now=now)
if isinstance(result, StillRunning):
return
@@ -341,6 +334,52 @@ class Scheduler:
run=result.result,
run_id=run_id,
)
self.run_store.clear_executing(run_id)
kind = {
StoredRunStatus.COMPLETED: "completed",
StoredRunStatus.INTERRUPTED: "interrupted",
StoredRunStatus.FAILED: "failed",
}[stopped.status]
self._record(
kind=kind, # type: ignore[arg-type]
sched_id=admission.schedule_id,
intended=_admission_intended(self.run_store, run_id),
run_id=run_id,
now=now,
started_at=now,
)
def record_stopped_execution(self, run_id: str, result: Any, now: datetime) -> None:
"""Settle a hanging execution with its late stopped result.
Async-completion seam for dispatchers that returned
:class:`StillRunning`: the executor later produced a genuine stopped
:class:`wf_core.RunState`. Persists through the shared lifecycle
boundary, clears the executing mark, and records terminal history.
Never re-invokes the dispatcher.
"""
from wf_api.run_lifecycle import persist_stopped_run
from wf_artifacts.runs.models import StoredRunStatus
try:
admission = self.run_store.get_admission(run_id)
except KeyError as exc:
raise BlockedSchedule(f"settle missing admission: {run_id!r}") from exc
if admission.schedule_id is None:
raise BlockedSchedule(f"settle missing schedule owner: {run_id!r}")
try:
record = self.run_store.get_run(run_id)
except KeyError as exc:
raise BlockedSchedule(f"settle missing run view: {run_id!r}") from exc
if self._status_value(record) != "admitted":
raise BlockedSchedule(f"settle non-admitted run: {run_id!r}")
stopped = persist_stopped_run(
store=self.run_store,
environment=admission.environment,
run=result,
run_id=run_id,
)
self.run_store.clear_executing(run_id)
kind = {
StoredRunStatus.COMPLETED: "completed",
StoredRunStatus.INTERRUPTED: "interrupted",
@@ -380,20 +419,56 @@ class Scheduler:
"""Dispatch recovery-materialized runs through capacity checks.
Recovery NEVER executes: it only completes missing views flagged
pending for this sweep. Pending runs of blocked schedules stay
pending. Hanging admitted runs without a pending marker are never
re-executed here.
pending for this sweep. Each pending marker is validated before it
is trusted: markers without an admission (or without a schedule
owner) fail the run closed instead of dispatching; markers on
stopped runs are stale and cleared without redispatch; unknown
owners and blocked schedules are left untouched. Hanging admitted
runs without a pending marker are never re-executed here.
"""
from wf_scheduling.recovery import _is_pending, clear_pending
from wf_scheduling.recovery import (
CORRUPT_PENDING_REASON,
_fail_run,
clear_executing,
clear_pending,
is_executing,
)
from wf_scheduling.recovery import _is_pending as _pending
for run in sorted(self.run_store.list_runs(), key=lambda r: r.id):
if not _is_pending(self.run_store, run.id):
if not _pending(self.run_store, run.id):
continue
try:
admission = self.run_store.get_admission(run.id)
except KeyError:
_fail_run(
self.run_store,
run,
CORRUPT_PENDING_REASON,
now,
None,
)
clear_pending(self.run_store, run.id)
clear_executing(self.run_store, run.id)
continue
if admission.schedule_id is None:
_fail_run(
self.run_store,
run,
CORRUPT_PENDING_REASON,
now,
None,
)
clear_pending(self.run_store, run.id)
clear_executing(self.run_store, run.id)
continue
if self._status_value(run) != "admitted":
# Stale marker on a stopped run: clear it, never redispatch.
# Terminal-history reconciliation is owned by recovery (which
# dedups); the sweep only removes the untrustworthy marker.
clear_pending(self.run_store, run.id)
if is_executing(self.run_store, run.id):
clear_executing(self.run_store, run.id)
continue
try:
sched = self.schedule_store.get_schedule(admission.schedule_id)
@@ -403,8 +478,7 @@ class Scheduler:
continue
if self._task_load() >= self.capacity:
continue
self._dispatch(run.id, now)
clear_pending(self.run_store, run.id)
self._execute_guarded(run.id, now)
def _poll_one(self, sched: Any, now: datetime) -> str:
if sched.deleted: