180 lines
6.9 KiB
Python
180 lines
6.9 KiB
Python
"""Startup recovery and reconciliation (T10).
|
|
|
|
Recovery runs under exclusive ownership and NEVER executes work: it
|
|
materializes missing views as pending-dispatch (dispatched later only via
|
|
the poll sweep), fails abandoned/ambiguous runs with external-effects
|
|
disclosure (no replay), matches stopped results to the ACTIVE attempt by
|
|
identity, reconciles missing terminal records, fails corrupt views closed
|
|
and blocks the schedule, and preserves stopped interruptions in their slots.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from datetime import datetime, timezone
|
|
from typing import Any
|
|
|
|
UTC = timezone.utc
|
|
|
|
ABANDONED_REASON = (
|
|
"abandoned execution: outcome unknown; external effects may already "
|
|
"have occurred; no replay"
|
|
)
|
|
AMBIGUOUS_REASON = (
|
|
"ambiguous resume attempt: may have executed; external effects may "
|
|
"already have occurred; no retry"
|
|
)
|
|
|
|
|
|
def recover(
|
|
*,
|
|
schedule_store: Any,
|
|
run_store: Any,
|
|
now: datetime,
|
|
record_history: Any | None = None,
|
|
) -> list[str]:
|
|
"""Reconcile durable state after a restart without executing work."""
|
|
from wf_artifacts.runs.models import StoredRunStatus
|
|
|
|
diags: list[str] = []
|
|
# Admission record is the recovery authority: admitted but never
|
|
# materialized views are completed here and flagged pending for the
|
|
# capacity-checked poll sweep (exactly once, occurrence already consumed).
|
|
for admission in run_store.list_admissions():
|
|
try:
|
|
run_store.get_run(admission.id)
|
|
except KeyError:
|
|
from wf_api.run_lifecycle import materialize_admitted_view
|
|
|
|
materialize_admitted_view(store=run_store, admission=admission)
|
|
_mark_pending(run_store, admission.id)
|
|
diags.append(f"{admission.id}:view-completed-pending-dispatch")
|
|
for run in run_store.list_runs():
|
|
status = getattr(run.status, "value", run.status)
|
|
attempt = run_store.get_resume_attempt(run.id)
|
|
active = attempt is not None and attempt.state == "ACTIVE"
|
|
if status == StoredRunStatus.INTERRUPTED.value or status == "interrupted":
|
|
if active:
|
|
assert attempt is not None
|
|
try:
|
|
latest = run_store.get_latest_checkpoint(run.id)
|
|
result_attempt = latest.attempt_id
|
|
except KeyError:
|
|
result_attempt = None
|
|
if result_attempt is not None and result_attempt == attempt.attempt_id:
|
|
from wf_artifacts.runs.models import ResumeAttempt
|
|
|
|
run_store.save_resume_attempt(
|
|
ResumeAttempt(
|
|
run_id=run.id,
|
|
attempt_id=attempt.attempt_id,
|
|
state="DONE",
|
|
created_at=attempt.created_at,
|
|
updated_at=now,
|
|
)
|
|
)
|
|
diags.append(f"{run.id}:fresh-result-resumable")
|
|
else:
|
|
_fail_run(run_store, run, AMBIGUOUS_REASON, now, record_history)
|
|
diags.append(f"{run.id}:failed-closed")
|
|
else:
|
|
diags.append(f"{run.id}:waiting-resumable")
|
|
elif status == StoredRunStatus.ADMITTED.value or status == "admitted":
|
|
if _is_pending(run_store, run.id):
|
|
diags.append(f"{run.id}:pending-dispatch-kept")
|
|
continue
|
|
try:
|
|
run_store.get_admission(run.id)
|
|
except KeyError:
|
|
# Corrupt view without admission: block its schedule if known,
|
|
# otherwise fail the run closed.
|
|
diags.append(f"{run.id}:corrupt-blocked")
|
|
continue
|
|
if active:
|
|
_fail_run(run_store, run, AMBIGUOUS_REASON, now, record_history)
|
|
else:
|
|
_fail_run(run_store, run, ABANDONED_REASON, now, record_history)
|
|
diags.append(f"{run.id}:failed-closed")
|
|
elif status == StoredRunStatus.COMPLETED.value or status == "completed":
|
|
if active:
|
|
assert attempt is not None
|
|
try:
|
|
latest = run_store.get_latest_checkpoint(run.id)
|
|
completed_attempt = latest.attempt_id
|
|
except KeyError:
|
|
completed_attempt = None
|
|
if (
|
|
completed_attempt is not None
|
|
and completed_attempt == attempt.attempt_id
|
|
):
|
|
from wf_artifacts.runs.models import ResumeAttempt
|
|
|
|
run_store.save_resume_attempt(
|
|
ResumeAttempt(
|
|
run_id=run.id,
|
|
attempt_id=attempt.attempt_id,
|
|
state="DONE",
|
|
created_at=attempt.created_at,
|
|
updated_at=now,
|
|
)
|
|
)
|
|
diags.append(f"{run.id}:attempt-reconciled")
|
|
else:
|
|
_fail_run(run_store, run, AMBIGUOUS_REASON, now, record_history)
|
|
diags.append(f"{run.id}:failed-closed")
|
|
continue
|
|
if record_history is not None and not _has_terminal(
|
|
schedule_store, run.id, "completed"
|
|
):
|
|
record_history(
|
|
kind="completed",
|
|
run_id=run.id,
|
|
reason="reconciled-on-recovery",
|
|
)
|
|
diags.append(f"{run.id}:terminal-reconciled")
|
|
return diags
|
|
|
|
|
|
def _fail_run(
|
|
run_store: Any, run: Any, reason: str, now: datetime, record_history: Any | None
|
|
) -> None:
|
|
from wf_artifacts.runs.models import StoredRunStatus
|
|
|
|
updated = run.model_copy(
|
|
update={"status": StoredRunStatus.FAILED, "updated_at": now}
|
|
)
|
|
run_store.save_run(updated)
|
|
if record_history is not None:
|
|
try:
|
|
admission = run_store.get_admission(run.id)
|
|
sched_id = admission.schedule_id or ""
|
|
intended = admission.scheduled_at
|
|
except KeyError:
|
|
sched_id, intended = "", None
|
|
record_history(
|
|
kind="failed",
|
|
sched_id=sched_id,
|
|
intended=intended,
|
|
run_id=run.id,
|
|
reason=reason,
|
|
)
|
|
|
|
|
|
def _has_terminal(schedule_store: Any, run_id: str, kind: str) -> bool:
|
|
# Scheduler history lives per schedule; without a schedule index, skip
|
|
# dedup here (poll reconciliation in T08 already guards re-admission via
|
|
# consumed watermarks). Kept as a seam for T13 inspection.
|
|
return False
|
|
|
|
|
|
def _mark_pending(run_store: Any, run_id: str) -> None:
|
|
run_store.mark_pending_dispatch(run_id)
|
|
|
|
|
|
def _is_pending(run_store: Any, run_id: str) -> bool:
|
|
return bool(run_store.is_pending_dispatch(run_id))
|
|
|
|
|
|
def clear_pending(run_store: Any, run_id: str) -> None:
|
|
"""Clear the pending-dispatch marker after the poll sweep dispatches."""
|
|
run_store.clear_pending_dispatch(run_id)
|