fix: harden workflow demo and evaluation tooling

This commit is contained in:
lda
2026-07-01 08:20:38 +07:00 Verified
parent 7bdfa3132d
commit 7e5f1c88da
29 changed files with 365 additions and 71 deletions
+18 -3
View File
@@ -17,6 +17,8 @@ FIGURE_STEMS = (
"agent-challenge-token-volume",
)
_MANUAL_OUTCOMES = frozenset({"pass", "invalid", "fail"})
_TASK_OUTCOMES = frozenset({"success", "failed", "timeout", "runner_error"})
_EVALUATION_PROFILES = frozenset({"none", "skills", "all"})
_CHALLENGE_ORDER = {"browser": 0, "report": 1}
_MODEL_ORDER = {"deepseek": 0, "mimo": 1}
_PROFILE_ORDER = {"none": 0, "skills": 1, "all": 2}
@@ -105,6 +107,19 @@ def _load_trial(
raise ValueError(
f"unsupported manual outcome {manual_outcome!r}: {report_path}"
)
profile = _string(run.get("profile"), field="runs[].profile")
if profile not in _EVALUATION_PROFILES:
raise ValueError(f"unsupported evaluation profile {profile!r}: {report_path}")
task_outcome = _string(run.get("task_outcome"), field="runs[].task_outcome")
if task_outcome not in _TASK_OUTCOMES:
raise ValueError(f"unsupported task outcome {task_outcome!r}: {report_path}")
raw_audit_notes = run.get("audit_notes")
if raw_audit_notes is None:
audit_notes = ""
elif isinstance(raw_audit_notes, str):
audit_notes = raw_audit_notes
else:
raise ValueError(f"runs[].audit_notes must be a string: {report_path}")
duration = run.get("duration_seconds")
if not isinstance(duration, int | float):
@@ -117,7 +132,7 @@ def _load_trial(
_string(run.get("challenge"), field="runs[].challenge")
),
model=short_model_name(_string(run.get("model"), field="runs[].model")),
profile=_string(run.get("profile"), field="runs[].profile"),
profile=profile,
trial_index=_integer(run.get("trial_index"), field="runs[].trial_index"),
repository_commit=_string(
run.get("repository_commit"), field="runs[].repository_commit"
@@ -126,10 +141,10 @@ def _load_trial(
run.get("base_prompt_hash"), field="runs[].base_prompt_hash"
),
manual_outcome=manual_outcome,
task_outcome=_string(run.get("task_outcome"), field="runs[].task_outcome"),
task_outcome=task_outcome,
duration_seconds=float(duration),
tokens_total=_integer(run.get("tokens_total"), field="runs[].tokens_total"),
audit_notes=str(run.get("audit_notes") or ""),
audit_notes=audit_notes,
)
@@ -24,6 +24,7 @@ _MODEL_STYLES = {
_PROFILE_ORDER = {"none": 0, "skills": 1, "all": 2}
_CHALLENGE_ORDER = {"browser": 0, "report": 1}
_MODEL_ORDER = {"deepseek": 0, "mimo": 1}
_TASK_OUTCOME_ORDER = ("success", "failed", "timeout", "runner_error")
def _configure_matplotlib() -> Any:
@@ -170,8 +171,17 @@ def _outcomes_by_cell(cohort: EvaluationCohort, plt: Any) -> Figure:
def _automatic_vs_manual(cohort: EvaluationCohort, plt: Any) -> Figure:
from matplotlib.colors import LinearSegmentedColormap
task_labels = ("success", "failed")
task_labels = tuple(
outcome
for outcome in _TASK_OUTCOME_ORDER
if any(trial.task_outcome == outcome for trial in cohort.trials)
)
manual_labels = ("pass", "invalid", "fail")
unknown_task_outcomes = sorted(
{trial.task_outcome for trial in cohort.trials} - set(_TASK_OUTCOME_ORDER)
)
if unknown_task_outcomes:
raise ValueError(f"unsupported task outcomes: {unknown_task_outcomes}")
matrix = [
[
sum(
@@ -211,7 +221,7 @@ def _automatic_vs_manual(cohort: EvaluationCohort, plt: Any) -> Figure:
axis.set_ylabel("Automatic task outcome")
axis.set_title("Automatic completion does not imply clean evaluation evidence")
axis.set_xticks([value - 0.5 for value in range(1, len(manual_labels))], minor=True)
axis.set_yticks([0.5], minor=True)
axis.set_yticks([value - 0.5 for value in range(1, len(task_labels))], minor=True)
axis.grid(which="minor", color="white", linewidth=2)
axis.tick_params(which="minor", bottom=False, left=False)
figure.colorbar(image, ax=axis, label="Trial count", shrink=0.82)
@@ -275,7 +285,11 @@ def _scatter_metric(
profiles = ("none", "skills", "all")
wave_offsets = {1: -0.055, 2: 0.0, 3: 0.055}
for trial in sorted(trials, key=_trial_sort_key):
style = _MODEL_STYLES[trial.model]
style = _MODEL_STYLES.get(trial.model)
if style is None:
raise ValueError(f"unsupported evaluation model {trial.model!r}")
if trial.profile not in profiles:
raise ValueError(f"unsupported evaluation profile {trial.profile!r}")
x = (
profiles.index(trial.profile)
+ float(style["offset"])
+1 -13
View File
@@ -97,20 +97,8 @@ def resume_command_from_result(
variant: str | None = None,
prompt_mode: PromptMode = "auto",
) -> list[str]:
"""Build a resume command from new metadata or recover it from old raw results."""
"""Rebuild a resume command from validated metadata or old raw results."""
opencode = result.get("opencode")
if isinstance(opencode, dict):
command = opencode.get("resume_command")
has_override = (
any(value is not None for value in (session_id, attach_url, model, variant))
or prompt_mode != "auto"
)
if (
not has_override
and isinstance(command, list)
and all(isinstance(part, str) for part in command)
):
return command
stdout = result.get("stdout")
stdout_text = stdout if isinstance(stdout, str) else ""
+2 -5
View File
@@ -249,13 +249,10 @@ def render_trial_report_markdown(report: TrialReport) -> str:
)
lines.append("")
if report.opencode is not None:
if report.opencode is not None and report.opencode.session_id:
lines.append("## OpenCode Resume")
lines.append("")
if report.opencode.session_id:
lines.append(f"- Session: `{report.opencode.session_id}`")
else:
lines.append("- Session: not captured")
lines.append(f"- Session: `{report.opencode.session_id}`")
if report.opencode.attach_url:
lines.append(f"- Attach URL: `{report.opencode.attach_url}`")
if report.opencode.resume_command:
+1 -1
View File
@@ -154,7 +154,7 @@ def main(argv: list[str] | None = None) -> int:
)
else:
print(_display_command(command))
except ValueError as exc:
except (OSError, ValueError) as exc:
parser.error(str(exc))
return 0
+1
View File
@@ -6,6 +6,7 @@ param(
[object[]]$models = @( # object[] because ModelProfile is not known yet
[ModelProfile]::new("opencode/deepseek-v4-flash-free", "max"),
[ModelProfile]::new("opencode/mimo-v2.5-free", "high")
# Disabled after repeated provider timeouts and resource-exhausted failures.
# [ModelProfile]::new("opencode/nemotron-3-ultra-free", "high")
)
)
+1
View File
@@ -45,6 +45,7 @@ class ModelProfile:
DEFAULT_MODELS = (
ModelProfile("opencode/deepseek-v4-flash-free", "max"),
ModelProfile("opencode/mimo-v2.5-free", "high"),
# Disabled after repeated provider timeouts and resource-exhausted failures.
# ModelProfile("opencode/nemotron-3-ultra-free", "high"),
)
+4 -1
View File
@@ -42,8 +42,11 @@ uv run wf --config examples/lda_report_workflow/wf.config.json --local run inspe
Resume with selected issues:
Copy one or more actual issue ids from the interrupt payload and replace the
placeholder below.
```powershell
uv run wf --config examples/lda_report_workflow/wf.config.json --local run resume <run_id> --payload '{"approved":true,"selected_issue_ids":["risk-1"],"comment":"Create selected issues."}'
uv run wf --config examples/lda_report_workflow/wf.config.json --local run resume <run_id> --payload '{"approved":true,"selected_issue_ids":["<issue_id_from_interrupt>"],"comment":"Create selected issues."}'
```
## Cleanup
@@ -71,6 +71,11 @@ def build_workflow() -> Workflow:
input=[{"path": "input.selected_documents", "target": "names"}],
output=[{"source": "documents", "target": "state.documents"}],
)
reset_board = builder.use_ref(
"local.issue_board.reset_issue_board",
id="reset_board",
input=[{"path": "input.board_path", "target": "board_path"}],
)
analyze = builder.use_ref(
"local.lda_report.analyze_documents",
id="analyze",
@@ -164,7 +169,8 @@ def build_workflow() -> Workflow:
end_completed = builder.end("completed", id="end_completed")
end_cancelled = builder.end("cancelled", id="end_cancelled")
builder.set_entry_point(read_docs)
builder.set_entry_point(reset_board)
builder.connect(reset_board, "ok", read_docs)
builder.connect(read_docs, "ok", analyze)
builder.connect(analyze, "ok", build_report)
builder.connect(build_report, "ok", draft_issues)
@@ -1,5 +1,10 @@
{
"edges": [
{
"from": "reset_board",
"outcome": "ok",
"to": "read_docs"
},
{
"from": "read_docs",
"outcome": "ok",
@@ -86,6 +91,21 @@
"timeout_seconds": null,
"type": "node"
},
{
"desc": null,
"id": "reset_board",
"input": [
{
"path": "input.board_path",
"target": "board_path"
}
],
"node": "local.issue_board.reset_issue_board",
"output": [],
"retry": null,
"timeout_seconds": null,
"type": "node"
},
{
"desc": null,
"id": "analyze",
@@ -368,7 +388,7 @@
"required": [],
"type": "object"
},
"start": "read_docs",
"start": "reset_board",
"state_schema": {
"properties": {
"analysis": {