feat: parse structured challenge reports
This commit is contained in:
@@ -50,6 +50,27 @@ The baseline challenge does not require Playwright MCP. The score is based on
|
|||||||
whether the agent used the workflow product path and produced the expected
|
whether the agent used the workflow product path and produced the expected
|
||||||
workflow output.
|
workflow output.
|
||||||
|
|
||||||
|
## Required Agent Report
|
||||||
|
|
||||||
|
The prompt asks the agent to end with one fenced YAML block:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
challenge_report:
|
||||||
|
used_product_path: true
|
||||||
|
used_helper_script: false
|
||||||
|
workflow_file: "path/to/workflow.json-or-yaml"
|
||||||
|
deployment_id: "browser_click_case_study.default"
|
||||||
|
run_id: "run_..."
|
||||||
|
before_clicked: false
|
||||||
|
after_clicked: true
|
||||||
|
run_failed: false
|
||||||
|
leftover_processes: false
|
||||||
|
notes: "short explanation"
|
||||||
|
```
|
||||||
|
|
||||||
|
The harness parses this report first. If the report is missing, it falls back to
|
||||||
|
best-effort prose classification.
|
||||||
|
|
||||||
## Classification
|
## Classification
|
||||||
|
|
||||||
Each trial is classified as one of:
|
Each trial is classified as one of:
|
||||||
|
|||||||
@@ -35,4 +35,24 @@ You may inspect and use it. A successful final answer must include:
|
|||||||
- evidence that `after.clicked` is `true`,
|
- evidence that `after.clicked` is `true`,
|
||||||
- whether any server/browser process remains running.
|
- whether any server/browser process remains running.
|
||||||
|
|
||||||
|
End your answer with exactly one fenced YAML block using this shape:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
challenge_report:
|
||||||
|
used_product_path: true
|
||||||
|
used_helper_script: false
|
||||||
|
workflow_file: "path/to/workflow.json-or-yaml"
|
||||||
|
deployment_id: "browser_click_case_study.default"
|
||||||
|
run_id: "run_..."
|
||||||
|
before_clicked: false
|
||||||
|
after_clicked: true
|
||||||
|
run_failed: false
|
||||||
|
leftover_processes: false
|
||||||
|
notes: "short explanation"
|
||||||
|
```
|
||||||
|
|
||||||
|
Set `used_product_path` to true only if the workflow was applied and run through
|
||||||
|
the `wf` CLI or `wf-rpc-server` path. Set `used_helper_script` to true if you
|
||||||
|
created a Python script to drive `WorkflowApi` directly.
|
||||||
|
|
||||||
If something fails, report the exact command and error instead of hiding it.
|
If something fails, report the exact command and error instead of hiding it.
|
||||||
|
|||||||
@@ -8,6 +8,8 @@ from dataclasses import asdict, dataclass
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Literal
|
from typing import Any, Literal
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
|
||||||
Classification = Literal[
|
Classification = Literal[
|
||||||
"success",
|
"success",
|
||||||
"workflow_script",
|
"workflow_script",
|
||||||
@@ -111,6 +113,10 @@ def _event_text(event: dict[str, Any]) -> str | None:
|
|||||||
|
|
||||||
|
|
||||||
def classify_output(text: str) -> Classification:
|
def classify_output(text: str) -> Classification:
|
||||||
|
report = extract_challenge_report(text)
|
||||||
|
if report is not None:
|
||||||
|
return classify_challenge_report(report)
|
||||||
|
|
||||||
lowered = text.lower()
|
lowered = text.lower()
|
||||||
product_command_markers = [
|
product_command_markers = [
|
||||||
"wf ",
|
"wf ",
|
||||||
@@ -121,9 +127,7 @@ def classify_output(text: str) -> Classification:
|
|||||||
"run id",
|
"run id",
|
||||||
"run_",
|
"run_",
|
||||||
]
|
]
|
||||||
used_product_command = any(
|
used_product_command = any(marker in lowered for marker in product_command_markers)
|
||||||
marker in lowered for marker in product_command_markers
|
|
||||||
)
|
|
||||||
has_workflow_evidence = any(
|
has_workflow_evidence = any(
|
||||||
marker in lowered for marker in workflow_evidence_markers
|
marker in lowered for marker in workflow_evidence_markers
|
||||||
)
|
)
|
||||||
@@ -162,6 +166,59 @@ def classify_output(text: str) -> Classification:
|
|||||||
return "unknown"
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def extract_challenge_report(text: str) -> dict[str, Any] | None:
|
||||||
|
"""Extract the required final YAML challenge report from agent output."""
|
||||||
|
marker = "```yaml"
|
||||||
|
start = text.lower().rfind(marker)
|
||||||
|
if start == -1:
|
||||||
|
return None
|
||||||
|
body_start = text.find("\n", start)
|
||||||
|
if body_start == -1:
|
||||||
|
return None
|
||||||
|
end = text.find("```", body_start + 1)
|
||||||
|
if end == -1:
|
||||||
|
return None
|
||||||
|
raw_yaml = text[body_start + 1 : end]
|
||||||
|
loaded = yaml.safe_load(raw_yaml)
|
||||||
|
if not isinstance(loaded, dict):
|
||||||
|
return None
|
||||||
|
report = loaded.get("challenge_report")
|
||||||
|
return report if isinstance(report, dict) else None
|
||||||
|
|
||||||
|
|
||||||
|
def classify_challenge_report(report: dict[str, Any]) -> Classification:
|
||||||
|
used_product_path = report.get("used_product_path") is True
|
||||||
|
used_helper_script = report.get("used_helper_script") is True
|
||||||
|
workflow_file = report.get("workflow_file")
|
||||||
|
deployment_id = report.get("deployment_id")
|
||||||
|
run_id = report.get("run_id")
|
||||||
|
before_clicked = report.get("before_clicked")
|
||||||
|
after_clicked = report.get("after_clicked")
|
||||||
|
failed = report.get("run_failed") is True
|
||||||
|
|
||||||
|
if failed:
|
||||||
|
return "run_failed"
|
||||||
|
if used_helper_script:
|
||||||
|
return "workflow_script"
|
||||||
|
if (
|
||||||
|
used_product_path
|
||||||
|
and isinstance(workflow_file, str)
|
||||||
|
and bool(workflow_file)
|
||||||
|
and isinstance(deployment_id, str)
|
||||||
|
and bool(deployment_id)
|
||||||
|
and isinstance(run_id, str)
|
||||||
|
and bool(run_id)
|
||||||
|
and before_clicked is False
|
||||||
|
and after_clicked is True
|
||||||
|
):
|
||||||
|
return "success"
|
||||||
|
if not used_product_path and (
|
||||||
|
before_clicked is not None or after_clicked is not None
|
||||||
|
):
|
||||||
|
return "workflow_not_used"
|
||||||
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
|
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
|
||||||
marker_index = text.find(marker)
|
marker_index = text.find(marker)
|
||||||
if marker_index == -1:
|
if marker_index == -1:
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ dependencies = [
|
|||||||
"mcp[cli,rich]>=1",
|
"mcp[cli,rich]>=1",
|
||||||
"openapi-core>=0.19",
|
"openapi-core>=0.19",
|
||||||
"pydantic>=2",
|
"pydantic>=2",
|
||||||
|
"pyyaml>=6.0.3",
|
||||||
"typer>=0.24.2",
|
"typer>=0.24.2",
|
||||||
"uvicorn>=0.46.0",
|
"uvicorn>=0.46.0",
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -6,7 +6,9 @@ from pathlib import Path
|
|||||||
from examples.agent_challenges.browser_click_challenge.run_opencode_trials import (
|
from examples.agent_challenges.browser_click_challenge.run_opencode_trials import (
|
||||||
TrialConfig,
|
TrialConfig,
|
||||||
build_opencode_command,
|
build_opencode_command,
|
||||||
|
classify_challenge_report,
|
||||||
classify_output,
|
classify_output,
|
||||||
|
extract_challenge_report,
|
||||||
parse_opencode_output,
|
parse_opencode_output,
|
||||||
trial_output_path,
|
trial_output_path,
|
||||||
)
|
)
|
||||||
@@ -52,7 +54,9 @@ def test_build_opencode_command_with_attach(tmp_path: Path) -> None:
|
|||||||
|
|
||||||
|
|
||||||
def test_parse_opencode_output_reads_json_object() -> None:
|
def test_parse_opencode_output_reads_json_object() -> None:
|
||||||
payload = {"text": "wf run start demo.default\nbefore.clicked false\nafter.clicked true"}
|
payload = {
|
||||||
|
"text": "wf run start demo.default\nbefore.clicked false\nafter.clicked true"
|
||||||
|
}
|
||||||
|
|
||||||
parsed = parse_opencode_output(json.dumps(payload))
|
parsed = parse_opencode_output(json.dumps(payload))
|
||||||
|
|
||||||
@@ -108,6 +112,91 @@ def test_classify_output_success() -> None:
|
|||||||
assert result == "success"
|
assert result == "success"
|
||||||
|
|
||||||
|
|
||||||
|
def test_extract_challenge_report_from_yaml_block() -> None:
|
||||||
|
text = """
|
||||||
|
The run worked.
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
challenge_report:
|
||||||
|
used_product_path: true
|
||||||
|
used_helper_script: false
|
||||||
|
workflow_file: "browser-click.workflow.yaml"
|
||||||
|
deployment_id: "browser_click_case_study.default"
|
||||||
|
run_id: "run_123"
|
||||||
|
before_clicked: false
|
||||||
|
after_clicked: true
|
||||||
|
run_failed: false
|
||||||
|
leftover_processes: false
|
||||||
|
notes: "ok"
|
||||||
|
```
|
||||||
|
"""
|
||||||
|
|
||||||
|
report = extract_challenge_report(text)
|
||||||
|
|
||||||
|
assert report is not None
|
||||||
|
assert report["used_product_path"] is True
|
||||||
|
assert report["before_clicked"] is False
|
||||||
|
assert report["after_clicked"] is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_challenge_report_success() -> None:
|
||||||
|
result = classify_challenge_report(
|
||||||
|
{
|
||||||
|
"used_product_path": True,
|
||||||
|
"used_helper_script": False,
|
||||||
|
"workflow_file": "browser-click.workflow.yaml",
|
||||||
|
"deployment_id": "browser_click_case_study.default",
|
||||||
|
"run_id": "run_123",
|
||||||
|
"before_clicked": False,
|
||||||
|
"after_clicked": True,
|
||||||
|
"run_failed": False,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result == "success"
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_output_prefers_yaml_report() -> None:
|
||||||
|
result = classify_output(
|
||||||
|
"""
|
||||||
|
Some prose that would otherwise be ambiguous.
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
challenge_report:
|
||||||
|
used_product_path: true
|
||||||
|
used_helper_script: false
|
||||||
|
workflow_file: "browser-click.workflow.yaml"
|
||||||
|
deployment_id: "browser_click_case_study.default"
|
||||||
|
run_id: "run_123"
|
||||||
|
before_clicked: false
|
||||||
|
after_clicked: true
|
||||||
|
run_failed: false
|
||||||
|
leftover_processes: false
|
||||||
|
notes: "ok"
|
||||||
|
```
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result == "success"
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_challenge_report_detects_helper_script() -> None:
|
||||||
|
result = classify_challenge_report(
|
||||||
|
{
|
||||||
|
"used_product_path": False,
|
||||||
|
"used_helper_script": True,
|
||||||
|
"workflow_file": "",
|
||||||
|
"deployment_id": "browser_click_case_study.default",
|
||||||
|
"run_id": "run_123",
|
||||||
|
"before_clicked": False,
|
||||||
|
"after_clicked": True,
|
||||||
|
"run_failed": False,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result == "workflow_script"
|
||||||
|
|
||||||
|
|
||||||
def test_classify_output_workflow_script() -> None:
|
def test_classify_output_workflow_script() -> None:
|
||||||
result = classify_output(
|
result = classify_output(
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -662,6 +662,7 @@ dependencies = [
|
|||||||
{ name = "mcp", extra = ["cli", "rich"] },
|
{ name = "mcp", extra = ["cli", "rich"] },
|
||||||
{ name = "openapi-core" },
|
{ name = "openapi-core" },
|
||||||
{ name = "pydantic" },
|
{ name = "pydantic" },
|
||||||
|
{ name = "pyyaml" },
|
||||||
{ name = "typer" },
|
{ name = "typer" },
|
||||||
{ name = "uvicorn" },
|
{ name = "uvicorn" },
|
||||||
]
|
]
|
||||||
@@ -686,6 +687,7 @@ requires-dist = [
|
|||||||
{ name = "mcp", extras = ["cli", "rich"], specifier = ">=1" },
|
{ name = "mcp", extras = ["cli", "rich"], specifier = ">=1" },
|
||||||
{ name = "openapi-core", specifier = ">=0.19" },
|
{ name = "openapi-core", specifier = ">=0.19" },
|
||||||
{ name = "pydantic", specifier = ">=2" },
|
{ name = "pydantic", specifier = ">=2" },
|
||||||
|
{ name = "pyyaml", specifier = ">=6.0.3" },
|
||||||
{ name = "typer", specifier = ">=0.24.2" },
|
{ name = "typer", specifier = ">=0.24.2" },
|
||||||
{ name = "uvicorn", specifier = ">=0.46.0" },
|
{ name = "uvicorn", specifier = ">=0.46.0" },
|
||||||
]
|
]
|
||||||
|
|||||||
Reference in New Issue
Block a user