refactor: extract browser challenge harness modules

This commit is contained in:
lda
2026-06-15 19:38:46 +07:00 Verified
parent 8d67e87a71
commit 66eaa4a67d
9 changed files with 761 additions and 304 deletions
@@ -74,6 +74,20 @@ examples/agent_challenges/browser_click_challenge/workspaces/
files can be graded by hand without polluting the repository. The template's
store directory is also ignored.
## Saving Trial Reports
To save an agent's final answer from a harness result into its trial workspace:
```powershell
uv run python examples/agent_challenges/browser_click_challenge/save_trial_report.py `
--from-result examples/agent_challenges/browser_click_challenge/results/<trial>.json
```
The script infers the workspace from the result file and only writes
`<trial>/final-report.md`. Add evaluator commentary by editing that file after it
is saved. For manually copied reports, pass an explicit workspace and
`--input-file final-answer.md`.
## Optional Opencode Server Attachment
`--attach` is opencode's server attach flag. It connects this non-interactive
@@ -113,6 +127,18 @@ challenge_report:
after_clicked: true
run_failed: false
leftover_processes: false
read:
skills: true
docs: true
product_code: false
adjacent_attempts: false
prior_store: false
existing_solution: false
attempts:
total: 1
failed: 0
missed_requirements:
- "none"
notes: "short explanation"
```
@@ -0,0 +1,103 @@
from __future__ import annotations
from dataclasses import dataclass
from pathlib import Path
from typing import Literal
Classification = Literal[
"success",
"workflow_script",
"workflow_not_used",
"run_failed",
"timeout",
"parse_error",
"unknown",
]
CHALLENGE_REPORT_REQUIRED_FIELDS = {
"used_product_path",
"used_helper_script",
"workflow_file",
"deployment_id",
"run_id",
"before_clicked",
"after_clicked",
"run_failed",
"leftover_processes",
"read",
"attempts",
"missed_requirements",
"notes",
}
CHALLENGE_REPORT_READ_FIELDS = {
"skills",
"docs",
"product_code",
"adjacent_attempts",
"prior_store",
"existing_solution",
}
CHALLENGE_REPORT_ATTEMPT_FIELDS = {"total", "failed"}
ROOT = Path(__file__).resolve().parents[3]
CHALLENGE_DIR = Path(__file__).resolve().parent
DEFAULT_RESULTS_DIR = CHALLENGE_DIR / "results"
DEFAULT_WORKSPACES_DIR = CHALLENGE_DIR / "workspaces"
DEFAULT_WORKSPACE_TEMPLATE = CHALLENGE_DIR / "workspace_template"
DEFAULT_PROMPT = DEFAULT_WORKSPACE_TEMPLATE / "prompt.md"
DEFAULT_SERVER_PORT = 8772
EXAMPLE_CONFIG = ROOT / "examples" / "browser_click_workflow" / "wf.config.json"
EXAMPLE_SOURCE_ROOT = ROOT / "examples" / "browser_click_workflow"
EXAMPLE_CONFIG_ARG = "examples/browser_click_workflow/wf.config.json"
LOCAL_WF_COMMAND_PREFIX = f"uv run wf --config {EXAMPLE_CONFIG_ARG} --local"
@dataclass(frozen=True, slots=True)
class TrialConfig:
model: str
variant: str
prompt_path: Path
attach_url: str | None
timeout_seconds: int
wf_command_prefix: str
server_context: str
@dataclass(frozen=True, slots=True)
class TrialWorkspace:
"""Per-trial scratch area copied from the challenge workspace template."""
root: Path
config_path: Path
prompt_path: Path
def render_prompt(
prompt_path: Path,
*,
wf_command_prefix: str,
server_context: str,
) -> str:
return (
prompt_path.read_text(encoding="utf-8")
.replace("{{wf_command_prefix}}", wf_command_prefix)
.replace("{{server_context}}", server_context)
)
def rpc_url_for_port(port: int) -> str:
return f"http://127.0.0.1:{port}/rpc"
def server_command(*, port: int) -> list[str]:
return [
"uv",
"run",
"wf-rpc-server",
"--config",
EXAMPLE_CONFIG_ARG,
"--host",
"127.0.0.1",
"--port",
str(port),
]
@@ -0,0 +1,183 @@
from __future__ import annotations
from typing import Any
import yaml
from examples.agent_challenges.browser_click_challenge.challenge import (
CHALLENGE_REPORT_ATTEMPT_FIELDS,
CHALLENGE_REPORT_READ_FIELDS,
CHALLENGE_REPORT_REQUIRED_FIELDS,
Classification,
)
def classify_output(text: str) -> Classification:
report = extract_challenge_report(text)
if report is not None:
return classify_challenge_report(report)
lowered = text.lower()
product_command_markers = [
"wf ",
"wf-rpc-server",
]
workflow_evidence_markers = [
"deployment",
"run id",
"run_",
]
used_product_command = any(marker in lowered for marker in product_command_markers)
has_workflow_evidence = any(
marker in lowered for marker in workflow_evidence_markers
)
used_helper_script = (
"uv run python" in lowered
or "python examples/" in lowered
or "run_workflow.py" in lowered
)
failed = any(
marker in lowered
for marker in [
"error:",
"failed",
"traceback",
"exception",
"validation failed",
]
)
before_false = _contains_bool_marker(lowered, "before.clicked", "false") or (
'"before"' in lowered and '"clicked": false' in lowered
)
after_true = _contains_bool_marker(lowered, "after.clicked", "true") or (
'"after"' in lowered and '"clicked": true' in lowered
)
if used_product_command and before_false and after_true and not failed:
return "success"
if has_workflow_evidence and used_helper_script and before_false and after_true:
return "workflow_script"
if (used_product_command or has_workflow_evidence) and failed:
return "run_failed"
if not has_workflow_evidence and (
before_false or after_true or "playwright" in lowered
):
return "workflow_not_used"
return "unknown"
def extract_challenge_report(text: str) -> dict[str, Any] | None:
"""Extract the required final YAML challenge report from agent output."""
marker = "```yaml"
start = text.lower().rfind(marker)
if start == -1:
return None
body_start = text.find("\n", start)
if body_start == -1:
return None
end = text.find("```", body_start + 1)
if end == -1:
return None
raw_yaml = text[body_start + 1 : end]
loaded = yaml.safe_load(raw_yaml)
if not isinstance(loaded, dict):
return None
report = loaded.get("challenge_report")
return report if isinstance(report, dict) else None
def classify_challenge_report(report: dict[str, Any]) -> Classification:
if challenge_report_schema_errors(report):
return "unknown"
used_product_path = report.get("used_product_path") is True
used_helper_script = report.get("used_helper_script") is True
workflow_file = report.get("workflow_file")
deployment_id = report.get("deployment_id")
run_id = report.get("run_id")
before_clicked = report.get("before_clicked")
after_clicked = report.get("after_clicked")
failed = report.get("run_failed") is True
if failed:
return "run_failed"
if used_helper_script:
return "workflow_script"
if (
used_product_path
and isinstance(workflow_file, str)
and bool(workflow_file)
and isinstance(deployment_id, str)
and bool(deployment_id)
and isinstance(run_id, str)
and bool(run_id)
and before_clicked is False
and after_clicked is True
):
return "success"
if not used_product_path and (
before_clicked is not None or after_clicked is not None
):
return "workflow_not_used"
return "unknown"
def challenge_report_schema_errors(report: dict[str, Any]) -> list[str]:
"""Return human-readable schema errors for the final YAML report block."""
errors: list[str] = []
missing = sorted(CHALLENGE_REPORT_REQUIRED_FIELDS.difference(report))
errors.extend(f"missing challenge_report.{field}" for field in missing)
for field in (
"used_product_path",
"used_helper_script",
"before_clicked",
"after_clicked",
"run_failed",
"leftover_processes",
):
if field in report and not isinstance(report[field], bool):
errors.append(f"challenge_report.{field} must be boolean")
for field in ("workflow_file", "deployment_id", "run_id", "notes"):
if field in report and not isinstance(report[field], str):
errors.append(f"challenge_report.{field} must be string")
read = report.get("read")
if isinstance(read, dict):
missing_read = sorted(CHALLENGE_REPORT_READ_FIELDS.difference(read))
errors.extend(f"missing challenge_report.read.{field}" for field in missing_read)
for field in CHALLENGE_REPORT_READ_FIELDS.intersection(read):
if not isinstance(read[field], bool):
errors.append(f"challenge_report.read.{field} must be boolean")
elif "read" in report:
errors.append("challenge_report.read must be object")
attempts = report.get("attempts")
if isinstance(attempts, dict):
missing_attempts = sorted(CHALLENGE_REPORT_ATTEMPT_FIELDS.difference(attempts))
errors.extend(
f"missing challenge_report.attempts.{field}"
for field in missing_attempts
)
for field in CHALLENGE_REPORT_ATTEMPT_FIELDS.intersection(attempts):
value = attempts[field]
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
errors.append(f"challenge_report.attempts.{field} must be >= 0 integer")
elif "attempts" in report:
errors.append("challenge_report.attempts must be object")
missed = report.get("missed_requirements")
if "missed_requirements" in report and (
not isinstance(missed, list)
or any(not isinstance(item, str) for item in missed)
):
errors.append("challenge_report.missed_requirements must be list of strings")
return errors
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
marker_index = text.find(marker)
if marker_index == -1:
return False
return value in text[marker_index : marker_index + 80]
@@ -0,0 +1,98 @@
from __future__ import annotations
import json
from typing import Any
from examples.agent_challenges.browser_click_challenge.challenge import (
TrialConfig,
render_prompt,
)
def build_opencode_command(config: TrialConfig) -> list[str]:
prompt_text = render_prompt(
config.prompt_path,
wf_command_prefix=config.wf_command_prefix,
server_context=config.server_context,
)
command = [
"opencode",
"run",
]
if config.attach_url is not None:
command.extend(["--attach", config.attach_url])
command.extend(
[
prompt_text,
"--format",
"json",
"--model",
config.model,
"--variant",
config.variant,
]
)
return command
def parse_opencode_output(stdout: str) -> dict[str, Any]:
text = stdout.strip()
if not text:
raise ValueError("opencode produced no JSON output")
try:
parsed = json.loads(text)
except json.JSONDecodeError:
parsed = _parse_jsonl_tail(text)
if not isinstance(parsed, dict):
raise ValueError("opencode output was not a JSON object")
return parsed
def _parse_jsonl_tail(text: str) -> dict[str, Any]:
last_error: json.JSONDecodeError | None = None
parsed_events: list[dict[str, Any]] = []
for line in reversed(text.splitlines()):
stripped = line.strip()
if not stripped:
continue
try:
parsed = json.loads(stripped)
except json.JSONDecodeError as exc:
last_error = exc
continue
if isinstance(parsed, dict):
parsed_events.append(parsed)
event_text = _event_text(parsed)
if event_text is not None:
return {"text": event_text, "event": parsed}
if parsed_events:
return parsed_events[0]
if last_error is not None:
raise last_error
raise ValueError("opencode output did not contain JSON lines")
def _event_text(event: dict[str, Any]) -> str | None:
"""Extract assistant text from opencode JSON events.
`opencode run --format json` emits many events. The final event can be a
`step_finish`; the answer text is usually the previous `text` event under
`part.text`.
"""
part = event.get("part")
if isinstance(part, dict):
text = part.get("text")
if isinstance(text, str):
return text
text = event.get("text")
return text if isinstance(text, str) else None
def _result_text(parsed: dict[str, Any]) -> str:
for key in ("text", "message", "content", "output"):
value = parsed.get(key)
if isinstance(value, str):
return value
return json.dumps(parsed, sort_keys=True)
@@ -0,0 +1,94 @@
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from examples.agent_challenges.browser_click_challenge.classification import (
extract_challenge_report,
)
from examples.agent_challenges.browser_click_challenge.opencode_io import _result_text
def save_report(
*,
workspace: Path,
report_text: str,
output_name: str = "final-report.md",
) -> Path:
"""Copy one finished trial report into an existing trial workspace."""
if not workspace.is_dir():
raise ValueError(f"workspace does not exist or is not a directory: {workspace}")
output_path = workspace / output_name
output_path.write_text(report_text.rstrip() + "\n", encoding="utf-8")
return output_path
def report_from_result(result_path: Path) -> tuple[Path, str]:
"""Return inferred workspace and final report text from one harness result.
The workspace is usually recovered from the serialized prompt path. Older
or hand-edited result files may only expose the workflow path inside the
final YAML report, so that path remains a secondary inference source.
"""
result = json.loads(result_path.read_text(encoding="utf-8"))
if not isinstance(result, dict):
raise ValueError("result file must contain a JSON object")
parsed = result.get("parsed")
if not isinstance(parsed, dict):
raise ValueError("result file is missing parsed output")
report_text = _result_text(parsed)
workspace = _workspace_from_result(result, report_text)
return workspace, report_text
def _workspace_from_result(result: dict[str, object], report_text: str) -> Path:
config = result.get("config")
if isinstance(config, dict):
prompt_path = config.get("prompt_path")
if isinstance(prompt_path, str) and prompt_path:
return Path(prompt_path).parent
report = extract_challenge_report(report_text)
if report is not None:
workflow_file = report.get("workflow_file")
if isinstance(workflow_file, str) and workflow_file:
return Path(workflow_file).parent
raise ValueError("could not infer trial workspace from result")
def _read_report_text(input_file: Path | None) -> str:
if input_file is None:
return sys.stdin.read()
return input_file.read_text(encoding="utf-8")
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("workspace", type=Path, nargs="?")
parser.add_argument("--from-result", type=Path, default=None)
parser.add_argument("--input-file", type=Path, default=None)
parser.add_argument("--output-name", default="final-report.md")
args = parser.parse_args(argv)
try:
if args.from_result is not None:
if args.workspace is not None or args.input_file is not None:
parser.error("--from-result cannot be combined with workspace/input")
workspace, report_text = report_from_result(args.from_result)
else:
if args.workspace is None:
parser.error("workspace is required unless --from-result is used")
workspace = args.workspace
report_text = _read_report_text(args.input_file)
output_path = save_report(
workspace=workspace,
report_text=report_text,
output_name=args.output_name,
)
except ValueError as exc:
parser.error(str(exc))
print(output_path.as_posix())
return 0
@@ -9,62 +9,90 @@ import sys
import time
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import Any, Literal
import yaml
Classification = Literal[
"success",
"workflow_script",
"workflow_not_used",
"run_failed",
"timeout",
"parse_error",
"unknown",
]
from typing import Any
ROOT = Path(__file__).resolve().parents[3]
CHALLENGE_DIR = Path(__file__).resolve().parent
DEFAULT_PROMPT = CHALLENGE_DIR / "prompt.md"
DEFAULT_RESULTS_DIR = CHALLENGE_DIR / "results"
DEFAULT_WORKSPACES_DIR = CHALLENGE_DIR / "workspaces"
DEFAULT_WORKSPACE_TEMPLATE = CHALLENGE_DIR / "workspace_template"
DEFAULT_SERVER_PORT = 8772
EXAMPLE_CONFIG = ROOT / "examples" / "browser_click_workflow" / "wf.config.json"
EXAMPLE_SOURCE_ROOT = ROOT / "examples" / "browser_click_workflow"
EXAMPLE_CONFIG_ARG = "examples/browser_click_workflow/wf.config.json"
LOCAL_WF_COMMAND_PREFIX = f"uv run wf --config {EXAMPLE_CONFIG_ARG} --local"
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
def rpc_url_for_port(port: int) -> str:
return f"http://127.0.0.1:{port}/rpc"
def server_command(*, port: int) -> list[str]:
return [
"uv",
"run",
"wf-rpc-server",
"--config",
from examples.agent_challenges.browser_click_challenge.challenge import ( # noqa: E402
CHALLENGE_DIR,
CHALLENGE_REPORT_ATTEMPT_FIELDS,
CHALLENGE_REPORT_READ_FIELDS,
CHALLENGE_REPORT_REQUIRED_FIELDS,
DEFAULT_PROMPT,
DEFAULT_RESULTS_DIR,
DEFAULT_SERVER_PORT,
DEFAULT_WORKSPACE_TEMPLATE,
DEFAULT_WORKSPACES_DIR,
EXAMPLE_CONFIG,
EXAMPLE_CONFIG_ARG,
"--host",
"127.0.0.1",
"--port",
str(port),
]
def render_prompt(
prompt_path: Path,
*,
wf_command_prefix: str,
server_context: str,
) -> str:
return (
prompt_path.read_text(encoding="utf-8")
.replace("{{wf_command_prefix}}", wf_command_prefix)
.replace("{{server_context}}", server_context)
EXAMPLE_SOURCE_ROOT,
LOCAL_WF_COMMAND_PREFIX,
Classification,
TrialConfig,
TrialWorkspace,
render_prompt,
rpc_url_for_port,
server_command,
)
from examples.agent_challenges.browser_click_challenge.classification import ( # noqa: E402
_contains_bool_marker,
challenge_report_schema_errors,
classify_challenge_report,
classify_output,
extract_challenge_report,
)
from examples.agent_challenges.browser_click_challenge.opencode_io import ( # noqa: E402
_event_text,
_parse_jsonl_tail,
_result_text,
build_opencode_command,
parse_opencode_output,
)
__all__ = [
"CHALLENGE_DIR",
"CHALLENGE_REPORT_ATTEMPT_FIELDS",
"CHALLENGE_REPORT_READ_FIELDS",
"CHALLENGE_REPORT_REQUIRED_FIELDS",
"DEFAULT_PROMPT",
"DEFAULT_RESULTS_DIR",
"DEFAULT_SERVER_PORT",
"DEFAULT_WORKSPACE_TEMPLATE",
"DEFAULT_WORKSPACES_DIR",
"EXAMPLE_CONFIG",
"EXAMPLE_CONFIG_ARG",
"EXAMPLE_SOURCE_ROOT",
"LOCAL_WF_COMMAND_PREFIX",
"ROOT",
"Classification",
"ManagedServer",
"TrialConfig",
"TrialWorkspace",
"_contains_bool_marker",
"_event_text",
"_parse_jsonl_tail",
"_result_text",
"build_opencode_command",
"challenge_report_schema_errors",
"classify_challenge_report",
"classify_output",
"extract_challenge_report",
"main",
"parse_opencode_output",
"prepare_trial_workspace",
"render_prompt",
"rpc_url_for_port",
"run_trial",
"server_command",
"start_server",
"starting_trial_index",
"stop_server",
"trial_output_path",
"wf_command_prefix_for_config",
"write_trial_config",
]
@dataclass(slots=True)
@@ -73,15 +101,6 @@ class ManagedServer:
rpc_url: str
@dataclass(frozen=True, slots=True)
class TrialWorkspace:
"""Per-trial scratch area copied from the challenge workspace template."""
root: Path
config_path: Path
prompt_path: Path
def start_server(*, port: int, timeout_seconds: int = 30) -> ManagedServer:
command = server_command(port=port)
process = subprocess.Popen(
@@ -234,212 +253,6 @@ def stop_server(process: subprocess.Popen[str]) -> None:
process.wait(timeout=10)
@dataclass(frozen=True, slots=True)
class TrialConfig:
model: str
variant: str
prompt_path: Path
attach_url: str | None
timeout_seconds: int
wf_command_prefix: str
server_context: str
def build_opencode_command(config: TrialConfig) -> list[str]:
prompt_text = render_prompt(
config.prompt_path,
wf_command_prefix=config.wf_command_prefix,
server_context=config.server_context,
)
command = [
"opencode",
"run",
]
if config.attach_url is not None:
command.extend(["--attach", config.attach_url])
command.extend(
[
prompt_text,
"--format",
"json",
"--model",
config.model,
"--variant",
config.variant,
]
)
return command
def parse_opencode_output(stdout: str) -> dict[str, Any]:
text = stdout.strip()
if not text:
raise ValueError("opencode produced no JSON output")
try:
parsed = json.loads(text)
except json.JSONDecodeError:
parsed = _parse_jsonl_tail(text)
if not isinstance(parsed, dict):
raise ValueError("opencode output was not a JSON object")
return parsed
def _parse_jsonl_tail(text: str) -> dict[str, Any]:
last_error: json.JSONDecodeError | None = None
parsed_events: list[dict[str, Any]] = []
for line in reversed(text.splitlines()):
stripped = line.strip()
if not stripped:
continue
try:
parsed = json.loads(stripped)
except json.JSONDecodeError as exc:
last_error = exc
continue
if isinstance(parsed, dict):
parsed_events.append(parsed)
event_text = _event_text(parsed)
if event_text is not None:
return {"text": event_text, "event": parsed}
if parsed_events:
return parsed_events[0]
if last_error is not None:
raise last_error
raise ValueError("opencode output did not contain JSON lines")
def _event_text(event: dict[str, Any]) -> str | None:
"""Extract assistant text from opencode JSON events.
`opencode run --format json` emits many events. The final event can be a
`step_finish`; the answer text is usually the previous `text` event under
`part.text`.
"""
part = event.get("part")
if isinstance(part, dict):
text = part.get("text")
if isinstance(text, str):
return text
text = event.get("text")
return text if isinstance(text, str) else None
def classify_output(text: str) -> Classification:
report = extract_challenge_report(text)
if report is not None:
return classify_challenge_report(report)
lowered = text.lower()
product_command_markers = [
"wf ",
"wf-rpc-server",
]
workflow_evidence_markers = [
"deployment",
"run id",
"run_",
]
used_product_command = any(marker in lowered for marker in product_command_markers)
has_workflow_evidence = any(
marker in lowered for marker in workflow_evidence_markers
)
used_helper_script = (
"uv run python" in lowered
or "python examples/" in lowered
or "run_workflow.py" in lowered
)
failed = any(
marker in lowered
for marker in [
"error:",
"failed",
"traceback",
"exception",
"validation failed",
]
)
before_false = _contains_bool_marker(lowered, "before.clicked", "false") or (
'"before"' in lowered and '"clicked": false' in lowered
)
after_true = _contains_bool_marker(lowered, "after.clicked", "true") or (
'"after"' in lowered and '"clicked": true' in lowered
)
if used_product_command and before_false and after_true and not failed:
return "success"
if has_workflow_evidence and used_helper_script and before_false and after_true:
return "workflow_script"
if (used_product_command or has_workflow_evidence) and failed:
return "run_failed"
if not has_workflow_evidence and (
before_false or after_true or "playwright" in lowered
):
return "workflow_not_used"
return "unknown"
def extract_challenge_report(text: str) -> dict[str, Any] | None:
"""Extract the required final YAML challenge report from agent output."""
marker = "```yaml"
start = text.lower().rfind(marker)
if start == -1:
return None
body_start = text.find("\n", start)
if body_start == -1:
return None
end = text.find("```", body_start + 1)
if end == -1:
return None
raw_yaml = text[body_start + 1 : end]
loaded = yaml.safe_load(raw_yaml)
if not isinstance(loaded, dict):
return None
report = loaded.get("challenge_report")
return report if isinstance(report, dict) else None
def classify_challenge_report(report: dict[str, Any]) -> Classification:
used_product_path = report.get("used_product_path") is True
used_helper_script = report.get("used_helper_script") is True
workflow_file = report.get("workflow_file")
deployment_id = report.get("deployment_id")
run_id = report.get("run_id")
before_clicked = report.get("before_clicked")
after_clicked = report.get("after_clicked")
failed = report.get("run_failed") is True
if failed:
return "run_failed"
if used_helper_script:
return "workflow_script"
if (
used_product_path
and isinstance(workflow_file, str)
and bool(workflow_file)
and isinstance(deployment_id, str)
and bool(deployment_id)
and isinstance(run_id, str)
and bool(run_id)
and before_clicked is False
and after_clicked is True
):
return "success"
if not used_product_path and (
before_clicked is not None or after_clicked is not None
):
return "workflow_not_used"
return "unknown"
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
marker_index = text.find(marker)
if marker_index == -1:
return False
return value in text[marker_index : marker_index + 80]
def trial_output_path(results_dir: Path, *, model: str, index: int) -> Path:
return results_dir / f"{_safe_model_name(model)}-trial-{index:03d}.json"
@@ -496,14 +309,6 @@ def run_trial(config: TrialConfig, *, index: int, results_dir: Path) -> dict[str
return payload
def _result_text(parsed: dict[str, Any]) -> str:
for key in ("text", "message", "content", "output"):
value = parsed.get(key)
if isinstance(value, str):
return value
return json.dumps(parsed, sort_keys=True)
def _jsonable_config(config: TrialConfig) -> dict[str, Any]:
payload = asdict(config)
payload["prompt_path"] = str(config.prompt_path)
@@ -527,7 +332,7 @@ def main(argv: list[str] | None = None) -> int:
parser.add_argument("--model", default="opencode/mimo-v2.5-free")
parser.add_argument("--variant", default="high")
parser.add_argument("--trials", type=int, default=1)
parser.add_argument("--timeout-seconds", type=int, default=600)
parser.add_argument("--timeout-seconds", type=int, default=1000)
parser.add_argument(
"--attach",
dest="attach_url",
@@ -0,0 +1,21 @@
from __future__ import annotations
import sys
from pathlib import Path
# Support direct execution as `python examples/.../save_trial_report.py`.
ROOT = Path(__file__).resolve().parents[3]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from examples.agent_challenges.browser_click_challenge.reports import ( # noqa: E402
main,
report_from_result,
save_report,
)
__all__ = ["main", "report_from_result", "save_report"]
if __name__ == "__main__":
raise SystemExit(main())
@@ -55,14 +55,17 @@ The repository already includes a deterministic source example at:
examples/browser_click_workflow/
```
You may inspect and use it. A successful final answer must include:
You may inspect and use it.
Your final answer should include a short human-readable report with:
- the commands you ran,
- the deployment id,
- the run id if one was produced,
- evidence that `before.clicked` is `false`,
- evidence that `after.clicked` is `true`,
- whether any server/browser process remains running.
- whether any server/browser process remains running,
- important failed attempts and how you fixed them.
End your answer with exactly one fenced YAML block using this shape:
@@ -77,6 +80,18 @@ challenge_report:
after_clicked: true
run_failed: false
leftover_processes: false
read:
skills: true
docs: true
product_code: false
adjacent_attempts: false
prior_store: false
existing_solution: false
attempts:
total: 1
failed: 0
missed_requirements:
- "none"
notes: "short explanation"
```
@@ -3,21 +3,62 @@ from __future__ import annotations
import json
from pathlib import Path
from examples.agent_challenges.browser_click_challenge.run_opencode_trials import (
from examples.agent_challenges.browser_click_challenge.challenge import (
LOCAL_WF_COMMAND_PREFIX,
TrialConfig,
build_opencode_command,
render_prompt,
server_command,
)
from examples.agent_challenges.browser_click_challenge.classification import (
challenge_report_schema_errors,
classify_challenge_report,
classify_output,
extract_challenge_report,
)
from examples.agent_challenges.browser_click_challenge.opencode_io import (
build_opencode_command,
parse_opencode_output,
)
from examples.agent_challenges.browser_click_challenge.reports import (
report_from_result,
save_report,
)
from examples.agent_challenges.browser_click_challenge.run_opencode_trials import (
prepare_trial_workspace,
render_prompt,
server_command,
starting_trial_index,
trial_output_path,
wf_command_prefix_for_config,
)
from examples.agent_challenges.browser_click_challenge.save_trial_report import (
main as save_trial_report_main,
)
def _valid_challenge_report(**overrides: object) -> dict[str, object]:
report: dict[str, object] = {
"used_product_path": True,
"used_helper_script": False,
"workflow_file": "browser-click.workflow.yaml",
"deployment_id": "browser_click_case_study.default",
"run_id": "run_123",
"before_clicked": False,
"after_clicked": True,
"run_failed": False,
"leftover_processes": False,
"read": {
"skills": True,
"docs": True,
"product_code": False,
"adjacent_attempts": False,
"prior_store": False,
"existing_solution": False,
},
"attempts": {"total": 1, "failed": 0},
"missed_requirements": ["none"],
"notes": "ok",
}
report.update(overrides)
return report
def test_build_opencode_command_without_attach(tmp_path: Path) -> None:
@@ -137,6 +178,18 @@ def test_extract_challenge_report_from_yaml_block() -> None:
after_clicked: true
run_failed: false
leftover_processes: false
read:
skills: true
docs: true
product_code: false
adjacent_attempts: false
prior_store: false
existing_solution: false
attempts:
total: 1
failed: 0
missed_requirements:
- "none"
notes: "ok"
```
"""
@@ -150,22 +203,19 @@ def test_extract_challenge_report_from_yaml_block() -> None:
def test_classify_challenge_report_success() -> None:
result = classify_challenge_report(
{
"used_product_path": True,
"used_helper_script": False,
"workflow_file": "browser-click.workflow.yaml",
"deployment_id": "browser_click_case_study.default",
"run_id": "run_123",
"before_clicked": False,
"after_clicked": True,
"run_failed": False,
}
)
result = classify_challenge_report(_valid_challenge_report())
assert result == "success"
def test_challenge_report_schema_errors_reject_missing_read_block() -> None:
report = _valid_challenge_report()
report.pop("read")
assert challenge_report_schema_errors(report) == ["missing challenge_report.read"]
assert classify_challenge_report(report) == "unknown"
def test_classify_output_prefers_yaml_report() -> None:
result = classify_output(
"""
@@ -182,6 +232,18 @@ def test_classify_output_prefers_yaml_report() -> None:
after_clicked: true
run_failed: false
leftover_processes: false
read:
skills: true
docs: true
product_code: false
adjacent_attempts: false
prior_store: false
existing_solution: false
attempts:
total: 1
failed: 0
missed_requirements:
- "none"
notes: "ok"
```
"""
@@ -192,16 +254,11 @@ def test_classify_output_prefers_yaml_report() -> None:
def test_classify_challenge_report_detects_helper_script() -> None:
result = classify_challenge_report(
{
"used_product_path": False,
"used_helper_script": True,
"workflow_file": "",
"deployment_id": "browser_click_case_study.default",
"run_id": "run_123",
"before_clicked": False,
"after_clicked": True,
"run_failed": False,
}
_valid_challenge_report(
used_product_path=False,
used_helper_script=True,
workflow_file="",
)
)
assert result == "workflow_script"
@@ -287,6 +344,61 @@ def test_prepare_trial_workspace_copies_template_to_model_trial_dir(
assert (prepared.root / ".gitignore").read_text(encoding="utf-8") == ".wf_store/\n"
def test_save_trial_report_copies_report_into_workspace(tmp_path: Path) -> None:
workspace = tmp_path / "trial"
workspace.mkdir()
output = save_report(workspace=workspace, report_text="# Report\n\nok\n")
assert output == workspace / "final-report.md"
assert output.read_text(encoding="utf-8") == "# Report\n\nok\n"
def test_report_from_result_infers_workspace_from_prompt_path(tmp_path: Path) -> None:
workspace = tmp_path / "trial"
workspace.mkdir()
result_path = tmp_path / "result.json"
result_path.write_text(
json.dumps(
{
"config": {"prompt_path": str(workspace / "prompt.md")},
"parsed": {"text": "# Report\n\nok"},
}
),
encoding="utf-8",
)
inferred_workspace, report_text = report_from_result(result_path)
assert inferred_workspace == workspace
assert report_text == "# Report\n\nok"
def test_save_trial_report_from_result_writes_inferred_workspace(
tmp_path: Path,
capsys,
) -> None:
workspace = tmp_path / "trial"
workspace.mkdir()
result_path = tmp_path / "result.json"
result_path.write_text(
json.dumps(
{
"config": {"prompt_path": str(workspace / "prompt.md")},
"parsed": {"text": "# Report\n\nok"},
}
),
encoding="utf-8",
)
assert save_trial_report_main(["--from-result", str(result_path)]) == 0
assert (workspace / "final-report.md").read_text(encoding="utf-8") == (
"# Report\n\nok\n"
)
assert (workspace / "final-report.md").as_posix() in capsys.readouterr().out
def test_prepare_trial_workspace_uses_next_available_directory(
tmp_path: Path,
) -> None: