refactor: share agent challenge harness
This commit is contained in:
@@ -0,0 +1,211 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
from examples.agent_challenges.classification import extract_challenge_report
|
||||
from examples.agent_challenges.reports import report_from_result
|
||||
|
||||
|
||||
def audit_from_result(
|
||||
result_path: Path,
|
||||
*,
|
||||
manual_classification: str,
|
||||
report_path: Path | None = None,
|
||||
audited_at: str | None = None,
|
||||
auditor: str = "human",
|
||||
read_overrides: dict[str, bool] | None = None,
|
||||
evidence_overrides: dict[str, object] | None = None,
|
||||
corrections: list[str] | None = None,
|
||||
notes: str = "",
|
||||
) -> tuple[Path, dict[str, Any]]:
|
||||
if report_path is None:
|
||||
workspace, report_text = report_from_result(result_path)
|
||||
else:
|
||||
workspace = report_path.parent
|
||||
report_text = report_path.read_text(encoding="utf-8")
|
||||
result = _load_result(result_path)
|
||||
challenge_report = extract_challenge_report(report_text) or {}
|
||||
read_flags = _dict_or_empty(challenge_report.get("read"))
|
||||
read_flags.update(read_overrides or {})
|
||||
evidence = _evidence_from_report(challenge_report)
|
||||
evidence.update(evidence_overrides or {})
|
||||
|
||||
audit = {
|
||||
"manual_audit": {
|
||||
"auditor": auditor,
|
||||
"audited_at": audited_at or _utc_now(),
|
||||
"auto_classification": _string_or_none(result.get("classification")),
|
||||
"manual_classification": manual_classification,
|
||||
"valid_product_run": _valid_product_run(challenge_report),
|
||||
"product_path_used": challenge_report.get("used_product_path") is True,
|
||||
"helper_script_used": challenge_report.get("used_helper_script") is True,
|
||||
"run_succeeded": challenge_report.get("run_failed") is False,
|
||||
"duration_seconds": result.get("duration_seconds"),
|
||||
"returncode": result.get("returncode"),
|
||||
"evidence": evidence,
|
||||
"read_flags": read_flags,
|
||||
"attempts": _dict_or_empty(challenge_report.get("attempts")),
|
||||
"missed_requirements": _list_or_empty(
|
||||
challenge_report.get("missed_requirements")
|
||||
),
|
||||
"agent_notes": challenge_report.get("notes"),
|
||||
"corrections": corrections or [],
|
||||
"notes": notes,
|
||||
}
|
||||
}
|
||||
return workspace, audit
|
||||
|
||||
|
||||
def save_manual_audit(
|
||||
result_path: Path,
|
||||
*,
|
||||
manual_classification: str,
|
||||
report_path: Path | None = None,
|
||||
audited_at: str | None = None,
|
||||
auditor: str = "human",
|
||||
read_overrides: dict[str, bool] | None = None,
|
||||
evidence_overrides: dict[str, object] | None = None,
|
||||
corrections: list[str] | None = None,
|
||||
notes: str = "",
|
||||
output_name: str = "manual-audit.yaml",
|
||||
) -> Path:
|
||||
workspace, audit = audit_from_result(
|
||||
result_path,
|
||||
manual_classification=manual_classification,
|
||||
report_path=report_path,
|
||||
audited_at=audited_at,
|
||||
auditor=auditor,
|
||||
read_overrides=read_overrides,
|
||||
evidence_overrides=evidence_overrides,
|
||||
corrections=corrections,
|
||||
notes=notes,
|
||||
)
|
||||
output_path = workspace / output_name
|
||||
output_path.write_text(
|
||||
yaml.safe_dump(audit, sort_keys=False, allow_unicode=True),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return output_path
|
||||
|
||||
|
||||
def _load_result(result_path: Path) -> dict[str, Any]:
|
||||
loaded = json.loads(result_path.read_text(encoding="utf-8"))
|
||||
if not isinstance(loaded, dict):
|
||||
raise ValueError("result file must contain a JSON object")
|
||||
return loaded
|
||||
|
||||
|
||||
def _evidence_from_report(report: dict[str, Any]) -> dict[str, object]:
|
||||
return {
|
||||
key: report[key]
|
||||
for key in (
|
||||
"deployment_id",
|
||||
"run_id",
|
||||
"before_clicked",
|
||||
"after_clicked",
|
||||
"leftover_processes",
|
||||
)
|
||||
if key in report
|
||||
}
|
||||
|
||||
|
||||
def _valid_product_run(report: dict[str, Any]) -> bool:
|
||||
return (
|
||||
report.get("used_product_path") is True
|
||||
and report.get("used_helper_script") is False
|
||||
and report.get("run_failed") is False
|
||||
)
|
||||
|
||||
|
||||
def _dict_or_empty(value: object) -> dict[str, Any]:
|
||||
return dict(value) if isinstance(value, dict) else {}
|
||||
|
||||
|
||||
def _list_or_empty(value: object) -> list[object]:
|
||||
return list(value) if isinstance(value, list) else []
|
||||
|
||||
|
||||
def _string_or_none(value: object) -> str | None:
|
||||
return value if isinstance(value, str) else None
|
||||
|
||||
|
||||
def _utc_now() -> str:
|
||||
return datetime.now(UTC).isoformat(timespec="seconds").replace("+00:00", "Z")
|
||||
|
||||
|
||||
def _parse_bool_assignment(value: str) -> tuple[str, bool]:
|
||||
key, separator, raw = value.partition("=")
|
||||
if separator != "=" or not key:
|
||||
raise ValueError("expected KEY=true or KEY=false")
|
||||
lowered = raw.lower()
|
||||
if lowered == "true":
|
||||
return key, True
|
||||
if lowered == "false":
|
||||
return key, False
|
||||
raise ValueError("boolean override value must be true or false")
|
||||
|
||||
|
||||
def _parse_value_assignment(value: str) -> tuple[str, object]:
|
||||
key, separator, raw = value.partition("=")
|
||||
if separator != "=" or not key:
|
||||
raise ValueError("expected KEY=VALUE")
|
||||
lowered = raw.lower()
|
||||
if lowered == "true":
|
||||
return key, True
|
||||
if lowered == "false":
|
||||
return key, False
|
||||
try:
|
||||
return key, int(raw)
|
||||
except ValueError:
|
||||
return key, raw
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--from-result", type=Path, required=True)
|
||||
parser.add_argument(
|
||||
"--from-report",
|
||||
type=Path,
|
||||
default=None,
|
||||
help=(
|
||||
"Use this final report Markdown for challenge_report YAML while "
|
||||
"keeping timeout/duration metadata from --from-result."
|
||||
),
|
||||
)
|
||||
parser.add_argument("--manual-classification", required=True)
|
||||
parser.add_argument("--auditor", default="human")
|
||||
parser.add_argument("--audited-at", default=None)
|
||||
parser.add_argument("--set-read", action="append", default=[])
|
||||
parser.add_argument("--set-evidence", action="append", default=[])
|
||||
parser.add_argument("--correction", action="append", default=[])
|
||||
parser.add_argument("--notes", default="")
|
||||
parser.add_argument("--output-name", default="manual-audit.yaml")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
try:
|
||||
read_overrides = dict(_parse_bool_assignment(item) for item in args.set_read)
|
||||
evidence_overrides = dict(
|
||||
_parse_value_assignment(item) for item in args.set_evidence
|
||||
)
|
||||
output_path = save_manual_audit(
|
||||
args.from_result,
|
||||
manual_classification=args.manual_classification,
|
||||
report_path=args.from_report,
|
||||
audited_at=args.audited_at,
|
||||
auditor=args.auditor,
|
||||
read_overrides=read_overrides,
|
||||
evidence_overrides=evidence_overrides,
|
||||
corrections=list(args.correction),
|
||||
notes=args.notes,
|
||||
output_name=args.output_name,
|
||||
)
|
||||
except ValueError as exc:
|
||||
parser.error(str(exc))
|
||||
print(output_path.as_posix())
|
||||
return 0
|
||||
@@ -3,6 +3,11 @@
|
||||
This harness runs agent trials against the browser-click workflow challenge.
|
||||
It is evidence tooling, not product runtime code.
|
||||
|
||||
The harness mechanics (workspace preparation, opencode I/O, report handling,
|
||||
trial runner) live in the shared modules under
|
||||
[`examples/agent_challenges/`](../). The browser-click challenge provides
|
||||
challenge-specific metadata, prompt template, classification, and defaults.
|
||||
|
||||
The deterministic workflow example is:
|
||||
|
||||
```text
|
||||
@@ -102,6 +107,26 @@ The script infers the workspace from the result file and only writes
|
||||
is saved. For manually copied reports, pass an explicit workspace and
|
||||
`--input-file final-answer.md`.
|
||||
|
||||
## Saving Manual Audits
|
||||
|
||||
Keep `final-report.md` as captured agent output. If manual review finds that
|
||||
the agent's YAML self-report was wrong or incomplete, write a sidecar audit:
|
||||
|
||||
```powershell
|
||||
uv run python examples/agent_challenges/browser_click_challenge/save_manual_audit.py `
|
||||
--from-result examples/agent_challenges/browser_click_challenge/results/<trial>.json `
|
||||
--manual-classification success_code_assisted `
|
||||
--set-read product_code=true `
|
||||
--set-evidence trace_count=3 `
|
||||
--correction "read.product_code: agent reported false, audited true" `
|
||||
--notes "Valid product run, but raw plan shape came from product code/tests."
|
||||
```
|
||||
|
||||
The script infers the workspace, run id, deployment id, automatic
|
||||
classification, read flags, attempts, and notes from the result file. It writes
|
||||
`<trial>/manual-audit.yaml`. Later benchmark summaries should treat the sidecar
|
||||
as the human override, not mutate the captured agent report.
|
||||
|
||||
## Optional Opencode Server Attachment
|
||||
|
||||
`--attach` is opencode's server attach flag. It connects this non-interactive
|
||||
@@ -173,4 +198,22 @@ Each trial is classified as one of:
|
||||
- `parse_error`: the harness could not read opencode JSON/JSONL output.
|
||||
- `unknown`: no clear success or failure signal was found.
|
||||
|
||||
## Shared Harness Modules
|
||||
|
||||
The generic modules in `examples/agent_challenges/` provide reusable harness
|
||||
logic:
|
||||
|
||||
| Module | Purpose |
|
||||
|--------|---------|
|
||||
| `workspace.py` | `ChallengeDef`, `TrialConfig`, `TrialWorkspace`, `prepare_trial_workspace`, `write_trial_config`, `starting_trial_index`, `wf_command_prefix_for_config`, `render_prompt`, `server_command` |
|
||||
| `runner.py` | `ManagedServer`, `start_server`, `run_trial`, `main` — generic trial runner parameterized by `ChallengeDef` and a classification function |
|
||||
| `opencode_io.py` | `build_opencode_command`, `parse_opencode_output`, `result_text` |
|
||||
| `reports.py` | `save_report`, `report_from_result`, `save_report_from_result_payload` |
|
||||
| `classification.py` | `extract_challenge_report` (generic YAML extraction), `_contains_bool_marker` |
|
||||
|
||||
The browser-click challenge's `run_opencode_trials.py` and `save_trial_report.py`
|
||||
are thin wrappers that pass `BROWSER_CLICK_DEF` and the browser-click
|
||||
classification function to the generic runner. The same pattern can be used to
|
||||
add new challenges without duplicating the harness.
|
||||
|
||||
Committed tests cover harness logic only. They do not invoke opencode.
|
||||
|
||||
@@ -1,9 +1,17 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
from examples.agent_challenges.workspace import ( # noqa: F401 - re-exported for backward compat
|
||||
ChallengeDef,
|
||||
TrialConfig,
|
||||
TrialWorkspace,
|
||||
render_prompt,
|
||||
rpc_url_for_port,
|
||||
server_command,
|
||||
)
|
||||
|
||||
Classification = Literal[
|
||||
"success",
|
||||
"workflow_script",
|
||||
@@ -41,63 +49,28 @@ CHALLENGE_REPORT_ATTEMPT_FIELDS = {"total", "failed"}
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[3]
|
||||
CHALLENGE_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_RESULTS_DIR = CHALLENGE_DIR / "results"
|
||||
DEFAULT_WORKSPACES_DIR = CHALLENGE_DIR / "workspaces"
|
||||
DEFAULT_WORKSPACE_TEMPLATE = CHALLENGE_DIR / "workspace_template"
|
||||
DEFAULT_PROMPT = DEFAULT_WORKSPACE_TEMPLATE / "prompt.md"
|
||||
DEFAULT_SERVER_PORT = 8772
|
||||
EXAMPLE_CONFIG = ROOT / "examples" / "browser_click_workflow" / "wf.config.json"
|
||||
EXAMPLE_SOURCE_ROOT = ROOT / "examples" / "browser_click_workflow"
|
||||
EXAMPLE_CONFIG_ARG = "examples/browser_click_workflow/wf.config.json"
|
||||
|
||||
BROWSER_CLICK_DEF = ChallengeDef(
|
||||
name="browser_click",
|
||||
source_root=ROOT / "examples" / "browser_click_workflow",
|
||||
source_id="local.browser_click",
|
||||
source_module="ops",
|
||||
source_registry="registry",
|
||||
store_root=".wf_browser_click_store",
|
||||
default_workspace_template=CHALLENGE_DIR / "workspace_template",
|
||||
default_workspaces_dir=CHALLENGE_DIR / "workspaces",
|
||||
default_results_dir=CHALLENGE_DIR / "results",
|
||||
default_prompt=CHALLENGE_DIR / "workspace_template" / "prompt.md",
|
||||
default_server_port=8772,
|
||||
server_config_arg="examples/browser_click_workflow/wf.config.json",
|
||||
)
|
||||
|
||||
EXAMPLE_CONFIG_ARG = BROWSER_CLICK_DEF.server_config_arg
|
||||
EXAMPLE_CONFIG = ROOT / EXAMPLE_CONFIG_ARG
|
||||
EXAMPLE_SOURCE_ROOT = BROWSER_CLICK_DEF.source_root
|
||||
LOCAL_WF_COMMAND_PREFIX = f"uv run wf --config {EXAMPLE_CONFIG_ARG} --local"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class TrialConfig:
|
||||
model: str
|
||||
variant: str
|
||||
prompt_path: Path
|
||||
attach_url: str | None
|
||||
timeout_seconds: int
|
||||
wf_command_prefix: str
|
||||
server_context: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class TrialWorkspace:
|
||||
"""Per-trial scratch area copied from the challenge workspace template."""
|
||||
|
||||
root: Path
|
||||
config_path: Path
|
||||
prompt_path: Path
|
||||
|
||||
|
||||
def render_prompt(
|
||||
prompt_path: Path,
|
||||
*,
|
||||
wf_command_prefix: str,
|
||||
server_context: str,
|
||||
) -> str:
|
||||
return (
|
||||
prompt_path.read_text(encoding="utf-8")
|
||||
.replace("{{wf_command_prefix}}", wf_command_prefix)
|
||||
.replace("{{server_context}}", server_context)
|
||||
)
|
||||
|
||||
|
||||
def rpc_url_for_port(port: int) -> str:
|
||||
return f"http://127.0.0.1:{port}/rpc"
|
||||
|
||||
|
||||
def server_command(*, port: int) -> list[str]:
|
||||
return [
|
||||
"uv",
|
||||
"run",
|
||||
"wf-rpc-server",
|
||||
"--config",
|
||||
EXAMPLE_CONFIG_ARG,
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
str(port),
|
||||
]
|
||||
DEFAULT_PROMPT = BROWSER_CLICK_DEF.default_prompt
|
||||
DEFAULT_RESULTS_DIR = BROWSER_CLICK_DEF.default_results_dir
|
||||
DEFAULT_WORKSPACES_DIR = BROWSER_CLICK_DEF.default_workspaces_dir
|
||||
DEFAULT_WORKSPACE_TEMPLATE = BROWSER_CLICK_DEF.default_workspace_template
|
||||
DEFAULT_SERVER_PORT = BROWSER_CLICK_DEF.default_server_port
|
||||
|
||||
@@ -2,14 +2,16 @@ from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
from examples.agent_challenges.browser_click_challenge.challenge import (
|
||||
CHALLENGE_REPORT_ATTEMPT_FIELDS,
|
||||
CHALLENGE_REPORT_READ_FIELDS,
|
||||
CHALLENGE_REPORT_REQUIRED_FIELDS,
|
||||
Classification,
|
||||
)
|
||||
from examples.agent_challenges.classification import (
|
||||
_contains_bool_marker,
|
||||
extract_challenge_report,
|
||||
)
|
||||
|
||||
|
||||
def classify_output(text: str) -> Classification:
|
||||
@@ -66,26 +68,6 @@ def classify_output(text: str) -> Classification:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def extract_challenge_report(text: str) -> dict[str, Any] | None:
|
||||
"""Extract the required final YAML challenge report from agent output."""
|
||||
marker = "```yaml"
|
||||
start = text.lower().rfind(marker)
|
||||
if start == -1:
|
||||
return None
|
||||
body_start = text.find("\n", start)
|
||||
if body_start == -1:
|
||||
return None
|
||||
end = text.find("```", body_start + 1)
|
||||
if end == -1:
|
||||
return None
|
||||
raw_yaml = text[body_start + 1 : end]
|
||||
loaded = yaml.safe_load(raw_yaml)
|
||||
if not isinstance(loaded, dict):
|
||||
return None
|
||||
report = loaded.get("challenge_report")
|
||||
return report if isinstance(report, dict) else None
|
||||
|
||||
|
||||
def classify_challenge_report(report: dict[str, Any]) -> Classification:
|
||||
if challenge_report_schema_errors(report):
|
||||
return "unknown"
|
||||
@@ -123,7 +105,6 @@ def classify_challenge_report(report: dict[str, Any]) -> Classification:
|
||||
|
||||
|
||||
def challenge_report_schema_errors(report: dict[str, Any]) -> list[str]:
|
||||
"""Return human-readable schema errors for the final YAML report block."""
|
||||
errors: list[str] = []
|
||||
missing = sorted(CHALLENGE_REPORT_REQUIRED_FIELDS.difference(report))
|
||||
errors.extend(f"missing challenge_report.{field}" for field in missing)
|
||||
@@ -175,10 +156,3 @@ def challenge_report_schema_errors(report: dict[str, Any]) -> list[str]:
|
||||
errors.append("challenge_report.missed_requirements must be list of strings")
|
||||
|
||||
return errors
|
||||
|
||||
|
||||
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
|
||||
marker_index = text.find(marker)
|
||||
if marker_index == -1:
|
||||
return False
|
||||
return value in text[marker_index : marker_index + 80]
|
||||
|
||||
@@ -1,98 +1,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from examples.agent_challenges.browser_click_challenge.challenge import (
|
||||
TrialConfig,
|
||||
render_prompt,
|
||||
from examples.agent_challenges.opencode_io import ( # noqa: F401
|
||||
_event_text,
|
||||
_parse_jsonl_tail,
|
||||
build_opencode_command,
|
||||
parse_opencode_output,
|
||||
result_text,
|
||||
)
|
||||
|
||||
|
||||
def build_opencode_command(config: TrialConfig) -> list[str]:
|
||||
prompt_text = render_prompt(
|
||||
config.prompt_path,
|
||||
wf_command_prefix=config.wf_command_prefix,
|
||||
server_context=config.server_context,
|
||||
)
|
||||
command = [
|
||||
"opencode",
|
||||
"run",
|
||||
]
|
||||
if config.attach_url is not None:
|
||||
command.extend(["--attach", config.attach_url])
|
||||
command.extend(
|
||||
[
|
||||
prompt_text,
|
||||
"--format",
|
||||
"json",
|
||||
"--model",
|
||||
config.model,
|
||||
"--variant",
|
||||
config.variant,
|
||||
]
|
||||
)
|
||||
return command
|
||||
|
||||
|
||||
def parse_opencode_output(stdout: str) -> dict[str, Any]:
|
||||
text = stdout.strip()
|
||||
if not text:
|
||||
raise ValueError("opencode produced no JSON output")
|
||||
|
||||
try:
|
||||
parsed = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
parsed = _parse_jsonl_tail(text)
|
||||
|
||||
if not isinstance(parsed, dict):
|
||||
raise ValueError("opencode output was not a JSON object")
|
||||
return parsed
|
||||
|
||||
|
||||
def _parse_jsonl_tail(text: str) -> dict[str, Any]:
|
||||
last_error: json.JSONDecodeError | None = None
|
||||
parsed_events: list[dict[str, Any]] = []
|
||||
for line in reversed(text.splitlines()):
|
||||
stripped = line.strip()
|
||||
if not stripped:
|
||||
continue
|
||||
try:
|
||||
parsed = json.loads(stripped)
|
||||
except json.JSONDecodeError as exc:
|
||||
last_error = exc
|
||||
continue
|
||||
if isinstance(parsed, dict):
|
||||
parsed_events.append(parsed)
|
||||
event_text = _event_text(parsed)
|
||||
if event_text is not None:
|
||||
return {"text": event_text, "event": parsed}
|
||||
if parsed_events:
|
||||
return parsed_events[0]
|
||||
if last_error is not None:
|
||||
raise last_error
|
||||
raise ValueError("opencode output did not contain JSON lines")
|
||||
|
||||
|
||||
def _event_text(event: dict[str, Any]) -> str | None:
|
||||
"""Extract assistant text from opencode JSON events.
|
||||
|
||||
`opencode run --format json` emits many events. The final event can be a
|
||||
`step_finish`; the answer text is usually the previous `text` event under
|
||||
`part.text`.
|
||||
"""
|
||||
part = event.get("part")
|
||||
if isinstance(part, dict):
|
||||
text = part.get("text")
|
||||
if isinstance(text, str):
|
||||
return text
|
||||
text = event.get("text")
|
||||
return text if isinstance(text, str) else None
|
||||
|
||||
|
||||
def result_text(parsed: dict[str, Any]) -> str:
|
||||
for key in ("text", "message", "content", "output"):
|
||||
value = parsed.get(key)
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
return json.dumps(parsed, sort_keys=True)
|
||||
|
||||
@@ -1,131 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from examples.agent_challenges.browser_click_challenge.classification import (
|
||||
extract_challenge_report,
|
||||
from examples.agent_challenges.reports import ( # noqa: F401
|
||||
main,
|
||||
report_from_result,
|
||||
save_report,
|
||||
save_report_from_result_payload,
|
||||
)
|
||||
from examples.agent_challenges.browser_click_challenge.opencode_io import (
|
||||
parse_opencode_output,
|
||||
result_text,
|
||||
)
|
||||
|
||||
|
||||
def save_report(
|
||||
*,
|
||||
workspace: Path,
|
||||
report_text: str,
|
||||
output_name: str = "final-report.md",
|
||||
) -> Path:
|
||||
"""Copy one finished trial report into an existing trial workspace."""
|
||||
if not workspace.is_dir():
|
||||
raise ValueError(f"workspace does not exist or is not a directory: {workspace}")
|
||||
output_path = workspace / output_name
|
||||
output_path.write_text(report_text.rstrip() + "\n", encoding="utf-8")
|
||||
return output_path
|
||||
|
||||
|
||||
def report_from_result(result_path: Path) -> tuple[Path, str]:
|
||||
"""Return inferred workspace and final report text from one harness result.
|
||||
|
||||
The workspace is usually recovered from the serialized prompt path. Older
|
||||
or hand-edited result files may only expose the workflow path inside the
|
||||
final YAML report, so that path remains a secondary inference source.
|
||||
"""
|
||||
result = json.loads(result_path.read_text(encoding="utf-8"))
|
||||
if not isinstance(result, dict):
|
||||
raise ValueError("result file must contain a JSON object")
|
||||
report_text = _report_text_from_result(result)
|
||||
workspace = _workspace_from_result(result, report_text)
|
||||
return workspace, report_text
|
||||
|
||||
|
||||
def save_report_from_result_payload(
|
||||
result: dict[str, object],
|
||||
*,
|
||||
output_name: str = "final-report.md",
|
||||
) -> Path:
|
||||
"""Save final report text from an in-memory harness result payload."""
|
||||
report_text = _report_text_from_result(result)
|
||||
workspace = _workspace_from_result(result, report_text)
|
||||
return save_report(
|
||||
workspace=workspace,
|
||||
report_text=report_text,
|
||||
output_name=output_name,
|
||||
)
|
||||
|
||||
|
||||
def _report_text_from_result(result: dict[str, object]) -> str:
|
||||
parsed = result.get("parsed")
|
||||
if isinstance(parsed, dict):
|
||||
return result_text(parsed)
|
||||
|
||||
stdout = result.get("stdout")
|
||||
if isinstance(stdout, str) and stdout.strip():
|
||||
try:
|
||||
recovered = parse_opencode_output(stdout)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
"result file is missing parsed output and stdout has no report text"
|
||||
) from exc
|
||||
return result_text(recovered)
|
||||
|
||||
raise ValueError("result file is missing parsed output")
|
||||
|
||||
|
||||
def _workspace_from_result(result: dict[str, object], report_text: str) -> Path:
|
||||
config = result.get("config")
|
||||
if isinstance(config, dict):
|
||||
prompt_path = config.get("prompt_path")
|
||||
if isinstance(prompt_path, str) and prompt_path:
|
||||
return Path(prompt_path).parent
|
||||
|
||||
report = extract_challenge_report(report_text)
|
||||
if report is not None:
|
||||
workflow_file = report.get("workflow_file")
|
||||
if isinstance(workflow_file, str) and workflow_file:
|
||||
return Path(workflow_file).parent
|
||||
|
||||
raise ValueError("could not infer trial workspace from result")
|
||||
|
||||
|
||||
def _read_report_text(input_file: Path | None) -> str:
|
||||
if input_file is None:
|
||||
if sys.stdin.isatty():
|
||||
raise ValueError(
|
||||
"manual report mode needs piped stdin, --input-file, or --from-result"
|
||||
)
|
||||
return sys.stdin.read()
|
||||
return input_file.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("workspace", type=Path, nargs="?")
|
||||
parser.add_argument("--from-result", type=Path, default=None)
|
||||
parser.add_argument("--input-file", type=Path, default=None)
|
||||
parser.add_argument("--output-name", default="final-report.md")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
try:
|
||||
if args.from_result is not None:
|
||||
if args.workspace is not None or args.input_file is not None:
|
||||
parser.error("--from-result cannot be combined with workspace/input")
|
||||
workspace, report_text = report_from_result(args.from_result)
|
||||
else:
|
||||
if args.workspace is None:
|
||||
parser.error("workspace is required unless --from-result is used")
|
||||
workspace = args.workspace
|
||||
report_text = _read_report_text(args.input_file)
|
||||
output_path = save_report(
|
||||
workspace=workspace,
|
||||
report_text=report_text,
|
||||
output_name=args.output_name,
|
||||
)
|
||||
except ValueError as exc:
|
||||
parser.error(str(exc))
|
||||
print(output_path.as_posix())
|
||||
return 0
|
||||
|
||||
@@ -1,25 +1,22 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from dataclasses import asdict, dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
# Support direct execution as `python examples/.../run_opencode_trials.py`.
|
||||
# ruff: noqa: I001 - imports must stay after sys.path.insert
|
||||
ROOT = Path(__file__).resolve().parents[3]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from examples.agent_challenges.browser_click_challenge.challenge import ( # noqa: E402
|
||||
BROWSER_CLICK_DEF,
|
||||
CHALLENGE_DIR,
|
||||
CHALLENGE_REPORT_ATTEMPT_FIELDS,
|
||||
CHALLENGE_REPORT_READ_FIELDS,
|
||||
CHALLENGE_REPORT_REQUIRED_FIELDS,
|
||||
Classification,
|
||||
DEFAULT_PROMPT,
|
||||
DEFAULT_RESULTS_DIR,
|
||||
DEFAULT_SERVER_PORT,
|
||||
@@ -29,12 +26,6 @@ from examples.agent_challenges.browser_click_challenge.challenge import ( # noq
|
||||
EXAMPLE_CONFIG_ARG,
|
||||
EXAMPLE_SOURCE_ROOT,
|
||||
LOCAL_WF_COMMAND_PREFIX,
|
||||
Classification,
|
||||
TrialConfig,
|
||||
TrialWorkspace,
|
||||
render_prompt,
|
||||
rpc_url_for_port,
|
||||
server_command,
|
||||
)
|
||||
from examples.agent_challenges.browser_click_challenge.classification import ( # noqa: E402
|
||||
_contains_bool_marker,
|
||||
@@ -43,22 +34,42 @@ from examples.agent_challenges.browser_click_challenge.classification import (
|
||||
classify_output,
|
||||
extract_challenge_report,
|
||||
)
|
||||
from examples.agent_challenges.browser_click_challenge.opencode_io import ( # noqa: E402
|
||||
from examples.agent_challenges.opencode_io import ( # noqa: E402
|
||||
_event_text,
|
||||
_parse_jsonl_tail,
|
||||
build_opencode_command,
|
||||
parse_opencode_output,
|
||||
result_text,
|
||||
)
|
||||
from examples.agent_challenges.browser_click_challenge.reports import ( # noqa: E402
|
||||
from examples.agent_challenges.reports import ( # noqa: E402
|
||||
save_report_from_result_payload,
|
||||
)
|
||||
from examples.agent_challenges.runner import ( # noqa: E402
|
||||
ManagedServer,
|
||||
main as _generic_main,
|
||||
run_trial as _generic_run_trial,
|
||||
start_server as _generic_start_server,
|
||||
stop_server as _generic_stop_server,
|
||||
)
|
||||
from examples.agent_challenges.workspace import ( # noqa: E402
|
||||
TrialConfig,
|
||||
TrialWorkspace,
|
||||
prepare_trial_workspace as _generic_prepare,
|
||||
render_prompt,
|
||||
rpc_url_for_port,
|
||||
server_command,
|
||||
starting_trial_index as _generic_starting_index,
|
||||
trial_output_path as _generic_trial_output_path,
|
||||
wf_command_prefix_for_config as _generic_wf_prefix,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"BROWSER_CLICK_DEF",
|
||||
"CHALLENGE_DIR",
|
||||
"CHALLENGE_REPORT_ATTEMPT_FIELDS",
|
||||
"CHALLENGE_REPORT_READ_FIELDS",
|
||||
"CHALLENGE_REPORT_REQUIRED_FIELDS",
|
||||
"Classification",
|
||||
"DEFAULT_PROMPT",
|
||||
"DEFAULT_RESULTS_DIR",
|
||||
"DEFAULT_SERVER_PORT",
|
||||
@@ -68,9 +79,8 @@ __all__ = [
|
||||
"EXAMPLE_CONFIG_ARG",
|
||||
"EXAMPLE_SOURCE_ROOT",
|
||||
"LOCAL_WF_COMMAND_PREFIX",
|
||||
"ROOT",
|
||||
"Classification",
|
||||
"ManagedServer",
|
||||
"ROOT",
|
||||
"TrialConfig",
|
||||
"TrialWorkspace",
|
||||
"_contains_bool_marker",
|
||||
@@ -95,55 +105,11 @@ __all__ = [
|
||||
"stop_server",
|
||||
"trial_output_path",
|
||||
"wf_command_prefix_for_config",
|
||||
"write_trial_config",
|
||||
]
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class ManagedServer:
|
||||
process: subprocess.Popen[str]
|
||||
rpc_url: str
|
||||
|
||||
|
||||
def start_server(*, port: int, timeout_seconds: int = 30) -> ManagedServer:
|
||||
command = server_command(port=port)
|
||||
process = subprocess.Popen(
|
||||
command,
|
||||
cwd=ROOT,
|
||||
text=True,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
rpc_url = rpc_url_for_port(port)
|
||||
try:
|
||||
wait_for_status(rpc_url=rpc_url, timeout_seconds=timeout_seconds)
|
||||
except Exception:
|
||||
stop_server(process)
|
||||
raise
|
||||
return ManagedServer(process=process, rpc_url=rpc_url)
|
||||
|
||||
|
||||
def wait_for_status(*, rpc_url: str, timeout_seconds: int) -> None:
|
||||
deadline = time.monotonic() + timeout_seconds
|
||||
command = ["uv", "run", "wf", "--url", rpc_url, "status"]
|
||||
last_stderr = ""
|
||||
while time.monotonic() < deadline:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
cwd=ROOT,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=False,
|
||||
)
|
||||
if completed.returncode == 0:
|
||||
return
|
||||
last_stderr = completed.stderr
|
||||
time.sleep(0.5)
|
||||
raise RuntimeError(f"wf status did not become ready: {last_stderr}")
|
||||
|
||||
|
||||
def _safe_model_name(model: str) -> str:
|
||||
return model.replace("/", "_").replace(":", "_")
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
return _generic_main(BROWSER_CLICK_DEF, classify_output, argv)
|
||||
|
||||
|
||||
def prepare_trial_workspace(
|
||||
@@ -154,18 +120,39 @@ def prepare_trial_workspace(
|
||||
template_dir: Path = DEFAULT_WORKSPACE_TEMPLATE,
|
||||
source_root: Path = EXAMPLE_SOURCE_ROOT,
|
||||
) -> TrialWorkspace:
|
||||
"""Copy the authoring template into a clean ignored per-trial directory."""
|
||||
root = workspaces_dir / f"{_safe_model_name(model)}-trial-{index:03d}"
|
||||
if root.exists():
|
||||
raise FileExistsError(f"trial workspace already exists: {root}")
|
||||
shutil.copytree(template_dir, root)
|
||||
workspace = TrialWorkspace(
|
||||
root=root,
|
||||
config_path=root / "wf.config.json",
|
||||
prompt_path=root / "prompt.md",
|
||||
return _generic_prepare(
|
||||
BROWSER_CLICK_DEF,
|
||||
model=model,
|
||||
index=index,
|
||||
workspaces_dir=workspaces_dir,
|
||||
template_dir=template_dir,
|
||||
source_root=source_root,
|
||||
)
|
||||
write_trial_config(workspace.config_path, source_root=source_root)
|
||||
return workspace
|
||||
|
||||
|
||||
def run_trial(
|
||||
config: TrialConfig,
|
||||
*,
|
||||
index: int,
|
||||
results_dir: Path,
|
||||
) -> dict:
|
||||
return _generic_run_trial(
|
||||
config, index=index, results_dir=results_dir, classify_fn=classify_output
|
||||
)
|
||||
|
||||
|
||||
def start_server(
|
||||
*,
|
||||
port: int,
|
||||
timeout_seconds: int = 30,
|
||||
) -> ManagedServer:
|
||||
return _generic_start_server(
|
||||
BROWSER_CLICK_DEF, port=port, timeout_seconds=timeout_seconds
|
||||
)
|
||||
|
||||
|
||||
def stop_server(process: subprocess.Popen[str]) -> None:
|
||||
return _generic_stop_server(process)
|
||||
|
||||
|
||||
def starting_trial_index(
|
||||
@@ -174,310 +161,17 @@ def starting_trial_index(
|
||||
results_dir: Path,
|
||||
workspaces_dir: Path,
|
||||
) -> int:
|
||||
"""Return the next global trial number for this model across invocations."""
|
||||
safe_model = _safe_model_name(model)
|
||||
highest = 0
|
||||
for directory in (results_dir, workspaces_dir):
|
||||
if not directory.exists():
|
||||
continue
|
||||
for path in directory.iterdir():
|
||||
index = _trial_index_from_name(path.name, safe_model=safe_model)
|
||||
if index is not None:
|
||||
highest = max(highest, index)
|
||||
return highest + 1
|
||||
|
||||
|
||||
def _trial_index_from_name(name: str, *, safe_model: str) -> int | None:
|
||||
prefix = f"{safe_model}-trial-"
|
||||
if not name.startswith(prefix):
|
||||
return None
|
||||
suffix = name.removeprefix(prefix)
|
||||
if "." in suffix:
|
||||
suffix = suffix.split(".", 1)[0]
|
||||
if not suffix.isdigit():
|
||||
return None
|
||||
return int(suffix)
|
||||
|
||||
|
||||
def write_trial_config(config_path: Path, *, source_root: Path) -> None:
|
||||
"""Write a per-trial config with Python source path relative to config."""
|
||||
relative_source = Path(os.path.relpath(source_root, config_path.parent)).as_posix()
|
||||
config = {
|
||||
"version": 1,
|
||||
"client": {"target": {"kind": "local"}},
|
||||
"server": {
|
||||
"store": {"kind": "filesystem", "root": ".wf_browser_click_store"},
|
||||
"sources": [
|
||||
{
|
||||
"kind": "python",
|
||||
"id": "local.browser_click",
|
||||
"path": relative_source,
|
||||
"module": "ops",
|
||||
"registry": "registry",
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
config_path.write_text(
|
||||
json.dumps(config, indent=2, sort_keys=True) + "\n",
|
||||
encoding="utf-8",
|
||||
return _generic_starting_index(
|
||||
model=model, results_dir=results_dir, workspaces_dir=workspaces_dir
|
||||
)
|
||||
|
||||
|
||||
def wf_command_prefix_for_config(config_path: Path) -> str:
|
||||
path = config_path
|
||||
if not path.is_absolute():
|
||||
path = ROOT / path
|
||||
try:
|
||||
path_arg = path.resolve().relative_to(ROOT.resolve()).as_posix()
|
||||
except ValueError:
|
||||
path_arg = str(path.resolve())
|
||||
return f"uv run wf --config {path_arg} --local"
|
||||
|
||||
|
||||
def _display_path(path: Path) -> str:
|
||||
try:
|
||||
return path.resolve().relative_to(ROOT.resolve()).as_posix()
|
||||
except ValueError:
|
||||
return str(path.resolve())
|
||||
|
||||
|
||||
def stop_server(process: subprocess.Popen[str]) -> None:
|
||||
if process.poll() is not None:
|
||||
return
|
||||
if sys.platform == "win32":
|
||||
subprocess.run(
|
||||
["taskkill", "/F", "/T", "/PID", str(process.pid)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
)
|
||||
else:
|
||||
try:
|
||||
process.terminate()
|
||||
except OSError:
|
||||
return
|
||||
try:
|
||||
process.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
process.kill()
|
||||
process.wait(timeout=10)
|
||||
|
||||
|
||||
def trial_output_path(results_dir: Path, *, model: str, index: int) -> Path:
|
||||
return results_dir / f"{_safe_model_name(model)}-trial-{index:03d}.json"
|
||||
return _generic_trial_output_path(results_dir, model=model, index=index)
|
||||
|
||||
|
||||
def run_trial(config: TrialConfig, *, index: int, results_dir: Path) -> dict[str, Any]:
|
||||
command = build_opencode_command(config)
|
||||
started = time.monotonic()
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
cwd=ROOT,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
timeout=config.timeout_seconds,
|
||||
check=False,
|
||||
)
|
||||
duration_seconds = time.monotonic() - started
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
payload = {
|
||||
"index": index,
|
||||
"config": _jsonable_config(config),
|
||||
"command": command,
|
||||
"classification": "timeout",
|
||||
"duration_seconds": config.timeout_seconds,
|
||||
"returncode": None,
|
||||
"stdout": exc.stdout or "",
|
||||
"stderr": exc.stderr or "",
|
||||
"parsed": None,
|
||||
}
|
||||
_write_trial_report(payload)
|
||||
_write_trial_result(results_dir, config=config, index=index, payload=payload)
|
||||
return payload
|
||||
|
||||
parsed: dict[str, Any] | None
|
||||
try:
|
||||
parsed = parse_opencode_output(completed.stdout)
|
||||
text = result_text(parsed)
|
||||
classification = classify_output(text)
|
||||
except Exception:
|
||||
parsed = None
|
||||
classification = "parse_error"
|
||||
|
||||
payload = {
|
||||
"index": index,
|
||||
"config": _jsonable_config(config),
|
||||
"command": command,
|
||||
"classification": classification,
|
||||
"duration_seconds": duration_seconds,
|
||||
"returncode": completed.returncode,
|
||||
"stdout": completed.stdout,
|
||||
"stderr": completed.stderr,
|
||||
"parsed": parsed,
|
||||
}
|
||||
_write_trial_report(payload)
|
||||
_write_trial_result(results_dir, config=config, index=index, payload=payload)
|
||||
return payload
|
||||
|
||||
|
||||
def _jsonable_config(config: TrialConfig) -> dict[str, Any]:
|
||||
payload = asdict(config)
|
||||
payload["prompt_path"] = str(config.prompt_path)
|
||||
return payload
|
||||
|
||||
|
||||
def _write_trial_result(
|
||||
results_dir: Path,
|
||||
*,
|
||||
config: TrialConfig,
|
||||
index: int,
|
||||
payload: dict[str, Any],
|
||||
) -> None:
|
||||
results_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = trial_output_path(results_dir, model=config.model, index=index)
|
||||
path.write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8")
|
||||
|
||||
|
||||
def _write_trial_report(payload: dict[str, Any]) -> None:
|
||||
try:
|
||||
report_path = save_report_from_result_payload(payload)
|
||||
except ValueError as exc:
|
||||
payload["report_save_error"] = str(exc)
|
||||
return
|
||||
payload["report_path"] = report_path.as_posix()
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--model", default="opencode/mimo-v2.5-free")
|
||||
parser.add_argument("--variant", default="high")
|
||||
parser.add_argument("--trials", type=int, default=1)
|
||||
parser.add_argument("--timeout-seconds", type=int, default=1000)
|
||||
parser.add_argument(
|
||||
"--attach",
|
||||
dest="attach_url",
|
||||
default=None,
|
||||
help=(
|
||||
"Attach to a running opencode server URL. This is not a direct MCP "
|
||||
"server URL."
|
||||
),
|
||||
)
|
||||
parser.add_argument("--prompt", type=Path, default=DEFAULT_PROMPT)
|
||||
parser.add_argument("--results-dir", type=Path, default=DEFAULT_RESULTS_DIR)
|
||||
parser.add_argument("--workspaces-dir", type=Path, default=DEFAULT_WORKSPACES_DIR)
|
||||
parser.add_argument(
|
||||
"--workspace-template",
|
||||
type=Path,
|
||||
default=DEFAULT_WORKSPACE_TEMPLATE,
|
||||
help="Template directory copied for each local-mode trial workspace.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--source-root",
|
||||
type=Path,
|
||||
default=EXAMPLE_SOURCE_ROOT,
|
||||
help="Python source root written into each generated trial config.",
|
||||
)
|
||||
parser.add_argument("--server-url", default=None)
|
||||
parser.add_argument("--start-server", action="store_true", default=False)
|
||||
parser.add_argument("--no-start-server", action="store_false", dest="start_server")
|
||||
parser.add_argument("--server-port", type=int, default=DEFAULT_SERVER_PORT)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
if args.trials < 1:
|
||||
parser.error("--trials must be >= 1")
|
||||
|
||||
if args.server_url is not None:
|
||||
rpc_url = args.server_url
|
||||
managed_server: ManagedServer | None = None
|
||||
wf_command_prefix = f"uv run wf --url {rpc_url}"
|
||||
server_context = f"A workflow RPC server is available at `{rpc_url}`."
|
||||
elif args.start_server:
|
||||
managed_server = start_server(port=args.server_port)
|
||||
rpc_url = managed_server.rpc_url
|
||||
wf_command_prefix = f"uv run wf --url {rpc_url}"
|
||||
server_context = (
|
||||
f"The harness started a workflow RPC server at `{rpc_url}` for this trial."
|
||||
)
|
||||
else:
|
||||
managed_server = None
|
||||
wf_command_prefix = LOCAL_WF_COMMAND_PREFIX
|
||||
server_context = (
|
||||
"No external workflow RPC server is staged. The command prefix uses "
|
||||
"`--local`, which builds the configured workflow server in the CLI "
|
||||
"process for each command."
|
||||
)
|
||||
|
||||
try:
|
||||
use_trial_workspace = args.server_url is None and not args.start_server
|
||||
first_index = starting_trial_index(
|
||||
model=args.model,
|
||||
results_dir=args.results_dir,
|
||||
workspaces_dir=args.workspaces_dir,
|
||||
)
|
||||
summaries: list[dict[str, Any]] = []
|
||||
for index in range(first_index, first_index + args.trials):
|
||||
prompt_path = args.prompt
|
||||
trial_wf_command_prefix = wf_command_prefix
|
||||
trial_server_context = server_context
|
||||
if use_trial_workspace:
|
||||
workspace = prepare_trial_workspace(
|
||||
model=args.model,
|
||||
index=index,
|
||||
workspaces_dir=args.workspaces_dir,
|
||||
template_dir=args.workspace_template,
|
||||
source_root=args.source_root,
|
||||
)
|
||||
if args.prompt == DEFAULT_PROMPT:
|
||||
prompt_path = workspace.prompt_path
|
||||
trial_wf_command_prefix = wf_command_prefix_for_config(
|
||||
workspace.config_path
|
||||
)
|
||||
workspace_path = _display_path(workspace.root)
|
||||
config_path = _display_path(workspace.config_path)
|
||||
trial_server_context = (
|
||||
"No external workflow RPC server is staged. Use the "
|
||||
"per-trial workspace config copied to "
|
||||
f"`{config_path}`. Your writable trial workspace is "
|
||||
f"`{workspace_path}`."
|
||||
)
|
||||
config = TrialConfig(
|
||||
model=args.model,
|
||||
variant=args.variant,
|
||||
prompt_path=prompt_path,
|
||||
attach_url=args.attach_url,
|
||||
timeout_seconds=args.timeout_seconds,
|
||||
wf_command_prefix=trial_wf_command_prefix,
|
||||
server_context=trial_server_context,
|
||||
)
|
||||
result = run_trial(config, index=index, results_dir=args.results_dir)
|
||||
summaries.append(
|
||||
{
|
||||
"index": index,
|
||||
"classification": result["classification"],
|
||||
"returncode": result["returncode"],
|
||||
"duration_seconds": round(float(result["duration_seconds"]), 3),
|
||||
"report_path": _optional_string(result.get("report_path")),
|
||||
"report_save_error": result.get("report_save_error"),
|
||||
}
|
||||
)
|
||||
print(json.dumps(summaries[-1], sort_keys=True))
|
||||
|
||||
success_count = sum(
|
||||
1 for item in summaries if item["classification"] == "success"
|
||||
)
|
||||
print(
|
||||
json.dumps({"success_count": success_count, "trial_count": len(summaries)})
|
||||
)
|
||||
return 0
|
||||
finally:
|
||||
if managed_server is not None:
|
||||
stop_server(managed_server.process)
|
||||
|
||||
|
||||
def _optional_string(value: object) -> str | None:
|
||||
return None if value is None else str(value)
|
||||
def wf_command_prefix_for_config(config_path: Path) -> str:
|
||||
return _generic_wf_prefix(config_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# Support direct execution as `python examples/.../save_manual_audit.py`.
|
||||
ROOT = Path(__file__).resolve().parents[3]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from examples.agent_challenges.audit import ( # noqa: E402
|
||||
audit_from_result,
|
||||
main,
|
||||
save_manual_audit,
|
||||
)
|
||||
|
||||
__all__ = ["audit_from_result", "main", "save_manual_audit"]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -8,7 +8,7 @@ ROOT = Path(__file__).resolve().parents[3]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from examples.agent_challenges.browser_click_challenge.reports import ( # noqa: E402
|
||||
from examples.agent_challenges.reports import ( # noqa: E402
|
||||
main,
|
||||
report_from_result,
|
||||
save_report,
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
|
||||
def extract_challenge_report(text: str) -> dict[str, Any] | None:
|
||||
marker = "```yaml"
|
||||
start = text.lower().rfind(marker)
|
||||
if start == -1:
|
||||
return None
|
||||
body_start = text.find("\n", start)
|
||||
if body_start == -1:
|
||||
return None
|
||||
end = text.find("```", body_start + 1)
|
||||
if end == -1:
|
||||
return None
|
||||
raw_yaml = text[body_start + 1 : end]
|
||||
loaded = yaml.safe_load(raw_yaml)
|
||||
if not isinstance(loaded, dict):
|
||||
return None
|
||||
report = loaded.get("challenge_report")
|
||||
return report if isinstance(report, dict) else None
|
||||
|
||||
|
||||
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
|
||||
marker_index = text.find(marker)
|
||||
if marker_index == -1:
|
||||
return False
|
||||
return value in text[marker_index : marker_index + 80]
|
||||
@@ -0,0 +1,89 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from examples.agent_challenges.workspace import TrialConfig, render_prompt
|
||||
|
||||
|
||||
def build_opencode_command(config: TrialConfig) -> list[str]:
|
||||
prompt_text = render_prompt(
|
||||
config.prompt_path,
|
||||
wf_command_prefix=config.wf_command_prefix,
|
||||
server_context=config.server_context,
|
||||
)
|
||||
command = [
|
||||
"opencode",
|
||||
"run",
|
||||
]
|
||||
if config.attach_url is not None:
|
||||
command.extend(["--attach", config.attach_url])
|
||||
command.extend(
|
||||
[
|
||||
prompt_text,
|
||||
"--format",
|
||||
"json",
|
||||
"--model",
|
||||
config.model,
|
||||
"--variant",
|
||||
config.variant,
|
||||
]
|
||||
)
|
||||
return command
|
||||
|
||||
|
||||
def parse_opencode_output(stdout: str) -> dict[str, Any]:
|
||||
text = stdout.strip()
|
||||
if not text:
|
||||
raise ValueError("opencode produced no JSON output")
|
||||
|
||||
try:
|
||||
parsed = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
parsed = _parse_jsonl_tail(text)
|
||||
|
||||
if not isinstance(parsed, dict):
|
||||
raise ValueError("opencode output was not a JSON object")
|
||||
return parsed
|
||||
|
||||
|
||||
def _parse_jsonl_tail(text: str) -> dict[str, Any]:
|
||||
last_error: json.JSONDecodeError | None = None
|
||||
parsed_events: list[dict[str, Any]] = []
|
||||
for line in reversed(text.splitlines()):
|
||||
stripped = line.strip()
|
||||
if not stripped:
|
||||
continue
|
||||
try:
|
||||
parsed = json.loads(stripped)
|
||||
except json.JSONDecodeError as exc:
|
||||
last_error = exc
|
||||
continue
|
||||
if isinstance(parsed, dict):
|
||||
parsed_events.append(parsed)
|
||||
event_text = _event_text(parsed)
|
||||
if event_text is not None:
|
||||
return {"text": event_text, "event": parsed}
|
||||
if parsed_events:
|
||||
return parsed_events[0]
|
||||
if last_error is not None:
|
||||
raise last_error
|
||||
raise ValueError("opencode output did not contain JSON lines")
|
||||
|
||||
|
||||
def _event_text(event: dict[str, Any]) -> str | None:
|
||||
part = event.get("part")
|
||||
if isinstance(part, dict):
|
||||
text = part.get("text")
|
||||
if isinstance(text, str):
|
||||
return text
|
||||
text = event.get("text")
|
||||
return text if isinstance(text, str) else None
|
||||
|
||||
|
||||
def result_text(parsed: dict[str, Any]) -> str:
|
||||
for key in ("text", "message", "content", "output"):
|
||||
value = parsed.get(key)
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
return json.dumps(parsed, sort_keys=True)
|
||||
@@ -0,0 +1,118 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from examples.agent_challenges.classification import extract_challenge_report
|
||||
from examples.agent_challenges.opencode_io import parse_opencode_output, result_text
|
||||
|
||||
|
||||
def save_report(
|
||||
*,
|
||||
workspace: Path,
|
||||
report_text: str,
|
||||
output_name: str = "final-report.md",
|
||||
) -> Path:
|
||||
if not workspace.is_dir():
|
||||
raise ValueError(f"workspace does not exist or is not a directory: {workspace}")
|
||||
output_path = workspace / output_name
|
||||
output_path.write_text(report_text.rstrip() + "\n", encoding="utf-8")
|
||||
return output_path
|
||||
|
||||
|
||||
def report_from_result(result_path: Path) -> tuple[Path, str]:
|
||||
result = json.loads(result_path.read_text(encoding="utf-8"))
|
||||
if not isinstance(result, dict):
|
||||
raise ValueError("result file must contain a JSON object")
|
||||
report_text = _report_text_from_result(result)
|
||||
workspace = _workspace_from_result(result, report_text)
|
||||
return workspace, report_text
|
||||
|
||||
|
||||
def save_report_from_result_payload(
|
||||
result: dict[str, object],
|
||||
*,
|
||||
output_name: str = "final-report.md",
|
||||
) -> Path:
|
||||
report_text = _report_text_from_result(result)
|
||||
workspace = _workspace_from_result(result, report_text)
|
||||
return save_report(
|
||||
workspace=workspace,
|
||||
report_text=report_text,
|
||||
output_name=output_name,
|
||||
)
|
||||
|
||||
|
||||
def _report_text_from_result(result: dict[str, object]) -> str:
|
||||
parsed = result.get("parsed")
|
||||
if isinstance(parsed, dict):
|
||||
return result_text(parsed)
|
||||
|
||||
stdout = result.get("stdout")
|
||||
if isinstance(stdout, str) and stdout.strip():
|
||||
try:
|
||||
recovered = parse_opencode_output(stdout)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
"result file is missing parsed output and stdout has no report text"
|
||||
) from exc
|
||||
return result_text(recovered)
|
||||
|
||||
raise ValueError("result file is missing parsed output")
|
||||
|
||||
|
||||
def _workspace_from_result(result: dict[str, object], report_text: str) -> Path:
|
||||
config = result.get("config")
|
||||
if isinstance(config, dict):
|
||||
prompt_path = config.get("prompt_path")
|
||||
if isinstance(prompt_path, str) and prompt_path:
|
||||
return Path(prompt_path).parent
|
||||
|
||||
report = extract_challenge_report(report_text)
|
||||
if report is not None:
|
||||
workflow_file = report.get("workflow_file")
|
||||
if isinstance(workflow_file, str) and workflow_file:
|
||||
return Path(workflow_file).parent
|
||||
|
||||
raise ValueError("could not infer trial workspace from result")
|
||||
|
||||
|
||||
def _read_report_text(input_file: Path | None) -> str:
|
||||
if input_file is None:
|
||||
if sys.stdin.isatty():
|
||||
raise ValueError(
|
||||
"manual report mode needs piped stdin, --input-file, or --from-result"
|
||||
)
|
||||
return sys.stdin.read()
|
||||
return input_file.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("workspace", type=Path, nargs="?")
|
||||
parser.add_argument("--from-result", type=Path, default=None)
|
||||
parser.add_argument("--input-file", type=Path, default=None)
|
||||
parser.add_argument("--output-name", default="final-report.md")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
try:
|
||||
if args.from_result is not None:
|
||||
if args.workspace is not None or args.input_file is not None:
|
||||
parser.error("--from-result cannot be combined with workspace/input")
|
||||
workspace, report_text = report_from_result(args.from_result)
|
||||
else:
|
||||
if args.workspace is None:
|
||||
parser.error("workspace is required unless --from-result is used")
|
||||
workspace = args.workspace
|
||||
report_text = _read_report_text(args.input_file)
|
||||
output_path = save_report(
|
||||
workspace=workspace,
|
||||
report_text=report_text,
|
||||
output_name=args.output_name,
|
||||
)
|
||||
except ValueError as exc:
|
||||
parser.error(str(exc))
|
||||
print(output_path.as_posix())
|
||||
return 0
|
||||
@@ -0,0 +1,334 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from dataclasses import asdict, dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
from examples.agent_challenges.opencode_io import ( # noqa: E402
|
||||
build_opencode_command,
|
||||
parse_opencode_output,
|
||||
result_text,
|
||||
)
|
||||
from examples.agent_challenges.reports import ( # noqa: E402
|
||||
save_report_from_result_payload,
|
||||
)
|
||||
from examples.agent_challenges.workspace import ( # noqa: E402
|
||||
ChallengeDef,
|
||||
TrialConfig,
|
||||
_display_path,
|
||||
prepare_trial_workspace,
|
||||
rpc_url_for_port,
|
||||
server_command,
|
||||
starting_trial_index,
|
||||
trial_output_path,
|
||||
wf_command_prefix_for_config,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class ManagedServer:
|
||||
process: subprocess.Popen[str]
|
||||
rpc_url: str
|
||||
|
||||
|
||||
def start_server(
|
||||
defn: ChallengeDef,
|
||||
*,
|
||||
port: int,
|
||||
timeout_seconds: int = 30,
|
||||
) -> ManagedServer:
|
||||
command = server_command(port=port, config_arg=defn.server_config_arg)
|
||||
process = subprocess.Popen(
|
||||
command,
|
||||
cwd=ROOT,
|
||||
text=True,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
rpc_url = rpc_url_for_port(port)
|
||||
try:
|
||||
wait_for_status(rpc_url=rpc_url, timeout_seconds=timeout_seconds)
|
||||
except Exception:
|
||||
stop_server(process)
|
||||
raise
|
||||
return ManagedServer(process=process, rpc_url=rpc_url)
|
||||
|
||||
|
||||
def wait_for_status(*, rpc_url: str, timeout_seconds: int) -> None:
|
||||
deadline = time.monotonic() + timeout_seconds
|
||||
command = ["uv", "run", "wf", "--url", rpc_url, "status"]
|
||||
last_stderr = ""
|
||||
while time.monotonic() < deadline:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
cwd=ROOT,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=False,
|
||||
)
|
||||
if completed.returncode == 0:
|
||||
return
|
||||
last_stderr = completed.stderr
|
||||
time.sleep(0.5)
|
||||
raise RuntimeError(f"wf status did not become ready: {last_stderr}")
|
||||
|
||||
|
||||
def stop_server(process: subprocess.Popen[str]) -> None:
|
||||
if process.poll() is not None:
|
||||
return
|
||||
if sys.platform == "win32":
|
||||
subprocess.run(
|
||||
["taskkill", "/F", "/T", "/PID", str(process.pid)],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
)
|
||||
else:
|
||||
try:
|
||||
process.terminate()
|
||||
except OSError:
|
||||
return
|
||||
try:
|
||||
process.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
process.kill()
|
||||
process.wait(timeout=10)
|
||||
|
||||
|
||||
def run_trial(
|
||||
config: TrialConfig,
|
||||
*,
|
||||
index: int,
|
||||
results_dir: Path,
|
||||
classify_fn: Callable[[str], str],
|
||||
) -> dict[str, Any]:
|
||||
command = build_opencode_command(config)
|
||||
started = time.monotonic()
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
cwd=ROOT,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
timeout=config.timeout_seconds,
|
||||
check=False,
|
||||
)
|
||||
duration_seconds = time.monotonic() - started
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
payload = {
|
||||
"index": index,
|
||||
"config": _jsonable_config(config),
|
||||
"command": command,
|
||||
"classification": "timeout",
|
||||
"duration_seconds": config.timeout_seconds,
|
||||
"returncode": None,
|
||||
"stdout": exc.stdout or "",
|
||||
"stderr": exc.stderr or "",
|
||||
"parsed": None,
|
||||
}
|
||||
_write_trial_report(payload)
|
||||
_write_trial_result(results_dir, config=config, index=index, payload=payload)
|
||||
return payload
|
||||
|
||||
parsed: dict[str, Any] | None
|
||||
try:
|
||||
parsed = parse_opencode_output(completed.stdout)
|
||||
text = result_text(parsed)
|
||||
classification = classify_fn(text)
|
||||
except Exception:
|
||||
parsed = None
|
||||
classification = "parse_error"
|
||||
|
||||
payload = {
|
||||
"index": index,
|
||||
"config": _jsonable_config(config),
|
||||
"command": command,
|
||||
"classification": classification,
|
||||
"duration_seconds": duration_seconds,
|
||||
"returncode": completed.returncode,
|
||||
"stdout": completed.stdout,
|
||||
"stderr": completed.stderr,
|
||||
"parsed": parsed,
|
||||
}
|
||||
_write_trial_report(payload)
|
||||
_write_trial_result(results_dir, config=config, index=index, payload=payload)
|
||||
return payload
|
||||
|
||||
|
||||
def _jsonable_config(config: TrialConfig) -> dict[str, Any]:
|
||||
payload = asdict(config)
|
||||
payload["prompt_path"] = str(config.prompt_path)
|
||||
return payload
|
||||
|
||||
|
||||
def _write_trial_result(
|
||||
results_dir: Path,
|
||||
*,
|
||||
config: TrialConfig,
|
||||
index: int,
|
||||
payload: dict[str, Any],
|
||||
) -> None:
|
||||
results_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = trial_output_path(results_dir, model=config.model, index=index)
|
||||
path.write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8")
|
||||
|
||||
|
||||
def _write_trial_report(payload: dict[str, Any]) -> None:
|
||||
try:
|
||||
report_path = save_report_from_result_payload(payload)
|
||||
except ValueError as exc:
|
||||
payload["report_save_error"] = str(exc)
|
||||
return
|
||||
payload["report_path"] = report_path.as_posix()
|
||||
|
||||
|
||||
def main(
|
||||
defn: ChallengeDef,
|
||||
classify_fn: Callable[[str], str],
|
||||
argv: list[str] | None = None,
|
||||
) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--model", default="opencode/mimo-v2.5-free")
|
||||
parser.add_argument("--variant", default="high")
|
||||
parser.add_argument("--trials", type=int, default=1)
|
||||
parser.add_argument("--timeout-seconds", type=int, default=1000)
|
||||
parser.add_argument(
|
||||
"--attach",
|
||||
dest="attach_url",
|
||||
default=None,
|
||||
help=(
|
||||
"Attach to a running opencode server URL. This is not a direct MCP "
|
||||
"server URL."
|
||||
),
|
||||
)
|
||||
parser.add_argument("--prompt", type=Path, default=defn.default_prompt)
|
||||
parser.add_argument("--results-dir", type=Path, default=defn.default_results_dir)
|
||||
parser.add_argument(
|
||||
"--workspaces-dir", type=Path, default=defn.default_workspaces_dir
|
||||
)
|
||||
parser.add_argument(
|
||||
"--workspace-template",
|
||||
type=Path,
|
||||
default=defn.default_workspace_template,
|
||||
help="Template directory copied for each local-mode trial workspace.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--source-root",
|
||||
type=Path,
|
||||
default=defn.source_root,
|
||||
help="Python source root written into each generated trial config.",
|
||||
)
|
||||
parser.add_argument("--server-url", default=None)
|
||||
parser.add_argument("--start-server", action="store_true", default=False)
|
||||
parser.add_argument("--no-start-server", action="store_false", dest="start_server")
|
||||
parser.add_argument("--server-port", type=int, default=defn.default_server_port)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
if args.trials < 1:
|
||||
parser.error("--trials must be >= 1")
|
||||
|
||||
if args.server_url is not None:
|
||||
rpc_url = args.server_url
|
||||
managed_server: ManagedServer | None = None
|
||||
wf_command_prefix = f"uv run wf --url {rpc_url}"
|
||||
server_context = f"A workflow RPC server is available at `{rpc_url}`."
|
||||
elif args.start_server:
|
||||
managed_server = start_server(defn, port=args.server_port)
|
||||
rpc_url = managed_server.rpc_url
|
||||
wf_command_prefix = f"uv run wf --url {rpc_url}"
|
||||
server_context = (
|
||||
f"The harness started a workflow RPC server at `{rpc_url}` for this trial."
|
||||
)
|
||||
else:
|
||||
managed_server = None
|
||||
local_prefix = f"uv run wf --config {defn.server_config_arg} --local"
|
||||
wf_command_prefix = local_prefix
|
||||
server_context = (
|
||||
"No external workflow RPC server is staged. The command prefix uses "
|
||||
"`--local`, which builds the configured workflow server in the CLI "
|
||||
"process for each command."
|
||||
)
|
||||
|
||||
try:
|
||||
use_trial_workspace = args.server_url is None and not args.start_server
|
||||
first_index = starting_trial_index(
|
||||
model=args.model,
|
||||
results_dir=args.results_dir,
|
||||
workspaces_dir=args.workspaces_dir,
|
||||
)
|
||||
summaries: list[dict[str, Any]] = []
|
||||
for index in range(first_index, first_index + args.trials):
|
||||
prompt_path = args.prompt
|
||||
trial_wf_command_prefix = wf_command_prefix
|
||||
trial_server_context = server_context
|
||||
if use_trial_workspace:
|
||||
workspace = prepare_trial_workspace(
|
||||
defn,
|
||||
model=args.model,
|
||||
index=index,
|
||||
workspaces_dir=args.workspaces_dir,
|
||||
template_dir=args.workspace_template,
|
||||
source_root=args.source_root,
|
||||
)
|
||||
if args.prompt == defn.default_prompt:
|
||||
prompt_path = workspace.prompt_path
|
||||
trial_wf_command_prefix = wf_command_prefix_for_config(
|
||||
workspace.config_path
|
||||
)
|
||||
workspace_path = _display_path(workspace.root)
|
||||
config_path = _display_path(workspace.config_path)
|
||||
trial_server_context = (
|
||||
"No external workflow RPC server is staged. Use the "
|
||||
"per-trial workspace config copied to "
|
||||
f"`{config_path}`. Your writable trial workspace is "
|
||||
f"`{workspace_path}`."
|
||||
)
|
||||
config = TrialConfig(
|
||||
model=args.model,
|
||||
variant=args.variant,
|
||||
prompt_path=prompt_path,
|
||||
attach_url=args.attach_url,
|
||||
timeout_seconds=args.timeout_seconds,
|
||||
wf_command_prefix=trial_wf_command_prefix,
|
||||
server_context=trial_server_context,
|
||||
)
|
||||
result = run_trial(
|
||||
config,
|
||||
index=index,
|
||||
results_dir=args.results_dir,
|
||||
classify_fn=classify_fn,
|
||||
)
|
||||
summaries.append(
|
||||
{
|
||||
"index": index,
|
||||
"classification": result["classification"],
|
||||
"returncode": result["returncode"],
|
||||
"duration_seconds": round(float(result["duration_seconds"]), 3),
|
||||
"report_path": _optional_string(result.get("report_path")),
|
||||
"report_save_error": result.get("report_save_error"),
|
||||
}
|
||||
)
|
||||
print(json.dumps(summaries[-1], sort_keys=True))
|
||||
|
||||
success_count = sum(
|
||||
1 for item in summaries if item["classification"] == "success"
|
||||
)
|
||||
print(
|
||||
json.dumps({"success_count": success_count, "trial_count": len(summaries)})
|
||||
)
|
||||
return 0
|
||||
finally:
|
||||
if managed_server is not None:
|
||||
stop_server(managed_server.process)
|
||||
|
||||
|
||||
def _optional_string(value: object) -> str | None:
|
||||
return None if value is None else str(value)
|
||||
@@ -0,0 +1,191 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ChallengeDef:
|
||||
name: str
|
||||
source_root: Path
|
||||
source_id: str
|
||||
source_module: str
|
||||
source_registry: str
|
||||
store_root: str
|
||||
default_workspace_template: Path
|
||||
default_workspaces_dir: Path
|
||||
default_results_dir: Path
|
||||
default_prompt: Path
|
||||
default_server_port: int
|
||||
server_config_arg: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class TrialConfig:
|
||||
model: str
|
||||
variant: str
|
||||
prompt_path: Path
|
||||
attach_url: str | None
|
||||
timeout_seconds: int
|
||||
wf_command_prefix: str
|
||||
server_context: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class TrialWorkspace:
|
||||
root: Path
|
||||
config_path: Path
|
||||
prompt_path: Path
|
||||
|
||||
|
||||
def render_prompt(
|
||||
prompt_path: Path,
|
||||
*,
|
||||
wf_command_prefix: str,
|
||||
server_context: str,
|
||||
) -> str:
|
||||
return (
|
||||
prompt_path.read_text(encoding="utf-8")
|
||||
.replace("{{wf_command_prefix}}", wf_command_prefix)
|
||||
.replace("{{server_context}}", server_context)
|
||||
)
|
||||
|
||||
|
||||
def rpc_url_for_port(port: int) -> str:
|
||||
return f"http://127.0.0.1:{port}/rpc"
|
||||
|
||||
|
||||
def server_command(*, port: int, config_arg: str) -> list[str]:
|
||||
return [
|
||||
"uv",
|
||||
"run",
|
||||
"wf-rpc-server",
|
||||
"--config",
|
||||
config_arg,
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
str(port),
|
||||
]
|
||||
|
||||
|
||||
def _safe_model_name(model: str) -> str:
|
||||
return model.replace("/", "_").replace(":", "_")
|
||||
|
||||
|
||||
def prepare_trial_workspace(
|
||||
defn: ChallengeDef,
|
||||
*,
|
||||
model: str,
|
||||
index: int,
|
||||
workspaces_dir: Path | None = None,
|
||||
template_dir: Path | None = None,
|
||||
source_root: Path | None = None,
|
||||
) -> TrialWorkspace:
|
||||
if workspaces_dir is None:
|
||||
workspaces_dir = defn.default_workspaces_dir
|
||||
if template_dir is None:
|
||||
template_dir = defn.default_workspace_template
|
||||
effective_source_root = source_root if source_root is not None else defn.source_root
|
||||
root = workspaces_dir / f"{_safe_model_name(model)}-trial-{index:03d}"
|
||||
if root.exists():
|
||||
raise FileExistsError(f"trial workspace already exists: {root}")
|
||||
shutil.copytree(template_dir, root)
|
||||
workspace = TrialWorkspace(
|
||||
root=root,
|
||||
config_path=root / "wf.config.json",
|
||||
prompt_path=root / "prompt.md",
|
||||
)
|
||||
write_trial_config(
|
||||
workspace.config_path, defn=defn, source_root=effective_source_root
|
||||
)
|
||||
return workspace
|
||||
|
||||
|
||||
def write_trial_config(
|
||||
config_path: Path,
|
||||
*,
|
||||
defn: ChallengeDef,
|
||||
source_root: Path | None = None,
|
||||
) -> None:
|
||||
effective_source_root = source_root if source_root is not None else defn.source_root
|
||||
relative_source = Path(
|
||||
os.path.relpath(effective_source_root, config_path.parent)
|
||||
).as_posix()
|
||||
config = {
|
||||
"version": 1,
|
||||
"client": {"target": {"kind": "local"}},
|
||||
"server": {
|
||||
"store": {"kind": "filesystem", "root": defn.store_root},
|
||||
"sources": [
|
||||
{
|
||||
"kind": "python",
|
||||
"id": defn.source_id,
|
||||
"path": relative_source,
|
||||
"module": defn.source_module,
|
||||
"registry": defn.source_registry,
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
config_path.write_text(
|
||||
json.dumps(config, indent=2, sort_keys=True) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
|
||||
def wf_command_prefix_for_config(config_path: Path) -> str:
|
||||
path = config_path
|
||||
if not path.is_absolute():
|
||||
path = PROJECT_ROOT / path
|
||||
try:
|
||||
path_arg = path.resolve().relative_to(PROJECT_ROOT.resolve()).as_posix()
|
||||
except ValueError:
|
||||
path_arg = str(path.resolve())
|
||||
return f"uv run wf --config {path_arg} --local"
|
||||
|
||||
|
||||
def _trial_index_from_name(name: str, *, safe_model: str) -> int | None:
|
||||
prefix = f"{safe_model}-trial-"
|
||||
if not name.startswith(prefix):
|
||||
return None
|
||||
suffix = name.removeprefix(prefix)
|
||||
if "." in suffix:
|
||||
suffix = suffix.split(".", 1)[0]
|
||||
if not suffix.isdigit():
|
||||
return None
|
||||
return int(suffix)
|
||||
|
||||
|
||||
def starting_trial_index(
|
||||
*,
|
||||
model: str,
|
||||
results_dir: Path,
|
||||
workspaces_dir: Path,
|
||||
) -> int:
|
||||
safe_model = _safe_model_name(model)
|
||||
highest = 0
|
||||
for directory in (results_dir, workspaces_dir):
|
||||
if not directory.exists():
|
||||
continue
|
||||
for path in directory.iterdir():
|
||||
index = _trial_index_from_name(path.name, safe_model=safe_model)
|
||||
if index is not None:
|
||||
highest = max(highest, index)
|
||||
return highest + 1
|
||||
|
||||
|
||||
def trial_output_path(results_dir: Path, *, model: str, index: int) -> Path:
|
||||
return results_dir / f"{_safe_model_name(model)}-trial-{index:03d}.json"
|
||||
|
||||
|
||||
def _display_path(path: Path) -> str:
|
||||
try:
|
||||
return path.resolve().relative_to(PROJECT_ROOT.resolve()).as_posix()
|
||||
except ValueError:
|
||||
return str(path.resolve())
|
||||
@@ -0,0 +1,33 @@
|
||||
{
|
||||
"version": 1,
|
||||
"client": {
|
||||
"target": {
|
||||
"kind": "rpc_http",
|
||||
"url": "http://127.0.0.1:8771/rpc",
|
||||
"timeout_seconds": 30
|
||||
}
|
||||
},
|
||||
"server": {
|
||||
"store": {
|
||||
"kind": "filesystem",
|
||||
"root": ".wf_report_store"
|
||||
},
|
||||
"transports": [
|
||||
{
|
||||
"kind": "rpc_http",
|
||||
"host": "127.0.0.1",
|
||||
"port": 8771,
|
||||
"path": "/rpc"
|
||||
}
|
||||
],
|
||||
"sources": [
|
||||
{
|
||||
"kind": "python",
|
||||
"id": "local.report",
|
||||
"path": ".",
|
||||
"module": "ops",
|
||||
"registry": "registry"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user