fix: classify opencode text events and helper scripts
This commit is contained in:
@@ -55,6 +55,8 @@ workflow output.
|
|||||||
Each trial is classified as one of:
|
Each trial is classified as one of:
|
||||||
|
|
||||||
- `success`: output shows workflow usage and before/after clicked states.
|
- `success`: output shows workflow usage and before/after clicked states.
|
||||||
|
- `workflow_script`: output shows a workflow run, but the agent drove it through
|
||||||
|
a new helper script instead of the product-facing CLI/server path.
|
||||||
- `workflow_not_used`: output appears to solve the task without `wf`,
|
- `workflow_not_used`: output appears to solve the task without `wf`,
|
||||||
`wf-rpc-server`, deployment, or run evidence.
|
`wf-rpc-server`, deployment, or run evidence.
|
||||||
- `run_failed`: output includes workflow usage but reports a failure.
|
- `run_failed`: output includes workflow usage but reports a failure.
|
||||||
|
|||||||
@@ -12,6 +12,14 @@ Use this repository's workflow product path. That means you should use the
|
|||||||
the deployment through the workflow API. Do not solve the challenge with only a
|
the deployment through the workflow API. Do not solve the challenge with only a
|
||||||
standalone Playwright/Python script.
|
standalone Playwright/Python script.
|
||||||
|
|
||||||
|
Do not create a new helper script whose only job is to drive `WorkflowApi`
|
||||||
|
directly. The challenge is about whether the product-facing CLI/server workflow
|
||||||
|
can be discovered and used.
|
||||||
|
|
||||||
|
If you need to write a workflow definition, write a declarative JSON/YAML file
|
||||||
|
and then apply/run it through the product-facing workflow tools. Do not hide the
|
||||||
|
workflow construction inside a Python script.
|
||||||
|
|
||||||
The repository already includes a deterministic source example at:
|
The repository already includes a deterministic source example at:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ from typing import Any, Literal
|
|||||||
|
|
||||||
Classification = Literal[
|
Classification = Literal[
|
||||||
"success",
|
"success",
|
||||||
|
"workflow_script",
|
||||||
"workflow_not_used",
|
"workflow_not_used",
|
||||||
"run_failed",
|
"run_failed",
|
||||||
"timeout",
|
"timeout",
|
||||||
@@ -71,6 +72,7 @@ def parse_opencode_output(stdout: str) -> dict[str, Any]:
|
|||||||
|
|
||||||
def _parse_jsonl_tail(text: str) -> dict[str, Any]:
|
def _parse_jsonl_tail(text: str) -> dict[str, Any]:
|
||||||
last_error: json.JSONDecodeError | None = None
|
last_error: json.JSONDecodeError | None = None
|
||||||
|
parsed_events: list[dict[str, Any]] = []
|
||||||
for line in reversed(text.splitlines()):
|
for line in reversed(text.splitlines()):
|
||||||
stripped = line.strip()
|
stripped = line.strip()
|
||||||
if not stripped:
|
if not stripped:
|
||||||
@@ -81,22 +83,55 @@ def _parse_jsonl_tail(text: str) -> dict[str, Any]:
|
|||||||
last_error = exc
|
last_error = exc
|
||||||
continue
|
continue
|
||||||
if isinstance(parsed, dict):
|
if isinstance(parsed, dict):
|
||||||
return parsed
|
parsed_events.append(parsed)
|
||||||
|
event_text = _event_text(parsed)
|
||||||
|
if event_text is not None:
|
||||||
|
return {"text": event_text, "event": parsed}
|
||||||
|
if parsed_events:
|
||||||
|
return parsed_events[0]
|
||||||
if last_error is not None:
|
if last_error is not None:
|
||||||
raise last_error
|
raise last_error
|
||||||
raise ValueError("opencode output did not contain JSON lines")
|
raise ValueError("opencode output did not contain JSON lines")
|
||||||
|
|
||||||
|
|
||||||
|
def _event_text(event: dict[str, Any]) -> str | None:
|
||||||
|
"""Extract assistant text from opencode JSON events.
|
||||||
|
|
||||||
|
`opencode run --format json` emits many events. The final event can be a
|
||||||
|
`step_finish`; the answer text is usually the previous `text` event under
|
||||||
|
`part.text`.
|
||||||
|
"""
|
||||||
|
part = event.get("part")
|
||||||
|
if isinstance(part, dict):
|
||||||
|
text = part.get("text")
|
||||||
|
if isinstance(text, str):
|
||||||
|
return text
|
||||||
|
text = event.get("text")
|
||||||
|
return text if isinstance(text, str) else None
|
||||||
|
|
||||||
|
|
||||||
def classify_output(text: str) -> Classification:
|
def classify_output(text: str) -> Classification:
|
||||||
lowered = text.lower()
|
lowered = text.lower()
|
||||||
workflow_markers = [
|
product_command_markers = [
|
||||||
"wf ",
|
"wf ",
|
||||||
"wf-rpc-server",
|
"wf-rpc-server",
|
||||||
|
]
|
||||||
|
workflow_evidence_markers = [
|
||||||
"deployment",
|
"deployment",
|
||||||
"run id",
|
"run id",
|
||||||
"run_",
|
"run_",
|
||||||
]
|
]
|
||||||
used_workflow = any(marker in lowered for marker in workflow_markers)
|
used_product_command = any(
|
||||||
|
marker in lowered for marker in product_command_markers
|
||||||
|
)
|
||||||
|
has_workflow_evidence = any(
|
||||||
|
marker in lowered for marker in workflow_evidence_markers
|
||||||
|
)
|
||||||
|
used_helper_script = (
|
||||||
|
"uv run python" in lowered
|
||||||
|
or "python examples/" in lowered
|
||||||
|
or "run_workflow.py" in lowered
|
||||||
|
)
|
||||||
failed = any(
|
failed = any(
|
||||||
marker in lowered
|
marker in lowered
|
||||||
for marker in [
|
for marker in [
|
||||||
@@ -107,26 +142,33 @@ def classify_output(text: str) -> Classification:
|
|||||||
"validation failed",
|
"validation failed",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
before_false = (
|
before_false = _contains_bool_marker(lowered, "before.clicked", "false") or (
|
||||||
"before.clicked is false" in lowered
|
'"before"' in lowered and '"clicked": false' in lowered
|
||||||
or '"before"' in lowered
|
|
||||||
and '"clicked": false' in lowered
|
|
||||||
)
|
)
|
||||||
after_true = (
|
after_true = _contains_bool_marker(lowered, "after.clicked", "true") or (
|
||||||
"after.clicked is true" in lowered
|
'"after"' in lowered and '"clicked": true' in lowered
|
||||||
or '"after"' in lowered
|
|
||||||
and '"clicked": true' in lowered
|
|
||||||
)
|
)
|
||||||
|
|
||||||
if used_workflow and before_false and after_true and not failed:
|
if used_product_command and before_false and after_true and not failed:
|
||||||
return "success"
|
return "success"
|
||||||
if used_workflow and failed:
|
if has_workflow_evidence and used_helper_script and before_false and after_true:
|
||||||
|
return "workflow_script"
|
||||||
|
if (used_product_command or has_workflow_evidence) and failed:
|
||||||
return "run_failed"
|
return "run_failed"
|
||||||
if not used_workflow and (before_false or after_true or "playwright" in lowered):
|
if not has_workflow_evidence and (
|
||||||
|
before_false or after_true or "playwright" in lowered
|
||||||
|
):
|
||||||
return "workflow_not_used"
|
return "workflow_not_used"
|
||||||
return "unknown"
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def _contains_bool_marker(text: str, marker: str, value: str) -> bool:
|
||||||
|
marker_index = text.find(marker)
|
||||||
|
if marker_index == -1:
|
||||||
|
return False
|
||||||
|
return value in text[marker_index : marker_index + 80]
|
||||||
|
|
||||||
|
|
||||||
def trial_output_path(results_dir: Path, *, model: str, index: int) -> Path:
|
def trial_output_path(results_dir: Path, *, model: str, index: int) -> Path:
|
||||||
safe_model = model.replace("/", "_").replace(":", "_")
|
safe_model = model.replace("/", "_").replace(":", "_")
|
||||||
return results_dir / f"{safe_model}-trial-{index:03d}.json"
|
return results_dir / f"{safe_model}-trial-{index:03d}.json"
|
||||||
|
|||||||
@@ -72,6 +72,27 @@ def test_parse_opencode_output_reads_last_jsonl_object() -> None:
|
|||||||
assert parsed["text"] == "final"
|
assert parsed["text"] == "final"
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_opencode_output_prefers_text_event_before_step_finish() -> None:
|
||||||
|
payload = "\n".join(
|
||||||
|
[
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"type": "text",
|
||||||
|
"part": {
|
||||||
|
"type": "text",
|
||||||
|
"text": "deployment id: demo.default\nbefore.clicked is false",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
),
|
||||||
|
json.dumps({"type": "step_finish", "part": {"type": "step-finish"}}),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
parsed = parse_opencode_output(payload)
|
||||||
|
|
||||||
|
assert parsed["text"] == "deployment id: demo.default\nbefore.clicked is false"
|
||||||
|
|
||||||
|
|
||||||
def test_classify_output_success() -> None:
|
def test_classify_output_success() -> None:
|
||||||
result = classify_output(
|
result = classify_output(
|
||||||
"""
|
"""
|
||||||
@@ -87,6 +108,20 @@ def test_classify_output_success() -> None:
|
|||||||
assert result == "success"
|
assert result == "success"
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_output_workflow_script() -> None:
|
||||||
|
result = classify_output(
|
||||||
|
"""
|
||||||
|
uv run python examples/browser_click_workflow/run_workflow.py
|
||||||
|
Deployment id: browser_click_case_study.default
|
||||||
|
Run id: run_123
|
||||||
|
`before.clicked`: `False`
|
||||||
|
`after.clicked`: `True`
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result == "workflow_script"
|
||||||
|
|
||||||
|
|
||||||
def test_classify_output_workflow_not_used() -> None:
|
def test_classify_output_workflow_not_used() -> None:
|
||||||
result = classify_output(
|
result = classify_output(
|
||||||
"""
|
"""
|
||||||
|
|||||||
Reference in New Issue
Block a user