feat: add debug profile for agent challenge UX mining
This commit is contained in:
@@ -248,11 +248,12 @@ stable.
|
|||||||
bundle for controlled challenge profiles, use `wf schema` for public shape
|
bundle for controlled challenge profiles, use `wf schema` for public shape
|
||||||
discovery, and avoid implementation/test-file guidance.
|
discovery, and avoid implementation/test-file guidance.
|
||||||
- Completed: the generic agent challenge harness now supports data-driven
|
- Completed: the generic agent challenge harness now supports data-driven
|
||||||
manifests, layered prompts, explicit `none|skills|all` profiles, one-hour hard
|
manifests, layered prompts, explicit `none|skills|all|debug` profiles,
|
||||||
ceilings, normalized OpenCode tool/token evidence, policy findings, and
|
one-hour hard ceilings, normalized OpenCode tool/token evidence, policy
|
||||||
manual-audited reports. Two data-driven challenges exist: browser-click and
|
findings, and manual-audited reports. Two data-driven challenges exist:
|
||||||
report-workflow. The central `run_trials.py` runner accepts any challenge
|
browser-click and report-workflow. The central `run_trials.py` runner accepts
|
||||||
manifest.
|
any challenge manifest. The `debug` profile is opt-in and captures
|
||||||
|
evidence-backed UX issue reports separately from normal benchmark scoring.
|
||||||
- Completed: report projections generate bounded Markdown and JSON reports for
|
- Completed: report projections generate bounded Markdown and JSON reports for
|
||||||
every V2 trial and regenerate both after audit without mutating raw evidence.
|
every V2 trial and regenerate both after audit without mutating raw evidence.
|
||||||
Implementation:
|
Implementation:
|
||||||
|
|||||||
@@ -107,10 +107,17 @@ Use profiles to separate product usability from instruction quality:
|
|||||||
| `none` | Base prompt plus challenge prompt only. | Tests discoverability with almost no agent instructions. |
|
| `none` | Base prompt plus challenge prompt only. | Tests discoverability with almost no agent instructions. |
|
||||||
| `skills` | Adds the workflow CLI skill bundle. | Tests the intended public agent instruction layer. |
|
| `skills` | Adds the workflow CLI skill bundle. | Tests the intended public agent instruction layer. |
|
||||||
| `all` | Allows broader docs/code exploration. | Tests whether the repository contains enough information to solve the task, but results are less clean. |
|
| `all` | Allows broader docs/code exploration. | Tests whether the repository contains enough information to solve the task, but results are less clean. |
|
||||||
|
| `debug` | Adds the workflow CLI skill bundle and asks the agent to keep detailed UX notes before escalating beyond public surfaces. | Mines command, docs, schema, and validation friction; keep separate from normal benchmark scoring. |
|
||||||
|
|
||||||
The same model/challenge should be run across profiles when comparing the value
|
The same model/challenge should be run across profiles when comparing the value
|
||||||
of the instruction layer.
|
of the instruction layer.
|
||||||
|
|
||||||
|
`debug` is opt-in and is not part of the default matrix. Use it when you want a
|
||||||
|
trial report to preserve failed commands, mistaken assumptions, confusing help
|
||||||
|
text, schema-shape problems, and any point where the agent became genuinely
|
||||||
|
blocked. A debug run may still complete the workflow, but its main value is the
|
||||||
|
`ux_issues_found` list in the self-report.
|
||||||
|
|
||||||
## Suggested Matrix
|
## Suggested Matrix
|
||||||
|
|
||||||
Start small and grow only after the harness output is stable:
|
Start small and grow only after the harness output is stable:
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ class InstructionProfile(StrEnum):
|
|||||||
NONE = "none"
|
NONE = "none"
|
||||||
SKILLS = "skills"
|
SKILLS = "skills"
|
||||||
ALL = "all"
|
ALL = "all"
|
||||||
|
DEBUG = "debug"
|
||||||
|
|
||||||
|
|
||||||
class SourceManifest(BaseModel):
|
class SourceManifest(BaseModel):
|
||||||
|
|||||||
@@ -207,7 +207,7 @@ def evaluate_policy(
|
|||||||
elif profile == InstructionProfile.SKILLS:
|
elif profile == InstructionProfile.SKILLS:
|
||||||
if category not in allowed_skills_categories:
|
if category not in allowed_skills_categories:
|
||||||
disallowed_reads.append(path_str)
|
disallowed_reads.append(path_str)
|
||||||
elif profile == InstructionProfile.ALL:
|
elif profile in (InstructionProfile.ALL, InstructionProfile.DEBUG):
|
||||||
if category in (
|
if category in (
|
||||||
"source",
|
"source",
|
||||||
"tests",
|
"tests",
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
## Instruction Profile: debug
|
||||||
|
|
||||||
|
Use the supplied skills under `.agent/skills/` plus public `wf` commands. This
|
||||||
|
profile is for product UX diagnosis, so keep working through public discovery
|
||||||
|
longer than you would in a normal benchmark run.
|
||||||
|
|
||||||
|
Do not read repository examples, tests, source, prior trials, or prior stores
|
||||||
|
unless you are genuinely blocked. "Genuinely blocked" means all of these are
|
||||||
|
true:
|
||||||
|
|
||||||
|
- You tried the relevant `wf --help` or subcommand `--help`.
|
||||||
|
- You tried `wf schema` or the specific `wf schema <name>` form when the issue
|
||||||
|
is about document shape.
|
||||||
|
- You tried validation or inspection commands such as `wf draft validate`,
|
||||||
|
`wf draft compile`, `wf deploy validate`, `wf cap inspect`, or bounded
|
||||||
|
`wf run trace` when those commands apply.
|
||||||
|
- You recorded the exact failing command, error text, and what you expected.
|
||||||
|
- You cannot proceed with public commands, supplied skills, challenge files, and
|
||||||
|
validation output alone.
|
||||||
|
|
||||||
|
If you escalate beyond public surfaces after being genuinely blocked, report it
|
||||||
|
honestly in the `read` flags and explain why. Never read or copy adjacent trial
|
||||||
|
answers, prior result files, prior stores, or complete ready-made solution plans
|
||||||
|
for the same challenge.
|
||||||
|
|
||||||
|
In the final `challenge_report`, include a `ux_issues_found` list. Use an empty
|
||||||
|
list if there were no issues. Each issue should be concrete and evidence-backed:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
ux_issues_found:
|
||||||
|
- command: "wf draft add-step ..."
|
||||||
|
issue: "Assumed --name existed; command requires --step."
|
||||||
|
evidence: "CLI returned: No such option --name"
|
||||||
|
workaround: "Used --step render"
|
||||||
|
suggested_fix: "Mention --step in help/example"
|
||||||
|
```
|
||||||
|
|
||||||
|
Log command typos, wrong assumptions, missing aliases, confusing help text,
|
||||||
|
schema/document-shape confusion, validation errors that were hard to repair,
|
||||||
|
and any point where you considered reading implementation code because public
|
||||||
|
surfaces were insufficient. A successful final run should still report the
|
||||||
|
issues and failed attempts that happened along the way.
|
||||||
@@ -125,6 +125,23 @@ def _summary_from_result(task: MatrixTask, result: dict[str, Any]) -> dict[str,
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _summary_from_exception(task: MatrixTask, exc: BaseException) -> dict[str, object]:
|
||||||
|
"""Return an auditable summary when a worker fails before writing a result."""
|
||||||
|
return {
|
||||||
|
"challenge": task.challenge.manifest.id,
|
||||||
|
"instruction_profile": task.profile.value,
|
||||||
|
"model": task.model,
|
||||||
|
"variant": task.variant,
|
||||||
|
"index": task.index,
|
||||||
|
"task_outcome": "runner_error",
|
||||||
|
"evaluation_validity": "unauditable",
|
||||||
|
"duration_seconds": 0.0,
|
||||||
|
"result_path": None,
|
||||||
|
"report_paths": None,
|
||||||
|
"error": {"type": type(exc).__name__, "message": str(exc)},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def main(argv: list[str] | None = None) -> int:
|
def main(argv: list[str] | None = None) -> int:
|
||||||
parser = argparse.ArgumentParser(description=__doc__)
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
@@ -141,7 +158,10 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
action="append",
|
action="append",
|
||||||
choices=[p.value for p in InstructionProfile],
|
choices=[p.value for p in InstructionProfile],
|
||||||
default=None,
|
default=None,
|
||||||
help="Instruction profile. May be repeated. Defaults to none, skills, all.",
|
help=(
|
||||||
|
"Instruction profile. May be repeated. Defaults to none, skills, "
|
||||||
|
"all; use debug explicitly for UX issue mining."
|
||||||
|
),
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--model",
|
"--model",
|
||||||
@@ -213,9 +233,13 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
|
|
||||||
summaries: list[dict[str, object]] = []
|
summaries: list[dict[str, object]] = []
|
||||||
with ThreadPoolExecutor(max_workers=min(args.concurrency, len(tasks))) as pool:
|
with ThreadPoolExecutor(max_workers=min(args.concurrency, len(tasks))) as pool:
|
||||||
futures = [pool.submit(_run_task, task) for task in tasks]
|
futures = {pool.submit(_run_task, task): task for task in tasks}
|
||||||
for future in as_completed(futures):
|
for future in as_completed(futures):
|
||||||
summary = future.result()
|
task = futures[future]
|
||||||
|
try:
|
||||||
|
summary = future.result()
|
||||||
|
except Exception as exc:
|
||||||
|
summary = _summary_from_exception(task, exc)
|
||||||
summaries.append(summary)
|
summaries.append(summary)
|
||||||
print(json.dumps(summary, sort_keys=True))
|
print(json.dumps(summary, sort_keys=True))
|
||||||
|
|
||||||
|
|||||||
@@ -620,6 +620,12 @@ def run_v2_trial(
|
|||||||
assertion_failures.append(
|
assertion_failures.append(
|
||||||
"could not extract challenge report for success_assertions evaluation"
|
"could not extract challenge report for success_assertions evaluation"
|
||||||
)
|
)
|
||||||
|
if profile == InstructionProfile.DEBUG and challenge_report is not None:
|
||||||
|
ux_issues = challenge_report.get("ux_issues_found")
|
||||||
|
if not isinstance(ux_issues, list):
|
||||||
|
assertion_failures.append(
|
||||||
|
"debug profile requires ux_issues_found list in challenge_report"
|
||||||
|
)
|
||||||
|
|
||||||
if assertion_failures:
|
if assertion_failures:
|
||||||
task_outcome = "failed"
|
task_outcome = "failed"
|
||||||
|
|||||||
@@ -283,7 +283,11 @@ def prepare_v2_trial_workspace(
|
|||||||
)
|
)
|
||||||
|
|
||||||
instruction_files: list[Path] = []
|
instruction_files: list[Path] = []
|
||||||
if profile in (InstructionProfile.SKILLS, InstructionProfile.ALL):
|
if profile in (
|
||||||
|
InstructionProfile.SKILLS,
|
||||||
|
InstructionProfile.ALL,
|
||||||
|
InstructionProfile.DEBUG,
|
||||||
|
):
|
||||||
bundle_entries = _load_instruction_bundle(instruction_bundle)
|
bundle_entries = _load_instruction_bundle(instruction_bundle)
|
||||||
for source_rel, destination_rel in bundle_entries:
|
for source_rel, destination_rel in bundle_entries:
|
||||||
source_file = PROJECT_ROOT / source_rel
|
source_file = PROJECT_ROOT / source_rel
|
||||||
|
|||||||
@@ -61,6 +61,7 @@ def test_instruction_profiles_are_exactly_the_supported_conditions() -> None:
|
|||||||
"none",
|
"none",
|
||||||
"skills",
|
"skills",
|
||||||
"all",
|
"all",
|
||||||
|
"debug",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
@@ -119,12 +120,14 @@ def test_challenge_prompt_is_identical_across_profiles(tmp_path: Path) -> None:
|
|||||||
assert {value.challenge_sha256 for value in rendered.values()} == {
|
assert {value.challenge_sha256 for value in rendered.values()} == {
|
||||||
rendered[InstructionProfile.NONE].challenge_sha256
|
rendered[InstructionProfile.NONE].challenge_sha256
|
||||||
}
|
}
|
||||||
assert len({value.rendered_sha256 for value in rendered.values()}) == 3
|
assert len({value.rendered_sha256 for value in rendered.values()}) == 4
|
||||||
assert "report the exact blocker" in rendered[InstructionProfile.NONE].text.replace(
|
assert "report the exact blocker" in rendered[InstructionProfile.NONE].text.replace(
|
||||||
"\n", " "
|
"\n", " "
|
||||||
)
|
)
|
||||||
assert ".agent/skills" in rendered[InstructionProfile.SKILLS].text
|
assert ".agent/skills" in rendered[InstructionProfile.SKILLS].text
|
||||||
assert "inspect broader repository" in rendered[InstructionProfile.ALL].text
|
assert "inspect broader repository" in rendered[InstructionProfile.ALL].text
|
||||||
|
assert "genuinely blocked" in rendered[InstructionProfile.DEBUG].text
|
||||||
|
assert "ux_issues_found" in rendered[InstructionProfile.DEBUG].text
|
||||||
|
|
||||||
|
|
||||||
def test_skills_profile_copies_bundle_but_none_does_not(tmp_path: Path) -> None:
|
def test_skills_profile_copies_bundle_but_none_does_not(tmp_path: Path) -> None:
|
||||||
@@ -537,6 +540,23 @@ def test_policy_evidence_classifies_reads(tmp_path: Path) -> None:
|
|||||||
assert all_policy.validity.value == "clean"
|
assert all_policy.validity.value == "clean"
|
||||||
assert all_policy.escalated_to_product_code is True
|
assert all_policy.escalated_to_product_code is True
|
||||||
|
|
||||||
|
debug_policy = evaluate_policy(
|
||||||
|
"debug",
|
||||||
|
[
|
||||||
|
workspace_read,
|
||||||
|
repository_index_read,
|
||||||
|
skills_read,
|
||||||
|
canonical_skills_read,
|
||||||
|
source_read,
|
||||||
|
test_read,
|
||||||
|
],
|
||||||
|
workspace_root=workspace_root,
|
||||||
|
repository_root=repository_root,
|
||||||
|
workspaces_root=workspaces_root,
|
||||||
|
)
|
||||||
|
assert debug_policy.validity.value == "clean"
|
||||||
|
assert debug_policy.escalated_to_product_code is True
|
||||||
|
|
||||||
bash_tc = ToolCallEvidence(
|
bash_tc = ToolCallEvidence(
|
||||||
ordinal=1,
|
ordinal=1,
|
||||||
call_id="c1",
|
call_id="c1",
|
||||||
@@ -1333,6 +1353,53 @@ def test_v2_runner_assertions_pass_on_matching_report(tmp_path: Path) -> None:
|
|||||||
assert result.get("challenge_report") is not None
|
assert result.get("challenge_report") is not None
|
||||||
|
|
||||||
|
|
||||||
|
def test_v2_runner_debug_profile_requires_ux_issues_found(
|
||||||
|
tmp_path: Path,
|
||||||
|
) -> None:
|
||||||
|
from examples.agent_challenges.runner import run_v2_trial
|
||||||
|
|
||||||
|
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
|
||||||
|
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
|
||||||
|
workspaces_dir = tmp_path / "workspaces"
|
||||||
|
results_dir = tmp_path / "results"
|
||||||
|
results_dir.mkdir()
|
||||||
|
|
||||||
|
report_yaml = (
|
||||||
|
"```yaml\nchallenge_report:\n value: expected\n run_failed: false\n```\n"
|
||||||
|
)
|
||||||
|
stdout_jsonl = json.dumps({"text": report_yaml})
|
||||||
|
|
||||||
|
def fake_run(
|
||||||
|
command: list[str],
|
||||||
|
*,
|
||||||
|
cwd: str,
|
||||||
|
text: bool,
|
||||||
|
capture_output: bool,
|
||||||
|
timeout: float | None,
|
||||||
|
check: bool,
|
||||||
|
) -> object:
|
||||||
|
return type(
|
||||||
|
"Result",
|
||||||
|
(),
|
||||||
|
{"returncode": 0, "stdout": stdout_jsonl, "stderr": ""},
|
||||||
|
)()
|
||||||
|
|
||||||
|
result = run_v2_trial(
|
||||||
|
challenge,
|
||||||
|
profile=InstructionProfile.DEBUG,
|
||||||
|
model="test-model",
|
||||||
|
variant="high",
|
||||||
|
index=1,
|
||||||
|
workspaces_dir=workspaces_dir,
|
||||||
|
results_dir=results_dir,
|
||||||
|
instruction_bundle=bundle,
|
||||||
|
run_fn=fake_run,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result["task_outcome"] == "failed"
|
||||||
|
assert any("ux_issues_found" in f for f in result["assertion_failures"])
|
||||||
|
|
||||||
|
|
||||||
def test_v2_runner_assertions_fail_on_mismatched_report(tmp_path: Path) -> None:
|
def test_v2_runner_assertions_fail_on_mismatched_report(tmp_path: Path) -> None:
|
||||||
from examples.agent_challenges.runner import run_v2_trial
|
from examples.agent_challenges.runner import run_v2_trial
|
||||||
|
|
||||||
@@ -1653,7 +1720,11 @@ def test_both_challenges_prepare_workspaces_under_each_profile(
|
|||||||
assert config["client"]["target"] == {"kind": "local"}
|
assert config["client"]["target"] == {"kind": "local"}
|
||||||
assert config["server"]["store"]["root"] == loaded.manifest.store_root
|
assert config["server"]["store"]["root"] == loaded.manifest.store_root
|
||||||
|
|
||||||
if profile in (InstructionProfile.SKILLS, InstructionProfile.ALL):
|
if profile in (
|
||||||
|
InstructionProfile.SKILLS,
|
||||||
|
InstructionProfile.ALL,
|
||||||
|
InstructionProfile.DEBUG,
|
||||||
|
):
|
||||||
assert (workspace.root / ".agent/skills/wf-cli/SKILL.md").is_file()
|
assert (workspace.root / ".agent/skills/wf-cli/SKILL.md").is_file()
|
||||||
else:
|
else:
|
||||||
assert not (workspace.root / ".agent").exists()
|
assert not (workspace.root / ".agent").exists()
|
||||||
|
|||||||
@@ -5,6 +5,7 @@ from pathlib import Path
|
|||||||
from examples.agent_challenges.manifests import load_challenge_manifest
|
from examples.agent_challenges.manifests import load_challenge_manifest
|
||||||
from examples.agent_challenges.models import InstructionProfile
|
from examples.agent_challenges.models import InstructionProfile
|
||||||
from examples.agent_challenges.run_matrix import (
|
from examples.agent_challenges.run_matrix import (
|
||||||
|
DEFAULT_PROFILES,
|
||||||
ModelProfile,
|
ModelProfile,
|
||||||
build_matrix_tasks,
|
build_matrix_tasks,
|
||||||
parse_model_profile,
|
parse_model_profile,
|
||||||
@@ -32,6 +33,14 @@ def test_parse_model_profile_rejects_empty_variant() -> None:
|
|||||||
parse_model_profile("opencode/deepseek-v4-flash-free=")
|
parse_model_profile("opencode/deepseek-v4-flash-free=")
|
||||||
|
|
||||||
|
|
||||||
|
def test_default_matrix_profiles_exclude_debug() -> None:
|
||||||
|
assert DEFAULT_PROFILES == (
|
||||||
|
InstructionProfile.NONE,
|
||||||
|
InstructionProfile.SKILLS,
|
||||||
|
InstructionProfile.ALL,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_matrix_tasks_allocate_indices_across_profiles(tmp_path: Path) -> None:
|
def test_matrix_tasks_allocate_indices_across_profiles(tmp_path: Path) -> None:
|
||||||
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
|
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
|
||||||
(challenge.root / "results").mkdir()
|
(challenge.root / "results").mkdir()
|
||||||
@@ -124,3 +133,39 @@ def test_run_trials_concurrency_invokes_all_indices(
|
|||||||
assert exit_code == 0
|
assert exit_code == 0
|
||||||
assert sorted(seen) == [1, 2, 3]
|
assert sorted(seen) == [1, 2, 3]
|
||||||
assert '"trial_count": 3' in capsys.readouterr().out
|
assert '"trial_count": 3' in capsys.readouterr().out
|
||||||
|
|
||||||
|
|
||||||
|
def test_run_matrix_reports_worker_exceptions(
|
||||||
|
monkeypatch,
|
||||||
|
tmp_path: Path,
|
||||||
|
capsys,
|
||||||
|
) -> None:
|
||||||
|
from examples.agent_challenges import run_matrix
|
||||||
|
|
||||||
|
manifest = _write_manifest(tmp_path / "challenge")
|
||||||
|
|
||||||
|
def fake_run_v2_trial(*args: object, **kwargs: object) -> dict[str, object]:
|
||||||
|
raise RuntimeError("lost result")
|
||||||
|
|
||||||
|
monkeypatch.setattr(run_matrix, "run_v2_trial", fake_run_v2_trial)
|
||||||
|
|
||||||
|
exit_code = run_matrix.main(
|
||||||
|
[
|
||||||
|
"--challenge",
|
||||||
|
str(manifest),
|
||||||
|
"--instruction-profile",
|
||||||
|
"debug",
|
||||||
|
"--model",
|
||||||
|
"opencode/test=max",
|
||||||
|
"--trials",
|
||||||
|
"1",
|
||||||
|
"--concurrency",
|
||||||
|
"1",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
output = capsys.readouterr().out
|
||||||
|
assert exit_code == 0
|
||||||
|
assert '"task_outcome": "runner_error"' in output
|
||||||
|
assert '"evaluation_validity": "unauditable"' in output
|
||||||
|
assert "lost result" in output
|
||||||
|
|||||||
Reference in New Issue
Block a user