feat: add profiled agent challenge harness

This commit is contained in:
lda
2026-06-23 00:19:31 +07:00 Verified
parent 3c3cd0e241
commit 893872f821
19 changed files with 2120 additions and 0 deletions
+4
View File
@@ -215,6 +215,10 @@ stable.
- Completed: workflow/CLI agent instructions now form an explicit copyable - Completed: workflow/CLI agent instructions now form an explicit copyable
bundle for controlled challenge profiles, use `wf schema` for public shape bundle for controlled challenge profiles, use `wf schema` for public shape
discovery, and avoid implementation/test-file guidance. discovery, and avoid implementation/test-file guidance.
- Completed: the generic agent challenge harness now supports data-driven
manifests, layered prompts, explicit `none|skills|all` profiles, one-hour hard
ceilings, normalized OpenCode tool/token evidence, policy findings, and
manual-audited reports.
## Historical References ## Historical References
+47
View File
@@ -209,3 +209,50 @@ def main(argv: list[str] | None = None) -> int:
parser.error(str(exc)) parser.error(str(exc))
print(output_path.as_posix()) print(output_path.as_posix())
return 0 return 0
def manual_audit_from_v2_result(
result: dict[str, object],
*,
workspace: Path | None = None,
official_outcome: str = "pending",
auditor_notes: str = "",
audited_at: str | None = None,
auditor: str = "human",
output_name: str = "manual-audit.yaml",
) -> tuple[Path, dict[str, Any]]:
task_outcome = result.get("task_outcome", "unknown")
evaluation_validity = result.get("evaluation_validity", "unknown")
policy = result.get("policy", {})
if not isinstance(policy, dict):
policy = {}
audit: dict[str, Any] = {
"manual_audit": {
"auditor": auditor,
"audited_at": audited_at or _utc_now(),
"task_outcome": task_outcome,
"evaluation_validity": evaluation_validity,
"official_outcome": official_outcome,
"auditor_notes": auditor_notes,
"automatic_evidence": {
"task_outcome": task_outcome,
"evaluation_validity": evaluation_validity,
"policy": policy,
},
"duration_seconds": result.get("duration_seconds"),
"returncode": result.get("returncode"),
}
}
output_path: Path
if workspace is not None:
output_path = workspace / output_name
output_path.write_text(
yaml.safe_dump(audit, sort_keys=False, allow_unicode=True),
encoding="utf-8",
)
else:
output_path = Path(output_name)
return output_path, audit
+16
View File
@@ -0,0 +1,16 @@
# Workflow Agent Challenge
Use the repository's public `wf` product path to complete the challenge below.
Do not replace the workflow lifecycle with a helper script that imports internal
workflow APIs. Preserve exact commands, failures, run ids, and evidence in your
final answer.
Use this command prefix:
{{wf_command_prefix}}
{{server_context}}
Your writable trial workspace is `{{workspace_path}}`. Write attempt files only
inside it. End with the challenge's requested YAML self-report. The self-report
will be checked against observed tool calls and manually audited.
+42
View File
@@ -0,0 +1,42 @@
from __future__ import annotations
from pathlib import Path
import yaml
from .models import ChallengeManifest, LoadedChallenge
def _inside(root: Path, relative: str, *, field: str) -> Path:
candidate = (root / relative).resolve()
if not candidate.is_relative_to(root):
raise ValueError(f"challenge {field} must stay inside challenge directory")
return candidate
def load_challenge_manifest(path: Path) -> LoadedChallenge:
manifest_path = path.resolve()
root = manifest_path.parent
loaded = yaml.safe_load(manifest_path.read_text(encoding="utf-8"))
manifest = ChallengeManifest.model_validate(loaded)
prompt_path = _inside(root, manifest.prompt, field="prompt")
workspace_template = _inside(
root, manifest.workspace_template, field="workspace_template"
)
source_root = (root / manifest.source.root).resolve()
server_config = (root / manifest.server.config).resolve()
if not prompt_path.is_file():
raise ValueError(f"challenge prompt does not exist: {prompt_path}")
if not workspace_template.is_dir():
raise ValueError(
f"challenge workspace_template does not exist: {workspace_template}"
)
return LoadedChallenge(
manifest_path=manifest_path,
root=root,
prompt_path=prompt_path,
workspace_template=workspace_template,
source_root=source_root,
server_config=server_config,
manifest=manifest,
)
+207
View File
@@ -0,0 +1,207 @@
from __future__ import annotations
import dataclasses
import hashlib
import json
from dataclasses import dataclass
from typing import Any
@dataclass(frozen=True, slots=True)
class TokenMetrics:
total: int = 0
input: int = 0
output: int = 0
reasoning: int = 0
cache_read: int = 0
cache_write: int = 0
@dataclass(frozen=True, slots=True)
class ToolCallEvidence:
ordinal: int
call_id: str
tool: str
status: str
title: str
input: dict[str, Any]
metadata: dict[str, Any]
output_chars: int
output_preview: str
output_sha256: str
failed: bool
@dataclass(slots=True)
class TrialMetrics:
step_count: int = 0
tool_call_count: int = 0
failed_tool_call_count: int = 0
tool_counts: dict[str, int] = dataclasses.field(default_factory=dict)
tokens: TokenMetrics = dataclasses.field(default_factory=TokenMetrics)
cost: float = 0.0
unknown_event_count: int = 0
tool_calls: list[ToolCallEvidence] = dataclasses.field(default_factory=list)
def _int(value: object, *, default: int = 0) -> int:
if isinstance(value, (int, float)):
return int(value)
return default
def _float(value: object, *, default: float = 0.0) -> float:
if isinstance(value, (int, float)):
return float(value)
return default
def _str(value: object, *, default: str = "") -> str:
return value if isinstance(value, str) else default
def _dict(value: object) -> dict[str, Any]:
return value if isinstance(value, dict) else {}
def _add_tokens(current: TokenMetrics, new: TokenMetrics) -> TokenMetrics:
return TokenMetrics(
total=current.total + new.total,
input=current.input + new.input,
output=current.output + new.output,
reasoning=current.reasoning + new.reasoning,
cache_read=current.cache_read + new.cache_read,
cache_write=current.cache_write + new.cache_write,
)
def _normalize_tool_event(event: dict[str, Any]) -> dict[str, Any]:
"""Normalize both flat and nested OpenCode tool_use event formats.
Flat format (test fixtures):
{"type": "tool_use", "tool": "read", "status": "success",
"call_id": "c1", "input": {"path": "foo.py"}, ...}
Nested format (real OpenCode JSONL):
{"type": "tool_use", "part": {"tool": "read", "callID": "c1",
"state": {"status": "success", "input": {"path": "foo.py"}, ...}}}
"""
part = event.get("part")
if not isinstance(part, dict):
return event
tool_name = _str(part.get("tool"))
if not tool_name:
tool_name = _str(event.get("tool"), default="unknown")
state = _dict(part.get("state"))
status = _str(
state.get("status"), default=_str(event.get("status"), default="unknown")
)
title = _str(state.get("title"), default=_str(event.get("title")))
output_raw = _str(state.get("output"), default=_str(event.get("output")))
metadata = _dict(state.get("metadata"))
if not metadata:
metadata = _dict(event.get("metadata"))
call_id = _str(part.get("callID"))
if not call_id:
call_id = _str(event.get("call_id"))
tool_input = _dict(state.get("input"))
if not tool_input:
tool_input = _dict(event.get("input"))
merged = dict(event)
merged["tool"] = tool_name
merged["status"] = status
merged["title"] = title
merged["output"] = output_raw
merged["metadata"] = metadata
merged["call_id"] = call_id
merged["input"] = tool_input
return merged
def extract_trial_metrics(stdout: str, *, preview_chars: int = 500) -> TrialMetrics:
metrics = TrialMetrics()
ordinal = 0
for line in stdout.splitlines():
stripped = line.strip()
if not stripped:
continue
try:
event = json.loads(stripped)
except json.JSONDecodeError:
metrics.unknown_event_count += 1
continue
if not isinstance(event, dict):
metrics.unknown_event_count += 1
continue
event_type = _str(event.get("type"))
if event_type == "step_start":
metrics.step_count += 1
elif event_type == "tool_use":
ordinal += 1
normalized = _normalize_tool_event(event)
tool_name = _str(normalized.get("tool"), default="unknown")
status = _str(normalized.get("status"), default="unknown")
failed = status in ("error", "failed")
output_raw = _str(normalized.get("output"))
output_sha256 = hashlib.sha256(output_raw.encode("utf-8")).hexdigest()
output_chars = len(output_raw)
output_preview = output_raw[:preview_chars]
metrics.tool_call_count += 1
if failed:
metrics.failed_tool_call_count += 1
metrics.tool_counts[tool_name] = metrics.tool_counts.get(tool_name, 0) + 1
call_id = _str(normalized.get("call_id"), default=f"call-{ordinal}")
metrics.tool_calls.append(
ToolCallEvidence(
ordinal=ordinal,
call_id=call_id,
tool=tool_name,
status=status,
title=_str(normalized.get("title")),
input=_dict(normalized.get("input")),
metadata=_dict(normalized.get("metadata")),
output_chars=output_chars,
output_preview=output_preview,
output_sha256=output_sha256,
failed=failed,
)
)
elif event_type == "step_finish":
tokens = _dict(event.get("tokens"))
cache = _dict(tokens.get("cache"))
step_tokens = TokenMetrics(
total=_int(tokens.get("total")),
input=_int(tokens.get("input")),
output=_int(tokens.get("output")),
reasoning=_int(tokens.get("reasoning")),
cache_read=_int(cache.get("read")),
cache_write=_int(cache.get("write")),
)
metrics.tokens = _add_tokens(metrics.tokens, step_tokens)
metrics.cost += _float(event.get("cost"))
else:
metrics.unknown_event_count += 1
return metrics
def metrics_payload(metrics: TrialMetrics) -> dict[str, Any]:
payload = dataclasses.asdict(metrics)
payload["tool_counts"] = dict(sorted(payload["tool_counts"].items()))
return payload
+56
View File
@@ -0,0 +1,56 @@
from __future__ import annotations
from enum import StrEnum
from pathlib import Path
from typing import Any
from pydantic import BaseModel, ConfigDict, Field
class InstructionProfile(StrEnum):
NONE = "none"
SKILLS = "skills"
ALL = "all"
class SourceManifest(BaseModel):
model_config = ConfigDict(extra="forbid")
id: str = Field(min_length=1)
root: str = Field(min_length=1)
module: str = Field(min_length=1)
registry: str = Field(min_length=1)
class ServerManifest(BaseModel):
model_config = ConfigDict(extra="forbid")
config: str = Field(min_length=1)
default_port: int = Field(ge=1, le=65535)
class ReportManifest(BaseModel):
model_config = ConfigDict(extra="forbid")
required_fields: list[str] = Field(default_factory=list)
success_assertions: dict[str, Any] = Field(default_factory=dict)
class ChallengeManifest(BaseModel):
model_config = ConfigDict(extra="forbid")
version: int = Field(ge=1)
id: str = Field(pattern=r"^[a-z][a-z0-9_-]*$")
prompt: str
workspace_template: str
source: SourceManifest
store_root: str
server: ServerManifest
report: ReportManifest
class LoadedChallenge(BaseModel):
model_config = ConfigDict(arbitrary_types_allowed=True)
manifest_path: Path
root: Path
prompt_path: Path
workspace_template: Path
source_root: Path
server_config: Path
manifest: ChallengeManifest
+149
View File
@@ -0,0 +1,149 @@
from __future__ import annotations
from dataclasses import dataclass
from enum import StrEnum
from pathlib import Path
from .metrics import ToolCallEvidence
from .models import InstructionProfile
class EvaluationValidity(StrEnum):
CLEAN = "clean"
CONTAMINATED = "contaminated"
UNAUDITABLE = "unauditable"
@dataclass(frozen=True, slots=True)
class PolicyEvidence:
validity: EvaluationValidity
disallowed_reads: tuple[str, ...]
escalated_to_product_code: bool
opaque_shell_commands: tuple[str, ...]
reads_by_category: dict[str, tuple[str, ...]]
def _classify_path(
path_str: str,
*,
workspace_root: str,
repository_root: str,
workspaces_root: str,
) -> str:
try:
p = Path(path_str).resolve()
except OSError, ValueError:
return "unknown"
workspace = Path(workspace_root).resolve()
repository = Path(repository_root).resolve()
workspaces = Path(workspaces_root).resolve()
if p.is_relative_to(workspace):
rel = p.relative_to(workspace)
if rel.parts and rel.parts[0] == ".agent":
return "supplied_skills"
return "workspace"
if p.is_relative_to(workspaces):
return "adjacent_attempts"
if p.is_relative_to(repository):
rel = p.relative_to(repository)
parts = rel.parts
if parts and parts[0] == ".wf_store":
return "prior_store"
if parts and parts[0] == "tests":
return "tests"
if parts and parts[0] == "src":
return "source"
if parts and parts[0] == "docs":
return "docs"
if parts and parts[0] == "examples":
return "examples"
return "source"
return "outside"
def _extract_paths_from_tool_call(
tc: ToolCallEvidence,
) -> list[str]:
tool_name = tc.tool.lower()
if tool_name in ("read", "glob", "grep", "list", "search"):
path_val = (
tc.input.get("path") or tc.input.get("file") or tc.input.get("pattern")
)
if isinstance(path_val, str) and path_val:
return [path_val]
return []
def _extract_shell_command(tc: ToolCallEvidence) -> str | None:
if tc.tool.lower() in ("bash", "shell", "exec", "run"):
cmd = tc.input.get("command") or tc.input.get("cmd")
if isinstance(cmd, str) and cmd:
return cmd
return None
def evaluate_policy(
profile: InstructionProfile | str,
tool_calls: list[ToolCallEvidence],
*,
workspace_root: str | Path,
repository_root: str | Path,
workspaces_root: str | Path,
) -> PolicyEvidence:
if isinstance(profile, str):
profile = InstructionProfile(profile)
workspace_root_str = str(workspace_root)
repository_root_str = str(repository_root)
workspaces_root_str = str(workspaces_root)
disallowed_reads: list[str] = []
opaque_shell_commands: list[str] = []
reads_by_category: dict[str, list[str]] = {}
escalated_to_product_code = False
for tc in tool_calls:
paths = _extract_paths_from_tool_call(tc)
for path_str in paths:
category = _classify_path(
path_str,
workspace_root=workspace_root_str,
repository_root=repository_root_str,
workspaces_root=workspaces_root_str,
)
reads_by_category.setdefault(category, []).append(path_str)
if profile == InstructionProfile.NONE:
if category not in ("workspace", "unknown"):
disallowed_reads.append(path_str)
elif profile == InstructionProfile.SKILLS:
if category not in ("workspace", "supplied_skills", "unknown"):
disallowed_reads.append(path_str)
elif profile == InstructionProfile.ALL:
if category in ("source", "tests", "docs", "examples"):
escalated_to_product_code = True
shell_cmd = _extract_shell_command(tc)
if shell_cmd is not None:
opaque_shell_commands.append(shell_cmd)
if disallowed_reads:
validity = EvaluationValidity.CONTAMINATED
elif opaque_shell_commands and not disallowed_reads:
validity = EvaluationValidity.UNAUDITABLE
else:
validity = EvaluationValidity.CLEAN
frozen_reads_by_category = {k: tuple(v) for k, v in reads_by_category.items()}
return PolicyEvidence(
validity=validity,
disallowed_reads=tuple(disallowed_reads),
escalated_to_product_code=escalated_to_product_code,
opaque_shell_commands=tuple(opaque_shell_commands),
reads_by_category=frozen_reads_by_category,
)
@@ -0,0 +1,5 @@
## Instruction Profile: all
Start with the supplied skills and public docs. If genuinely blocked, you may
inspect broader repository docs, examples, tests, and source. Report what you
read; observed tool calls will also be retained for audit.
@@ -0,0 +1,7 @@
## Instruction Profile: none
Use challenge files, `wf --help`, `wf schema`, validation, inspect, and bounded
trace commands. Do not read repository skills, docs, examples, tests, source,
prior trials, or prior stores. If public surfaces are insufficient, report the
exact blocker and finish the task as failed rather than reverse-engineering the
implementation.
@@ -0,0 +1,6 @@
## Instruction Profile: skills
Use the supplied skills under `.agent/skills/` plus public `wf` commands. Do not
read repository examples, tests, source, prior trials, or prior stores. If the
skills and public surfaces are insufficient, report the exact blocker rather
than reverse-engineering implementation code.
+72
View File
@@ -0,0 +1,72 @@
from __future__ import annotations
import hashlib
import re
from dataclasses import dataclass
from pathlib import Path
from .models import InstructionProfile, LoadedChallenge
_PROMPT_DIR = Path(__file__).resolve().parent
_BASE_PROMPT = _PROMPT_DIR / "base-prompt.md"
_PROFILE_DIR = _PROMPT_DIR / "profile-prompts"
_PLACEHOLDER_RE = re.compile(r"\{\{(\w+)\}\}")
def _sha256(text: str) -> str:
return hashlib.sha256(text.encode("utf-8")).hexdigest()
@dataclass(frozen=True, slots=True)
class RenderedPrompt:
text: str
base_sha256: str
profile_sha256: str
challenge_sha256: str
rendered_sha256: str
def _load_profile_fragment(profile: InstructionProfile) -> str:
path = _PROFILE_DIR / f"{profile.value}.md"
return path.read_text(encoding="utf-8")
def _resolve_placeholders(template: str, *, variables: dict[str, str]) -> str:
def _replace(match: re.Match[str]) -> str:
key = match.group(1)
if key not in variables:
raise ValueError(f"unresolved placeholder {{{{{key}}}}}")
return variables[key]
return _PLACEHOLDER_RE.sub(_replace, template)
def compose_trial_prompt(
challenge: LoadedChallenge,
*,
profile: InstructionProfile,
wf_command_prefix: str,
server_context: str,
workspace_path: Path,
) -> RenderedPrompt:
base_text = _BASE_PROMPT.read_text(encoding="utf-8")
profile_text = _load_profile_fragment(profile)
challenge_text = challenge.prompt_path.read_text(encoding="utf-8")
rendered = _resolve_placeholders(
base_text,
variables={
"wf_command_prefix": wf_command_prefix,
"server_context": server_context,
"workspace_path": str(workspace_path),
},
)
return RenderedPrompt(
text=f"{rendered}\n{profile_text}\n{challenge_text}",
base_sha256=_sha256(base_text),
profile_sha256=_sha256(profile_text),
challenge_sha256=_sha256(challenge_text),
rendered_sha256=_sha256(f"{rendered}\n{profile_text}\n{challenge_text}"),
)
+82
View File
@@ -116,3 +116,85 @@ def main(argv: list[str] | None = None) -> int:
parser.error(str(exc)) parser.error(str(exc))
print(output_path.as_posix()) print(output_path.as_posix())
return 0 return 0
def _format_tokens(tokens: dict[str, object]) -> str:
parts = []
for key in ("total", "input", "output", "reasoning", "cache_read", "cache_write"):
val = tokens.get(key)
if val is not None:
parts.append(f"{key}: {val}")
return ", ".join(parts)
def report_from_v2_result(result: dict[str, object]) -> str:
lines: list[str] = []
profile = result.get("instruction_profile", "unknown")
lines.append(f"Instruction profile: {profile}")
lines.append("")
task_outcome = result.get("task_outcome", "unknown")
evaluation_validity = result.get("evaluation_validity", "unknown")
lines.append(f"Task outcome: {task_outcome}")
lines.append(f"Evaluation validity: {evaluation_validity}")
lines.append("")
duration = result.get("duration_seconds", 0)
lines.append(f"Duration: {duration}s")
lines.append("")
metrics = result.get("metrics", {})
if isinstance(metrics, dict):
tokens = metrics.get("tokens", {})
if isinstance(tokens, dict):
lines.append("Observed token metrics:")
lines.append(f" {_format_tokens(tokens)}")
cost = metrics.get("cost")
if cost is not None:
lines.append(f" cost: {cost}")
tool_counts = metrics.get("tool_counts", {})
if isinstance(tool_counts, dict) and tool_counts:
lines.append("")
lines.append("Tool calls by tool:")
for tool, count in sorted(tool_counts.items()):
lines.append(f" {tool}: {count}")
tool_calls = metrics.get("tool_calls", [])
if isinstance(tool_calls, list) and tool_calls:
lines.append("")
lines.append("Tool call details:")
for tc in tool_calls:
if isinstance(tc, dict):
tool = tc.get("tool", "unknown")
status = tc.get("status", "unknown")
preview = tc.get("output_preview", "")
lines.append(f" [{tc.get('ordinal', '?')}] {tool} ({status})")
if preview:
lines.append(f" preview: {preview[:200]}")
lines.append("")
policy = result.get("policy", {})
if isinstance(policy, dict):
disallowed = policy.get("disallowed_reads", [])
if disallowed:
lines.append("Disallowed reads:")
for path in disallowed:
lines.append(f" - {path}")
lines.append("")
lines.append("Agent self-report discrepancies:")
lines.append(" (pending manual audit)")
lines.append("")
parsed = result.get("parsed")
if isinstance(parsed, dict):
text = parsed.get("text", "")
if text:
lines.append("Final agent answer:")
lines.append(text)
lines.append("")
lines.append("Manual audit: pending")
lines.append("")
return "\n".join(lines)
+114
View File
@@ -0,0 +1,114 @@
"""Central CLI for running profiled agent challenge trials."""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
try:
from .manifests import load_challenge_manifest
from .models import InstructionProfile
from .runner import run_v2_trial
from .workspace import starting_trial_index
except ImportError:
_project_root = Path(__file__).resolve().parents[2]
if str(_project_root) not in sys.path:
sys.path.insert(0, str(_project_root))
from examples.agent_challenges.manifests import load_challenge_manifest
from examples.agent_challenges.models import InstructionProfile
from examples.agent_challenges.runner import run_v2_trial
from examples.agent_challenges.workspace import starting_trial_index
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(
description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter,
)
parser.add_argument(
"--challenge",
type=Path,
required=True,
help="Path to challenge.yaml",
)
parser.add_argument(
"--instruction-profile",
type=str,
choices=[p.value for p in InstructionProfile],
required=True,
help="Instruction profile for this trial",
)
parser.add_argument("--model", default="opencode/mimo-v2.5-free")
parser.add_argument("--variant", default="high")
parser.add_argument("--trials", type=int, default=1)
parser.add_argument("--timeout-seconds", type=int, default=3600)
parser.add_argument(
"--attach",
dest="attach_url",
default=None,
help="Attach to a running opencode server URL.",
)
parser.add_argument("--results-dir", type=Path, default=None)
parser.add_argument("--workspaces-dir", type=Path, default=None)
parser.add_argument(
"--instruction-bundle",
type=Path,
default=Path(__file__).resolve().parents[2]
/ "examples"
/ "agent_challenges"
/ "instruction_bundles"
/ "workflow_cli.yaml",
)
parser.add_argument("--server-url", default=None)
parser.add_argument("--start-server", action="store_true", default=False)
parser.add_argument("--server-port", type=int, default=8772)
args = parser.parse_args(argv)
if args.trials < 1:
parser.error("--trials must be >= 1")
challenge = load_challenge_manifest(args.challenge)
profile = InstructionProfile(args.instruction_profile)
results_dir = args.results_dir or challenge.root / "results"
workspaces_dir = args.workspaces_dir or challenge.root / "workspaces"
first_index = starting_trial_index(
model=args.model,
results_dir=results_dir,
workspaces_dir=workspaces_dir,
)
summaries: list[dict[str, object]] = []
for index in range(first_index, first_index + args.trials):
result = run_v2_trial(
challenge,
profile=profile,
model=args.model,
variant=args.variant,
index=index,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=args.instruction_bundle,
timeout_seconds=args.timeout_seconds,
attach_url=args.attach_url,
)
summaries.append(
{
"index": index,
"task_outcome": result["task_outcome"],
"evaluation_validity": result["evaluation_validity"],
"duration_seconds": result["duration_seconds"],
}
)
print(json.dumps(summaries[-1], sort_keys=True))
success_count = sum(1 for s in summaries if s["task_outcome"] == "success")
print(json.dumps({"success_count": success_count, "trial_count": len(summaries)}))
return 0
if __name__ == "__main__":
sys.exit(main())
+277
View File
@@ -10,6 +10,8 @@ from dataclasses import asdict, dataclass
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
import yaml
ROOT = Path(__file__).resolve().parents[2] ROOT = Path(__file__).resolve().parents[2]
from examples.agent_challenges.opencode_io import ( # noqa: E402 from examples.agent_challenges.opencode_io import ( # noqa: E402
@@ -339,3 +341,278 @@ def main(
def _optional_string(value: object) -> str | None: def _optional_string(value: object) -> str | None:
return None if value is None else str(value) return None if value is None else str(value)
def _get_git_commit() -> str:
try:
completed = subprocess.run(
["git", "rev-parse", "HEAD"],
cwd=ROOT,
text=True,
capture_output=True,
check=False,
)
if completed.returncode == 0:
return completed.stdout.strip()
except Exception:
pass
return "unknown"
def _get_git_dirty() -> bool:
try:
completed = subprocess.run(
["git", "status", "--porcelain"],
cwd=ROOT,
text=True,
capture_output=True,
check=False,
)
if completed.returncode == 0:
return bool(completed.stdout.strip())
except Exception:
pass
return False
def run_v2_trial(
challenge: object,
*,
profile: object,
model: str,
variant: str,
index: int,
workspaces_dir: Path,
results_dir: Path,
instruction_bundle: Path,
timeout_seconds: int = 3600,
attach_url: str | None = None,
run_fn: Any = None,
) -> dict[str, Any]:
from .metrics import extract_trial_metrics, metrics_payload
from .models import InstructionProfile, LoadedChallenge
from .policy import evaluate_policy
from .prompts import compose_trial_prompt
from .workspace import (
_display_path,
prepare_v2_trial_workspace,
wf_command_prefix_for_config,
)
if run_fn is None:
run_fn = subprocess.run
if not isinstance(challenge, LoadedChallenge):
raise TypeError("challenge must be a LoadedChallenge")
if not isinstance(profile, InstructionProfile):
profile = InstructionProfile(profile)
workspace = prepare_v2_trial_workspace(
challenge,
profile=profile,
model=model,
index=index,
workspaces_dir=workspaces_dir,
instruction_bundle=instruction_bundle,
)
wf_command_prefix = wf_command_prefix_for_config(workspace.config_path)
workspace_path = _display_path(workspace.root)
config_path_display = _display_path(workspace.config_path)
server_context = (
"No external workflow RPC server is staged. Use the "
"per-trial workspace config copied to "
f"`{config_path_display}`. Your writable trial workspace is "
f"`{workspace_path}`."
)
rendered = compose_trial_prompt(
challenge,
profile=profile,
wf_command_prefix=wf_command_prefix,
server_context=server_context,
workspace_path=workspace.root,
)
workspace.rendered_prompt_path.write_text(rendered.text, encoding="utf-8")
command = [
"opencode",
"run",
]
if attach_url is not None:
command.extend(["--attach", attach_url])
command.extend(
[
rendered.text,
"--format",
"json",
"--model",
model,
"--variant",
variant,
]
)
started = time.monotonic()
stdout = ""
stderr = ""
returncode = 0
task_outcome = "success"
parse_error: dict[str, str] | None = None
try:
completed = run_fn(
command,
cwd=str(workspace.root),
text=True,
capture_output=True,
timeout=timeout_seconds,
check=False,
)
duration_seconds = time.monotonic() - started
stdout = completed.stdout
stderr = completed.stderr
returncode = completed.returncode
if returncode != 0:
task_outcome = "failed"
except subprocess.TimeoutExpired as exc:
duration_seconds = time.monotonic() - started
stdout = (
exc.stdout
if isinstance(exc.stdout, str)
else (exc.stdout or b"").decode("utf-8", errors="replace")
if exc.stdout
else ""
)
stderr = (
exc.stderr
if isinstance(exc.stderr, str)
else (exc.stderr or b"").decode("utf-8", errors="replace")
if exc.stderr
else ""
)
task_outcome = "timeout"
except Exception as exc:
duration_seconds = time.monotonic() - started
task_outcome = "parse_error"
parse_error = {
"type": type(exc).__name__,
"message": str(exc),
}
metrics = extract_trial_metrics(stdout)
metrics_dir = workspace.root
metrics_dir.mkdir(parents=True, exist_ok=True)
(metrics_dir / "metrics.json").write_text(
json.dumps(metrics_payload(metrics), indent=2, sort_keys=True),
encoding="utf-8",
)
repository_root = ROOT
policy = evaluate_policy(
profile,
metrics.tool_calls,
workspace_root=workspace.root,
repository_root=repository_root,
workspaces_root=workspaces_dir,
)
success_assertions = challenge.manifest.report.success_assertions
required_fields = challenge.manifest.report.required_fields
assertion_failures: list[str] = []
challenge_report: dict[str, Any] | None = None
parsed_output: dict[str, Any] | None = None
report_parse_error: dict[str, str] | None = None
if stdout.strip():
try:
parsed_output = parse_opencode_output(stdout)
report_text = result_text(parsed_output)
from examples.agent_challenges.classification import (
extract_challenge_report,
)
challenge_report = extract_challenge_report(report_text)
except (ValueError, KeyError, yaml.YAMLError) as exc:
challenge_report = None
report_parse_error = {
"type": type(exc).__name__,
"message": str(exc),
}
if task_outcome == "success":
if required_fields and challenge_report is None:
assertion_failures.append(
"could not extract challenge report for required_fields evaluation"
)
elif required_fields and challenge_report is not None:
for field in required_fields:
if field not in challenge_report:
assertion_failures.append(f"required field missing: {field}")
if success_assertions and challenge_report is not None:
for field, expected in success_assertions.items():
actual = challenge_report.get(field)
if actual != expected:
assertion_failures.append(
f"{field}: expected {expected!r}, got {actual!r}"
)
elif success_assertions and challenge_report is None:
if not assertion_failures:
assertion_failures.append(
"could not extract challenge report for success_assertions evaluation"
)
if assertion_failures:
task_outcome = "failed"
result: dict[str, Any] = {
"instruction_profile": profile.value,
"task_outcome": task_outcome,
"evaluation_validity": policy.validity.value,
"prompt_hashes": {
"base": rendered.base_sha256,
"profile": rendered.profile_sha256,
"challenge": rendered.challenge_sha256,
"rendered": rendered.rendered_sha256,
},
"metrics": metrics_payload(metrics),
"policy": {
"validity": policy.validity.value,
"disallowed_reads": list(policy.disallowed_reads),
"escalated_to_product_code": policy.escalated_to_product_code,
"opaque_shell_commands": list(policy.opaque_shell_commands),
},
"repository_commit": _get_git_commit(),
"repository_dirty": _get_git_dirty(),
"harness_version": "v2",
"index": index,
"model": model,
"variant": variant,
"duration_seconds": round(duration_seconds, 3),
"returncode": returncode,
"stdout": stdout,
"stderr": stderr,
"parsed": parsed_output,
}
if assertion_failures:
result["assertion_failures"] = assertion_failures
if parse_error is not None:
result["parse_error"] = parse_error
if report_parse_error is not None:
result["report_parse_error"] = report_parse_error
if challenge_report is not None:
result["challenge_report"] = challenge_report
results_dir.mkdir(parents=True, exist_ok=True)
result_path = (
results_dir
/ f"{model.replace('/', '_').replace(':', '_')}-trial-{index:03d}.json"
)
result_path.write_text(
json.dumps(result, indent=2, sort_keys=True), encoding="utf-8"
)
return result
@@ -0,0 +1,22 @@
"""Central CLI for saving agent challenge manual audits."""
from __future__ import annotations
import sys
from pathlib import Path
try:
from .audit import main as audit_main
except ImportError:
_project_root = Path(__file__).resolve().parents[2]
if str(_project_root) not in sys.path:
sys.path.insert(0, str(_project_root))
from examples.agent_challenges.audit import main as audit_main
def main(argv: list[str] | None = None) -> int:
return audit_main(argv)
if __name__ == "__main__":
sys.exit(main())
@@ -0,0 +1,22 @@
"""Central CLI for saving agent challenge trial reports."""
from __future__ import annotations
import sys
from pathlib import Path
try:
from .reports import main as reports_main
except ImportError:
_project_root = Path(__file__).resolve().parents[2]
if str(_project_root) not in sys.path:
sys.path.insert(0, str(_project_root))
from examples.agent_challenges.reports import main as reports_main
def main(argv: list[str] | None = None) -> int:
return reports_main(argv)
if __name__ == "__main__":
sys.exit(main())
+94
View File
@@ -189,3 +189,97 @@ def _display_path(path: Path) -> str:
return path.resolve().relative_to(PROJECT_ROOT.resolve()).as_posix() return path.resolve().relative_to(PROJECT_ROOT.resolve()).as_posix()
except ValueError: except ValueError:
return str(path.resolve()) return str(path.resolve())
@dataclass(frozen=True, slots=True)
class V2TrialWorkspace:
root: Path
config_path: Path
rendered_prompt_path: Path
instruction_files: tuple[Path, ...]
def _load_instruction_bundle(
bundle_path: Path,
) -> list[tuple[str, str]]:
import yaml
loaded = yaml.safe_load(bundle_path.read_text(encoding="utf-8"))
if not isinstance(loaded, dict) or not isinstance(loaded.get("files"), list):
raise ValueError(f"invalid instruction bundle: {bundle_path}")
entries: list[tuple[str, str]] = []
for entry in loaded["files"]:
if not isinstance(entry, dict):
raise ValueError(f"invalid bundle entry: {entry}")
source = entry.get("source")
destination = entry.get("destination")
if not isinstance(source, str) or not isinstance(destination, str):
raise ValueError(f"bundle entry missing source/destination: {entry}")
entries.append((source, destination))
return entries
def prepare_v2_trial_workspace(
challenge: object,
*,
profile: object,
model: str,
index: int,
workspaces_dir: Path,
instruction_bundle: Path,
) -> V2TrialWorkspace:
from .models import InstructionProfile, LoadedChallenge
if not isinstance(challenge, LoadedChallenge):
raise TypeError("challenge must be a LoadedChallenge")
if not isinstance(profile, InstructionProfile):
raise TypeError("profile must be an InstructionProfile")
root = workspaces_dir / f"{_safe_model_name(model)}-trial-{index:03d}"
if root.exists():
raise FileExistsError(f"trial workspace already exists: {root}")
shutil.copytree(challenge.workspace_template, root)
config_path = root / "wf.config.json"
relative_source = Path(
os.path.relpath(challenge.source_root, config_path.parent)
).as_posix()
config = {
"version": 1,
"client": {"target": {"kind": "local"}},
"server": {
"store": {"kind": "filesystem", "root": challenge.manifest.store_root},
"sources": [
{
"kind": "python",
"id": challenge.manifest.source.id,
"path": relative_source,
"module": challenge.manifest.source.module,
"registry": challenge.manifest.source.registry,
}
],
},
}
config_path.write_text(
json.dumps(config, indent=2, sort_keys=True) + "\n",
encoding="utf-8",
)
instruction_files: list[Path] = []
if profile in (InstructionProfile.SKILLS, InstructionProfile.ALL):
bundle_entries = _load_instruction_bundle(instruction_bundle)
for source_rel, destination_rel in bundle_entries:
source_file = PROJECT_ROOT / source_rel
dest_file = root / ".agent" / "skills" / destination_rel
dest_file.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(source_file, dest_file)
instruction_files.append(dest_file)
rendered_prompt_path = root / "rendered-prompt.md"
return V2TrialWorkspace(
root=root,
config_path=config_path,
rendered_prompt_path=rendered_prompt_path,
instruction_files=tuple(instruction_files),
)
@@ -0,0 +1,898 @@
from __future__ import annotations
import json
from pathlib import Path
import pytest
from examples.agent_challenges.manifests import load_challenge_manifest
from examples.agent_challenges.models import InstructionProfile
from examples.agent_challenges.prompts import compose_trial_prompt
from examples.agent_challenges.workspace import prepare_v2_trial_workspace
def _write_manifest(root: Path) -> Path:
(root / "workspace_template").mkdir(parents=True)
(root / "challenge-prompt.md").write_text("Build it.\n", encoding="utf-8")
path = root / "challenge.yaml"
path.write_text(
"""\
version: 1
id: fixture
prompt: challenge-prompt.md
workspace_template: workspace_template
source:
id: local.fixture
root: source
module: ops
registry: registry
store_root: .wf_fixture_store
server:
config: wf.config.json
default_port: 8779
report:
required_fields: [value, run_failed]
success_assertions:
value: expected
run_failed: false
""",
encoding="utf-8",
)
return path
def test_load_challenge_manifest_resolves_paths(tmp_path: Path) -> None:
manifest_path = _write_manifest(tmp_path)
loaded = load_challenge_manifest(manifest_path)
assert loaded.manifest.id == "fixture"
assert loaded.root == tmp_path.resolve()
assert loaded.prompt_path == (tmp_path / "challenge-prompt.md").resolve()
assert loaded.workspace_template == (tmp_path / "workspace_template").resolve()
assert loaded.manifest.report.success_assertions == {
"value": "expected",
"run_failed": False,
}
def test_instruction_profiles_are_exactly_the_supported_conditions() -> None:
assert [profile.value for profile in InstructionProfile] == [
"none",
"skills",
"all",
]
def test_invalid_manifest_rejects_parent_traversal(tmp_path: Path) -> None:
path = _write_manifest(tmp_path)
text = path.read_text(encoding="utf-8").replace(
"workspace_template: workspace_template",
"workspace_template: ../outside",
)
path.write_text(text, encoding="utf-8")
with pytest.raises(ValueError, match="workspace_template"):
load_challenge_manifest(path)
ROOT = Path(__file__).resolve().parents[2]
def test_challenge_prompt_is_identical_across_profiles(tmp_path: Path) -> None:
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
rendered = {
profile: compose_trial_prompt(
challenge,
profile=profile,
wf_command_prefix="uv run wf --config wf.config.json --local",
server_context="Local mode.",
workspace_path=tmp_path / profile.value,
)
for profile in InstructionProfile
}
assert {value.challenge_sha256 for value in rendered.values()} == {
rendered[InstructionProfile.NONE].challenge_sha256
}
assert len({value.rendered_sha256 for value in rendered.values()}) == 3
assert "report the exact blocker" in rendered[InstructionProfile.NONE].text.replace(
"\n", " "
)
assert ".agent/skills" in rendered[InstructionProfile.SKILLS].text
assert "inspect broader repository" in rendered[InstructionProfile.ALL].text
def test_skills_profile_copies_bundle_but_none_does_not(tmp_path: Path) -> None:
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
none_workspace = prepare_v2_trial_workspace(
challenge,
profile=InstructionProfile.NONE,
model="model",
index=1,
workspaces_dir=tmp_path / "workspaces",
instruction_bundle=bundle,
)
skills_workspace = prepare_v2_trial_workspace(
challenge,
profile=InstructionProfile.SKILLS,
model="model",
index=2,
workspaces_dir=tmp_path / "workspaces",
instruction_bundle=bundle,
)
assert not (none_workspace.root / ".agent/skills").exists()
assert (skills_workspace.root / ".agent/skills/wf-cli/SKILL.md").is_file()
assert skills_workspace.instruction_files
def test_extract_trial_metrics_parses_jsonl_events() -> None:
from examples.agent_challenges.metrics import extract_trial_metrics
stdout = "\n".join(
[
json.dumps({"type": "step_start", "step": 1}),
json.dumps(
{
"type": "tool_use",
"tool": "read",
"status": "success",
"title": "Read file",
"input": {"path": "foo.py"},
"metadata": {},
"output": "x" * 4000,
}
),
json.dumps(
{
"type": "tool_use",
"tool": "bash",
"status": "error",
"title": "Run command",
"input": {"command": "ls"},
"metadata": {},
"output": "error occurred",
}
),
json.dumps(
{
"type": "step_finish",
"tokens": {
"total": 120,
"input": 20,
"output": 30,
"reasoning": 10,
"cache": {"read": 60, "write": 0},
},
"cost": 0.01,
}
),
]
)
metrics = extract_trial_metrics(stdout)
assert metrics.step_count == 1
assert metrics.tool_call_count == 2
assert metrics.failed_tool_call_count == 1
assert metrics.tool_counts == {"bash": 1, "read": 1}
assert metrics.tokens.total == 120
assert metrics.tokens.input == 20
assert metrics.tokens.output == 30
assert metrics.tokens.reasoning == 10
assert metrics.tokens.cache_read == 60
assert metrics.cost == 0.01
assert metrics.tool_calls[0].tool == "read"
assert metrics.tool_calls[0].output_chars == 4000
assert len(metrics.tool_calls[0].output_preview) <= 500
def test_extract_trial_metrics_sums_tokens_across_steps() -> None:
from examples.agent_challenges.metrics import extract_trial_metrics
stdout = "\n".join(
[
json.dumps({"type": "step_start", "step": 1}),
json.dumps(
{
"type": "step_finish",
"tokens": {
"total": 50,
"input": 20,
"output": 15,
"reasoning": 5,
"cache": {"read": 10, "write": 0},
},
"cost": 0.003,
}
),
json.dumps({"type": "step_start", "step": 2}),
json.dumps(
{
"type": "step_finish",
"tokens": {
"total": 70,
"input": 30,
"output": 25,
"reasoning": 10,
"cache": {"read": 5, "write": 0},
},
"cost": 0.004,
}
),
]
)
metrics = extract_trial_metrics(stdout)
assert metrics.step_count == 2
assert metrics.tokens.total == 120
assert metrics.tokens.input == 50
assert metrics.tokens.output == 40
assert metrics.tokens.reasoning == 15
assert metrics.tokens.cache_read == 15
assert metrics.cost == 0.007
def test_extract_trial_metrics_handles_nested_part_state_format() -> None:
from examples.agent_challenges.metrics import extract_trial_metrics
stdout = "\n".join(
[
json.dumps({"type": "step_start", "step": 1}),
json.dumps(
{
"type": "tool_use",
"part": {
"tool": "read",
"callID": "call-abc",
"state": {
"status": "success",
"title": "Read file",
"input": {"path": "src/app.py"},
"output": "file content here",
"metadata": {"size": 100},
},
},
}
),
json.dumps(
{
"type": "tool_use",
"part": {
"tool": "bash",
"callID": "call-def",
"state": {
"status": "error",
"title": "Run command",
"input": {"command": "ls nonexistent"},
"output": "command failed",
},
},
}
),
json.dumps(
{
"type": "step_finish",
"tokens": {
"total": 100,
"input": 50,
"output": 30,
"reasoning": 10,
"cache": {"read": 10, "write": 0},
},
"cost": 0.005,
}
),
]
)
metrics = extract_trial_metrics(stdout)
assert metrics.step_count == 1
assert metrics.tool_call_count == 2
assert metrics.failed_tool_call_count == 1
assert metrics.tool_counts == {"bash": 1, "read": 1}
assert metrics.tool_calls[0].tool == "read"
assert metrics.tool_calls[0].status == "success"
assert metrics.tool_calls[0].call_id == "call-abc"
assert metrics.tool_calls[0].output_chars == 17
assert metrics.tool_calls[0].input == {"path": "src/app.py"}
assert metrics.tool_calls[1].tool == "bash"
assert metrics.tool_calls[1].status == "error"
assert metrics.tool_calls[1].failed is True
assert metrics.tool_calls[1].input == {"command": "ls nonexistent"}
assert metrics.tokens.total == 100
assert metrics.cost == 0.005
def test_policy_evidence_classifies_reads(tmp_path: Path) -> None:
from examples.agent_challenges.metrics import ToolCallEvidence
from examples.agent_challenges.policy import evaluate_policy
workspace_root = tmp_path / "workspace"
workspace_root.mkdir()
repository_root = tmp_path / "repo"
repository_root.mkdir()
workspaces_root = tmp_path / "workspaces"
workspaces_root.mkdir()
def _tc(tool: str, path: str) -> ToolCallEvidence:
return ToolCallEvidence(
ordinal=1,
call_id="c1",
tool=tool,
status="success",
title="read",
input={"path": path},
metadata={},
output_chars=100,
output_preview="",
output_sha256="abc",
failed=False,
)
source_read = _tc("read", str(repository_root / "src" / "app.py"))
skills_read = _tc(
"read",
str(workspace_root / ".agent" / "skills" / "wf-cli" / "SKILL.md"),
)
workspace_read = _tc("read", str(workspace_root / "attempt.md"))
test_read = _tc("read", str(repository_root / "tests" / "test_app.py"))
none_policy = evaluate_policy(
"none",
[workspace_read, source_read],
workspace_root=workspace_root,
repository_root=repository_root,
workspaces_root=workspaces_root,
)
assert none_policy.validity.value == "contaminated"
assert any("app.py" in p for p in none_policy.disallowed_reads)
skills_policy = evaluate_policy(
"skills",
[workspace_read, skills_read, source_read, test_read],
workspace_root=workspace_root,
repository_root=repository_root,
workspaces_root=workspaces_root,
)
assert skills_policy.validity.value == "contaminated"
assert not skills_policy.escalated_to_product_code
all_policy = evaluate_policy(
"all",
[workspace_read, skills_read, source_read, test_read],
workspace_root=workspace_root,
repository_root=repository_root,
workspaces_root=workspaces_root,
)
assert all_policy.validity.value == "clean"
assert all_policy.escalated_to_product_code is True
bash_tc = ToolCallEvidence(
ordinal=1,
call_id="c1",
tool="bash",
status="success",
title="run",
input={"command": "cat /etc/passwd"},
metadata={},
output_chars=100,
output_preview="",
output_sha256="def",
failed=False,
)
bash_policy = evaluate_policy(
"none",
[bash_tc],
workspace_root=workspace_root,
repository_root=repository_root,
workspaces_root=workspaces_root,
)
assert bash_policy.validity.value == "unauditable"
assert len(bash_policy.opaque_shell_commands) == 1
def test_v2_runner_default_timeout_and_workspace_cwd(tmp_path: Path) -> None:
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
captured: dict[str, object] = {}
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
captured["cwd"] = cwd
captured["timeout"] = timeout
captured["command"] = command
return type(
"Result",
(),
{
"returncode": 0,
"stdout": json.dumps(
{
"type": "step_finish",
"tokens": {"total": 10, "input": 5, "output": 5},
"cost": 0.001,
}
),
"stderr": "",
},
)()
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert captured["timeout"] == 3600
assert isinstance(captured["cwd"], str)
assert result["instruction_profile"] == "none"
assert "prompt_hashes" in result
assert "metrics" in result
assert "policy" in result
assert "repository_commit" in result
def test_v2_runner_timeout_preserves_partial_evidence(tmp_path: Path) -> None:
import subprocess as sp
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
raise sp.TimeoutExpired(cmd=command, timeout=3600)
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["task_outcome"] == "timeout"
assert "metrics" in result
def test_v2_report_contains_expected_sections(tmp_path: Path) -> None:
from examples.agent_challenges.reports import report_from_v2_result
result_payload = {
"instruction_profile": "skills",
"task_outcome": "success",
"evaluation_validity": "contaminated",
"prompt_hashes": {
"base": "abc123",
"profile": "def456",
"challenge": "ghi789",
"rendered": "jkl012",
},
"metrics": {
"step_count": 1,
"tool_call_count": 2,
"failed_tool_call_count": 0,
"tool_counts": {"bash": 1, "read": 1},
"tokens": {
"total": 100,
"input": 50,
"output": 30,
"reasoning": 10,
"cache_read": 10,
"cache_write": 0,
},
"cost": 0.005,
"unknown_event_count": 0,
"tool_calls": [
{
"ordinal": 1,
"call_id": "c1",
"tool": "read",
"status": "success",
"title": "Read file",
"input": {"path": "foo.py"},
"metadata": {},
"output_chars": 100,
"output_preview": "file content...",
"output_sha256": "abc",
"failed": False,
},
],
},
"policy": {
"validity": "contaminated",
"disallowed_reads": ["src/app.py"],
"escalated_to_product_code": False,
"opaque_shell_commands": [],
},
"repository_commit": "abc123",
"repository_dirty": False,
"harness_version": "v2",
"index": 1,
"model": "test-model",
"variant": "high",
"duration_seconds": 10.5,
"returncode": 0,
"stdout": "test stdout",
"stderr": "",
"parsed": {"text": "Agent answer here"},
}
report_text = report_from_v2_result(result_payload)
assert "Instruction profile: skills" in report_text
assert "Task outcome: success" in report_text
assert "Evaluation validity: contaminated" in report_text
assert "Duration" in report_text
assert "Observed token metrics" in report_text
assert "Tool calls by tool" in report_text
assert "Disallowed reads" in report_text
assert "Agent self-report discrepancies" in report_text
assert "Final agent answer" in report_text
assert "Manual audit: pending" in report_text
assert "file content..." in report_text
assert "src/app.py" in report_text
def test_v2_manual_audit_includes_automatic_evidence(tmp_path: Path) -> None:
from examples.agent_challenges.audit import manual_audit_from_v2_result
result_payload = {
"instruction_profile": "none",
"task_outcome": "success",
"evaluation_validity": "contaminated",
"policy": {
"validity": "contaminated",
"disallowed_reads": ["src/app.py"],
"escalated_to_product_code": False,
"opaque_shell_commands": [],
},
"metrics": {"tokens": {"total": 100}},
}
workspace, audit = manual_audit_from_v2_result(
result_payload,
official_outcome="pass",
auditor_notes="Reviewed and passed.",
)
assert audit["manual_audit"]["task_outcome"] == "success"
assert audit["manual_audit"]["evaluation_validity"] == "contaminated"
assert audit["manual_audit"]["official_outcome"] == "pass"
assert audit["manual_audit"]["auditor_notes"] == "Reviewed and passed."
assert "automatic_evidence" in audit["manual_audit"]
def test_v2_runner_assertions_pass_on_matching_report(tmp_path: Path) -> None:
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
report_yaml = (
"```yaml\nchallenge_report:\n value: expected\n run_failed: false\n```\n"
)
stdout_jsonl = json.dumps({"text": report_yaml})
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
return type(
"Result",
(),
{"returncode": 0, "stdout": stdout_jsonl, "stderr": ""},
)()
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["task_outcome"] == "success"
assert "assertion_failures" not in result
assert result.get("challenge_report") is not None
def test_v2_runner_assertions_fail_on_mismatched_report(tmp_path: Path) -> None:
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
report_yaml = (
"```yaml\nchallenge_report:\n value: wrong_value\n run_failed: true\n```\n"
)
stdout_jsonl = json.dumps({"text": report_yaml})
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
return type(
"Result",
(),
{"returncode": 0, "stdout": stdout_jsonl, "stderr": ""},
)()
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["task_outcome"] == "failed"
assert "assertion_failures" in result
assert len(result["assertion_failures"]) == 2
assert any("value" in f for f in result["assertion_failures"])
assert any("run_failed" in f for f in result["assertion_failures"])
def test_v2_runner_required_fields_missing(tmp_path: Path) -> None:
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
report_yaml = "```yaml\nchallenge_report:\n other_field: true\n```\n"
stdout_jsonl = json.dumps({"text": report_yaml})
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
return type(
"Result",
(),
{"returncode": 0, "stdout": stdout_jsonl, "stderr": ""},
)()
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["task_outcome"] == "failed"
assert "assertion_failures" in result
assert any("required field missing" in f for f in result["assertion_failures"])
def test_v2_runner_preserves_evidence_on_parse_failure(tmp_path: Path) -> None:
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
raise RuntimeError("subprocess exploded")
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["task_outcome"] == "parse_error"
assert "parse_error" in result
assert result["parse_error"]["type"] == "RuntimeError"
assert "subprocess exploded" in result["parse_error"]["message"]
assert result["duration_seconds"] >= 0
def test_v2_runner_to_report_shows_final_answer(tmp_path: Path) -> None:
from examples.agent_challenges.reports import report_from_v2_result
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
agent_answer = "The deployment succeeded with id dep_123."
challenge_report_yaml = (
"```yaml\nchallenge_report:\n value: expected\n run_failed: false\n```\n"
)
stdout_jsonl = "\n".join(
[
json.dumps({"type": "step_start", "step": 1}),
json.dumps({"type": "step_finish", "tokens": {"total": 50}, "cost": 0.001}),
json.dumps(
{"text": f"Final answer: {agent_answer}\n\n{challenge_report_yaml}"}
),
]
)
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
return type(
"Result",
(),
{"returncode": 0, "stdout": stdout_jsonl, "stderr": ""},
)()
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["parsed"] is not None
assert result["task_outcome"] == "success"
report_text = report_from_v2_result(result)
assert "Final agent answer" in report_text
assert agent_answer in report_text
def test_v2_runner_preserves_report_parse_error_on_malformed_yaml(
tmp_path: Path,
) -> None:
from examples.agent_challenges.runner import run_v2_trial
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
bundle = ROOT / "examples/agent_challenges/instruction_bundles/workflow_cli.yaml"
workspaces_dir = tmp_path / "workspaces"
results_dir = tmp_path / "results"
results_dir.mkdir()
malformed_yaml = (
"```yaml\nchallenge_report:\n value: expected\n run_failed: [unclosed\n```\n"
)
stdout_jsonl = "\n".join(
[
json.dumps({"type": "step_start", "step": 1}),
json.dumps({"type": "step_finish", "tokens": {"total": 50}, "cost": 0.001}),
json.dumps({"text": f"Some output.\n\n{malformed_yaml}"}),
]
)
def fake_run(
command: list[str],
*,
cwd: str,
text: bool,
capture_output: bool,
timeout: float | None,
check: bool,
) -> object:
return type(
"Result",
(),
{"returncode": 0, "stdout": stdout_jsonl, "stderr": ""},
)()
result = run_v2_trial(
challenge,
profile=InstructionProfile.NONE,
model="test-model",
variant="high",
index=1,
workspaces_dir=workspaces_dir,
results_dir=results_dir,
instruction_bundle=bundle,
run_fn=fake_run,
)
assert result["parsed"] is not None
assert result.get("report_parse_error") is not None
assert result["report_parse_error"]["type"] in (
"ParserError",
"ScannerError",
"YAMLError",
)
assert result.get("challenge_report") is None