fix: address challenge harness review findings
This commit is contained in:
@@ -1035,14 +1035,16 @@ The tested thesis path imports the complete three-node plan as an immutable
|
|||||||
artifact:
|
artifact:
|
||||||
|
|
||||||
```powershell
|
```powershell
|
||||||
wf artifact create-from-plan workflow.plan.json --artifact report_case_study --version 1 --title "Report Case Study" --outcome ok --binding local.report=local.report
|
wf artifact create-from-plan workflow.plan.json --artifact report_case_study --version 1 --title "Report Case Study" --outcome ok
|
||||||
```
|
```
|
||||||
|
|
||||||
Artifact creation captures the workflow graph, required capability snapshots,
|
Artifact creation captures the workflow graph, required capability snapshots,
|
||||||
declared outcome, and logical source requirements. Deployment saving then binds
|
declared outcome, and logical source requirements. Deployment saving then binds
|
||||||
the logical source `local.report` to the concrete configured source
|
the logical source `local.report` to the concrete configured source
|
||||||
`local.report`. Deployment validation checks that the bound source exists and
|
`local.report`, for example with
|
||||||
still satisfies the artifact's saved requirements before execution.
|
`wf deploy save report_case_study.default --artifact report_case_study --version 1 --binding local.report=local.report`.
|
||||||
|
Deployment validation checks that the bound source exists and still satisfies
|
||||||
|
the artifact's saved requirements before execution.
|
||||||
|
|
||||||
Run execution starts from the deployment, validates input, executes the
|
Run execution starts from the deployment, validates input, executes the
|
||||||
three-node pipeline, records trace frames, and stores a completed run record
|
three-node pipeline, records trace frames, and stores a completed run record
|
||||||
|
|||||||
@@ -104,17 +104,17 @@ one. Structurally identical definitions may be deduplicated.
|
|||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"schemas": [
|
"schemas": [
|
||||||
{
|
|
||||||
"name": "WorkflowDraft",
|
|
||||||
"aliases": ["draft"],
|
|
||||||
"kind": "root",
|
|
||||||
"description": "Patch-friendly JSON authoring document."
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"name": "NodeUse",
|
"name": "NodeUse",
|
||||||
"aliases": [],
|
"aliases": [],
|
||||||
"kind": "definition",
|
"kind": "definition",
|
||||||
"description": "Concrete use of a reusable node definition."
|
"description": "Concrete use of a reusable node definition."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "WorkflowDraft",
|
||||||
|
"aliases": ["draft"],
|
||||||
|
"kind": "root",
|
||||||
|
"description": "Patch-friendly JSON authoring document."
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
@@ -176,8 +176,8 @@ Example:
|
|||||||
constants, and basic validation bounds when present.
|
constants, and basic validation bounds when present.
|
||||||
- Preserve object property names.
|
- Preserve object property names.
|
||||||
- Convert local `$ref` values to canonical definition-name strings.
|
- Convert local `$ref` values to canonical definition-name strings.
|
||||||
- Convert `oneOf`/`anyOf` reference unions to `one_of` name lists when all
|
- Convert `oneOf` reference unions to `one_of` and `anyOf` reference unions to
|
||||||
branches are named references.
|
`any_of` when branches are named references.
|
||||||
- Preserve simple inline primitive unions in compact JSON form.
|
- Preserve simple inline primitive unions in compact JSON form.
|
||||||
- For arrays, summarize the item schema recursively.
|
- For arrays, summarize the item schema recursively.
|
||||||
- Add a sorted `related` list containing definitions referenced by the outline.
|
- Add a sorted `related` list containing definitions referenced by the outline.
|
||||||
@@ -203,19 +203,20 @@ For a component definition such as `NodeUse`, emit a valid root document:
|
|||||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||||
"$ref": "#/$defs/NodeUse",
|
"$ref": "#/$defs/NodeUse",
|
||||||
"$defs": {
|
"$defs": {
|
||||||
"NodeUse": {},
|
"NodeUse": {"type": "object", "properties": {"node": {"type": "string"}}},
|
||||||
"InputPathBinding": {},
|
"InputPathBinding": {"type": "object", "properties": {"path": {"type": "string"}}},
|
||||||
"InputValueBinding": {},
|
"InputValueBinding": {"type": "object", "properties": {"value": true}},
|
||||||
"OutputBinding": {}
|
"OutputBinding": {"type": "object", "properties": {"to": {"type": "string"}}}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The `$defs` table contains the full combined Pydantic-generated definition
|
The short `$defs` bodies above are representative excerpts, not complete
|
||||||
catalog. Do not hand-roll transitive reference pruning or JSON Schema reference
|
runtime output. The real `$defs` table contains the full combined
|
||||||
resolution in the first implementation. The larger verbose payload is an
|
Pydantic-generated definition catalog. Do not hand-roll transitive reference
|
||||||
acceptable tradeoff for correctness; `--verbose` is explicitly the unbounded
|
pruning or JSON Schema reference resolution in the first implementation. The
|
||||||
form.
|
larger verbose payload is an acceptable tradeoff for correctness; `--verbose`
|
||||||
|
is explicitly the unbounded form.
|
||||||
|
|
||||||
All verbose documents must pass `Draft202012Validator.check_schema()` and a
|
All verbose documents must pass `Draft202012Validator.check_schema()` and a
|
||||||
validator-backed local-reference resolution test.
|
validator-backed local-reference resolution test.
|
||||||
|
|||||||
@@ -317,14 +317,13 @@ def save_v2_manual_audit(
|
|||||||
f"got {harness_version!r}"
|
f"got {harness_version!r}"
|
||||||
)
|
)
|
||||||
|
|
||||||
workspace_path_str = result.get("workspace_path")
|
|
||||||
if not isinstance(workspace_path_str, str) or not workspace_path_str:
|
|
||||||
raise ValueError("result is missing workspace_path")
|
|
||||||
workspace_path = Path(workspace_path_str)
|
|
||||||
|
|
||||||
result_path_str = result.get("result_path")
|
result_path_str = result.get("result_path")
|
||||||
if not isinstance(result_path_str, str) or not result_path_str:
|
if not isinstance(result_path_str, str) or not result_path_str:
|
||||||
raise ValueError("result is missing result_path")
|
raise ValueError("result is missing result_path")
|
||||||
|
if Path(result_path_str).resolve() != result_path.resolve():
|
||||||
|
raise ValueError("result_path field does not match audited result file")
|
||||||
|
workspace_path = _trusted_workspace_for_result(result_path, result)
|
||||||
|
workspace_path_str = str(workspace_path)
|
||||||
|
|
||||||
audit_payload: dict[str, object] = {
|
audit_payload: dict[str, object] = {
|
||||||
"manual_audit": {
|
"manual_audit": {
|
||||||
@@ -371,3 +370,23 @@ def save_v2_manual_audit(
|
|||||||
machine=machine_path,
|
machine=machine_path,
|
||||||
results_markdown=results_markdown_path,
|
results_markdown=results_markdown_path,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _trusted_workspace_for_result(result_path: Path, result: dict[str, object]) -> Path:
|
||||||
|
"""Resolve the audited workspace without trusting arbitrary JSON paths."""
|
||||||
|
resolved_result = result_path.resolve()
|
||||||
|
challenge_root = resolved_result.parent.parent
|
||||||
|
if resolved_result.parent.name == "results":
|
||||||
|
sibling_workspace = (
|
||||||
|
challenge_root / "workspaces" / resolved_result.with_suffix("").name
|
||||||
|
).resolve()
|
||||||
|
if sibling_workspace.exists():
|
||||||
|
return sibling_workspace
|
||||||
|
|
||||||
|
workspace_path_str = result.get("workspace_path")
|
||||||
|
if not isinstance(workspace_path_str, str) or not workspace_path_str:
|
||||||
|
raise ValueError("result is missing workspace_path")
|
||||||
|
workspace_path = Path(workspace_path_str).resolve()
|
||||||
|
if not workspace_path.is_relative_to(challenge_root.resolve()):
|
||||||
|
raise ValueError("workspace_path escapes audited challenge directory")
|
||||||
|
return workspace_path
|
||||||
|
|||||||
@@ -7,13 +7,23 @@ import yaml
|
|||||||
from .models import ChallengeManifest, LoadedChallenge
|
from .models import ChallengeManifest, LoadedChallenge
|
||||||
|
|
||||||
|
|
||||||
def _inside(root: Path, relative: str, *, field: str) -> Path:
|
def _inside(
|
||||||
|
root: Path, relative: str, *, field: str, boundary: Path | None = None
|
||||||
|
) -> Path:
|
||||||
candidate = (root / relative).resolve()
|
candidate = (root / relative).resolve()
|
||||||
if not candidate.is_relative_to(root):
|
allowed_root = (boundary or root).resolve()
|
||||||
|
if not candidate.is_relative_to(allowed_root):
|
||||||
raise ValueError(f"challenge {field} must stay inside challenge directory")
|
raise ValueError(f"challenge {field} must stay inside challenge directory")
|
||||||
return candidate
|
return candidate
|
||||||
|
|
||||||
|
|
||||||
|
def _source_boundary(root: Path) -> Path:
|
||||||
|
"""Real challenge bundles may point at sibling example source directories."""
|
||||||
|
if root.parent.name == "agent_challenges":
|
||||||
|
return root.parent.parent
|
||||||
|
return root
|
||||||
|
|
||||||
|
|
||||||
def load_challenge_manifest(path: Path) -> LoadedChallenge:
|
def load_challenge_manifest(path: Path) -> LoadedChallenge:
|
||||||
manifest_path = path.resolve()
|
manifest_path = path.resolve()
|
||||||
root = manifest_path.parent
|
root = manifest_path.parent
|
||||||
@@ -23,8 +33,18 @@ def load_challenge_manifest(path: Path) -> LoadedChallenge:
|
|||||||
workspace_template = _inside(
|
workspace_template = _inside(
|
||||||
root, manifest.workspace_template, field="workspace_template"
|
root, manifest.workspace_template, field="workspace_template"
|
||||||
)
|
)
|
||||||
source_root = (root / manifest.source.root).resolve()
|
source_root = _inside(
|
||||||
server_config = (root / manifest.server.config).resolve()
|
root,
|
||||||
|
manifest.source.root,
|
||||||
|
field="source.root",
|
||||||
|
boundary=_source_boundary(root),
|
||||||
|
)
|
||||||
|
server_config = _inside(
|
||||||
|
root,
|
||||||
|
manifest.server.config,
|
||||||
|
field="server.config",
|
||||||
|
boundary=_source_boundary(root),
|
||||||
|
)
|
||||||
if not prompt_path.is_file():
|
if not prompt_path.is_file():
|
||||||
raise ValueError(f"challenge prompt does not exist: {prompt_path}")
|
raise ValueError(f"challenge prompt does not exist: {prompt_path}")
|
||||||
if not workspace_template.is_dir():
|
if not workspace_template.is_dir():
|
||||||
|
|||||||
@@ -147,12 +147,13 @@ def _build_tool_briefs(result: dict[str, object]) -> list[CommandToolBrief]:
|
|||||||
tc_input.get("path")
|
tc_input.get("path")
|
||||||
or tc_input.get("filePath")
|
or tc_input.get("filePath")
|
||||||
or tc_input.get("file")
|
or tc_input.get("file")
|
||||||
|
or tc_input.get("pattern")
|
||||||
or ""
|
or ""
|
||||||
)
|
)
|
||||||
if isinstance(path_val, str) and path_val:
|
if isinstance(path_val, str) and path_val:
|
||||||
detail_str = path_val[:_MAX_COMMAND_DETAIL_CHARS]
|
detail_str = path_val[:_MAX_COMMAND_DETAIL_CHARS]
|
||||||
elif "command" in tc_input:
|
elif "command" in tc_input or "cmd" in tc_input:
|
||||||
cmd = tc_input["command"]
|
cmd = tc_input.get("command") or tc_input.get("cmd")
|
||||||
if isinstance(cmd, str):
|
if isinstance(cmd, str):
|
||||||
detail_str = cmd[:_MAX_COMMAND_DETAIL_CHARS]
|
detail_str = cmd[:_MAX_COMMAND_DETAIL_CHARS]
|
||||||
|
|
||||||
|
|||||||
@@ -42,8 +42,9 @@ The hard timeout ceiling is 3,600 seconds per trial.
|
|||||||
|
|
||||||
## Workspace Layout
|
## Workspace Layout
|
||||||
|
|
||||||
- `workspace_template/` holds local store ignore rules (gitignored contents).
|
- `workspace_template/` seeds each isolated trial with task files such as
|
||||||
- `workspaces/` holds per-trial workspaces (gitignored).
|
`input.md`, `run-input.json`, and local store ignore rules.
|
||||||
|
- `workspaces/` holds copied per-trial workspaces (gitignored).
|
||||||
- `results/` holds per-trial raw result JSON and report projections
|
- `results/` holds per-trial raw result JSON and report projections
|
||||||
(gitignored).
|
(gitignored).
|
||||||
- `challenge.yaml` declares the manifest, source, server, and report schema.
|
- `challenge.yaml` declares the manifest, source, server, and report schema.
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
import sys
|
import sys
|
||||||
|
import uuid
|
||||||
from collections.abc import Sequence
|
from collections.abc import Sequence
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -152,7 +153,7 @@ def report_from_v2_result(result: dict[str, object]) -> str:
|
|||||||
metrics = result.get("metrics", {})
|
metrics = result.get("metrics", {})
|
||||||
if isinstance(metrics, dict):
|
if isinstance(metrics, dict):
|
||||||
tokens = metrics.get("tokens", {})
|
tokens = metrics.get("tokens", {})
|
||||||
if isinstance(tokens, dict):
|
if isinstance(tokens, dict) and tokens:
|
||||||
lines.append("Observed token metrics:")
|
lines.append("Observed token metrics:")
|
||||||
lines.append(f" {_format_tokens(tokens)}")
|
lines.append(f" {_format_tokens(tokens)}")
|
||||||
cost = metrics.get("cost")
|
cost = metrics.get("cost")
|
||||||
@@ -213,7 +214,7 @@ class TrialReportPaths:
|
|||||||
|
|
||||||
def _atomic_write_text(path: Path, text: str) -> None:
|
def _atomic_write_text(path: Path, text: str) -> None:
|
||||||
path.parent.mkdir(parents=True, exist_ok=True)
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
temporary = path.with_name(f".{path.name}.tmp")
|
temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp")
|
||||||
temporary.write_text(text, encoding="utf-8")
|
temporary.write_text(text, encoding="utf-8")
|
||||||
temporary.replace(path)
|
temporary.replace(path)
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
param(
|
param(
|
||||||
[string]$AttachUrl = "http://127.0.0.1:8192",
|
[string]$AttachUrl = "",
|
||||||
[int]$Trials = 5,
|
[int]$Trials = 5,
|
||||||
[int]$Concurrency = 2,
|
[int]$Concurrency = 2,
|
||||||
[int]$TimeoutSeconds = 3600,
|
[int]$TimeoutSeconds = 3600,
|
||||||
@@ -45,11 +45,13 @@ try {
|
|||||||
"$Trials",
|
"$Trials",
|
||||||
"--concurrency",
|
"--concurrency",
|
||||||
"$Concurrency",
|
"$Concurrency",
|
||||||
"--attach",
|
|
||||||
"$AttachUrl",
|
|
||||||
"--timeout-seconds",
|
"--timeout-seconds",
|
||||||
"$TimeoutSeconds"
|
"$TimeoutSeconds"
|
||||||
)
|
)
|
||||||
|
if ($AttachUrl) {
|
||||||
|
$argsList += "--attach"
|
||||||
|
$argsList += "$AttachUrl"
|
||||||
|
}
|
||||||
foreach ($model in $models) {
|
foreach ($model in $models) {
|
||||||
$argsList += "--model"
|
$argsList += "--model"
|
||||||
$argsList += "$($model.Model)=$($model.Variant)"
|
$argsList += "$($model.Model)=$($model.Variant)"
|
||||||
|
|||||||
@@ -65,6 +65,8 @@ def parse_model_profile(raw: str) -> ModelProfile:
|
|||||||
model, separator, variant = raw.partition("=")
|
model, separator, variant = raw.partition("=")
|
||||||
if not model:
|
if not model:
|
||||||
raise ValueError("model cannot be empty")
|
raise ValueError("model cannot be empty")
|
||||||
|
if separator and not variant:
|
||||||
|
raise ValueError("variant cannot be empty when using MODEL=VARIANT")
|
||||||
return ModelProfile(model=model, variant=variant if separator else "high")
|
return ModelProfile(model=model, variant=variant if separator else "high")
|
||||||
|
|
||||||
|
|
||||||
@@ -80,11 +82,15 @@ def build_matrix_tasks(
|
|||||||
for challenge in challenges:
|
for challenge in challenges:
|
||||||
results_dir = challenge.root / "results"
|
results_dir = challenge.root / "results"
|
||||||
workspaces_dir = challenge.root / "workspaces"
|
workspaces_dir = challenge.root / "workspaces"
|
||||||
|
next_indices: dict[str, int] = {}
|
||||||
for model in models:
|
for model in models:
|
||||||
next_index = starting_trial_index(
|
next_index = next_indices.setdefault(
|
||||||
|
model.model,
|
||||||
|
starting_trial_index(
|
||||||
model=model.model,
|
model=model.model,
|
||||||
results_dir=results_dir,
|
results_dir=results_dir,
|
||||||
workspaces_dir=workspaces_dir,
|
workspaces_dir=workspaces_dir,
|
||||||
|
),
|
||||||
)
|
)
|
||||||
for profile in profiles:
|
for profile in profiles:
|
||||||
for _ in range(trials):
|
for _ in range(trials):
|
||||||
@@ -100,6 +106,7 @@ def build_matrix_tasks(
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
next_index += 1
|
next_index += 1
|
||||||
|
next_indices[model.model] = next_index
|
||||||
return tasks
|
return tasks
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -482,6 +482,7 @@ def run_v2_trial(
|
|||||||
task_outcome = "failed"
|
task_outcome = "failed"
|
||||||
except subprocess.TimeoutExpired as exc:
|
except subprocess.TimeoutExpired as exc:
|
||||||
duration_seconds = time.monotonic() - started
|
duration_seconds = time.monotonic() - started
|
||||||
|
returncode = -1
|
||||||
stdout = (
|
stdout = (
|
||||||
exc.stdout
|
exc.stdout
|
||||||
if isinstance(exc.stdout, str)
|
if isinstance(exc.stdout, str)
|
||||||
@@ -499,7 +500,8 @@ def run_v2_trial(
|
|||||||
task_outcome = "timeout"
|
task_outcome = "timeout"
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
duration_seconds = time.monotonic() - started
|
duration_seconds = time.monotonic() - started
|
||||||
task_outcome = "parse_error"
|
returncode = -2
|
||||||
|
task_outcome = "runner_error"
|
||||||
parse_error = {
|
parse_error = {
|
||||||
"type": type(exc).__name__,
|
"type": type(exc).__name__,
|
||||||
"message": str(exc),
|
"message": str(exc),
|
||||||
|
|||||||
@@ -38,7 +38,7 @@ The full artifact/deployment/run path is covered by
|
|||||||
three-node lifecycle manually through the CLI, import the raw plan:
|
three-node lifecycle manually through the CLI, import the raw plan:
|
||||||
|
|
||||||
```powershell
|
```powershell
|
||||||
uv run wf --config examples/report_workflow/wf.config.json artifact create-from-plan examples/report_workflow/workflow.plan.json --artifact report_case_study --version 1 --title "Report Case Study" --outcome ok --binding local.report=local.report
|
uv run wf --config examples/report_workflow/wf.config.json artifact create-from-plan examples/report_workflow/workflow.plan.json --artifact report_case_study --version 1 --title "Report Case Study" --outcome ok
|
||||||
uv run wf --config examples/report_workflow/wf.config.json deploy save report_case_study.default --artifact report_case_study --version 1 --binding local.report=local.report
|
uv run wf --config examples/report_workflow/wf.config.json deploy save report_case_study.default --artifact report_case_study --version 1 --binding local.report=local.report
|
||||||
uv run wf --config examples/report_workflow/wf.config.json deploy validate report_case_study.default
|
uv run wf --config examples/report_workflow/wf.config.json deploy validate report_case_study.default
|
||||||
uv run wf --config examples/report_workflow/wf.config.json run start report_case_study.default --input-file examples/report_workflow/run-input.json --trace-from 0 --trace-limit 5
|
uv run wf --config examples/report_workflow/wf.config.json run start report_case_study.default --input-file examples/report_workflow/run-input.json --trace-from 0 --trace-limit 5
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from pydantic import BaseModel, Field
|
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||||
|
|
||||||
from wf_authoring import node
|
from wf_authoring import node
|
||||||
|
|
||||||
@@ -10,6 +10,10 @@ _EXAMPLE_DIR = Path(__file__).resolve().parent
|
|||||||
|
|
||||||
|
|
||||||
class ReadInput(BaseModel):
|
class ReadInput(BaseModel):
|
||||||
|
model_config = ConfigDict(
|
||||||
|
json_schema_extra={"oneOf": [{"required": ["text"]}, {"required": ["path"]}]}
|
||||||
|
)
|
||||||
|
|
||||||
text: str | None = Field(
|
text: str | None = Field(
|
||||||
default=None,
|
default=None,
|
||||||
description=(
|
description=(
|
||||||
@@ -22,6 +26,12 @@ class ReadInput(BaseModel):
|
|||||||
description="Legacy path to a UTF-8 Markdown notes file inside the example.",
|
description="Legacy path to a UTF-8 Markdown notes file inside the example.",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@model_validator(mode="after")
|
||||||
|
def require_exactly_one_input(self) -> ReadInput:
|
||||||
|
if (self.text is None) == (self.path is None):
|
||||||
|
raise ValueError("read_notes requires exactly one of text or path")
|
||||||
|
return self
|
||||||
|
|
||||||
|
|
||||||
class ReadOutput(BaseModel):
|
class ReadOutput(BaseModel):
|
||||||
text: str
|
text: str
|
||||||
@@ -94,7 +104,7 @@ def _read_notes(payload: ReadInput) -> ReadOutput:
|
|||||||
if payload.text is not None:
|
if payload.text is not None:
|
||||||
return ReadOutput(text=payload.text)
|
return ReadOutput(text=payload.text)
|
||||||
if payload.path is None:
|
if payload.path is None:
|
||||||
raise ValueError("read_notes requires either text or path")
|
raise ValueError("read_notes requires exactly one of text or path")
|
||||||
path = _resolve_example_path(payload.path)
|
path = _resolve_example_path(payload.path)
|
||||||
return ReadOutput(text=path.read_text(encoding="utf-8"))
|
return ReadOutput(text=path.read_text(encoding="utf-8"))
|
||||||
|
|
||||||
|
|||||||
@@ -8,8 +8,8 @@ unrunnable, or surprising.
|
|||||||
Check in this order:
|
Check in this order:
|
||||||
|
|
||||||
1. `wf status`
|
1. `wf status`
|
||||||
2. `wf cap inspect <capability>`
|
2. `wf cap list --format ids`
|
||||||
3. `wf cap list --format ids`
|
3. `wf cap inspect <capability>`
|
||||||
|
|
||||||
Remember: MCP control tools are not workflow capabilities. They appear in
|
Remember: MCP control tools are not workflow capabilities. They appear in
|
||||||
MCP `tools/list`, not `wf cap list`.
|
MCP `tools/list`, not `wf cap list`.
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ validated, runnable deployment.
|
|||||||
`wf artifact create-from-plan workflow.plan.json --artifact <artifact_id> --version 1 --title "Workflow Title" --outcome ok`
|
`wf artifact create-from-plan workflow.plan.json --artifact <artifact_id> --version 1 --title "Workflow Title" --outcome ok`
|
||||||
7. Save and validate a deployment.
|
7. Save and validate a deployment.
|
||||||
- `wf deploy save <deployment_id> --artifact <artifact_id> --version 1 --binding <logical_source>=<concrete_source>` (or `wf deploy create` alias)
|
- `wf deploy save <deployment_id> --artifact <artifact_id> --version 1 --binding <logical_source>=<concrete_source>` (or `wf deploy create` alias)
|
||||||
|
- `wf deploy validate <deployment_id>`
|
||||||
8. Run the deployment.
|
8. Run the deployment.
|
||||||
9. Inspect the run summary first; read bounded traces only when needed.
|
9. Inspect the run summary first; read bounded traces only when needed.
|
||||||
|
|
||||||
|
|||||||
@@ -25,6 +25,10 @@ def schema_command(
|
|||||||
) -> None:
|
) -> None:
|
||||||
"""Print a compact workflow schema outline or full JSON Schema."""
|
"""Print a compact workflow schema outline or full JSON Schema."""
|
||||||
if name is None or name == "list":
|
if name is None or name == "list":
|
||||||
|
if verbose:
|
||||||
|
raise typer.BadParameter(
|
||||||
|
"--verbose requires a schema name", param_hint="NAME"
|
||||||
|
)
|
||||||
emit_json(schema_catalog_payload())
|
emit_json(schema_catalog_payload())
|
||||||
return
|
return
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -76,6 +76,30 @@ def test_invalid_manifest_rejects_parent_traversal(tmp_path: Path) -> None:
|
|||||||
load_challenge_manifest(path)
|
load_challenge_manifest(path)
|
||||||
|
|
||||||
|
|
||||||
|
def test_invalid_manifest_rejects_source_root_escape(tmp_path: Path) -> None:
|
||||||
|
path = _write_manifest(tmp_path)
|
||||||
|
text = path.read_text(encoding="utf-8").replace(
|
||||||
|
" root: source",
|
||||||
|
" root: ../source",
|
||||||
|
)
|
||||||
|
path.write_text(text, encoding="utf-8")
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="source.root"):
|
||||||
|
load_challenge_manifest(path)
|
||||||
|
|
||||||
|
|
||||||
|
def test_invalid_manifest_rejects_server_config_escape(tmp_path: Path) -> None:
|
||||||
|
path = _write_manifest(tmp_path)
|
||||||
|
text = path.read_text(encoding="utf-8").replace(
|
||||||
|
" config: wf.config.json",
|
||||||
|
" config: ../wf.config.json",
|
||||||
|
)
|
||||||
|
path.write_text(text, encoding="utf-8")
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="server.config"):
|
||||||
|
load_challenge_manifest(path)
|
||||||
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[2]
|
ROOT = Path(__file__).resolve().parents[2]
|
||||||
|
|
||||||
|
|
||||||
@@ -968,6 +992,7 @@ def test_v2_runner_timeout_preserves_partial_evidence(tmp_path: Path) -> None:
|
|||||||
)
|
)
|
||||||
|
|
||||||
assert result["task_outcome"] == "timeout"
|
assert result["task_outcome"] == "timeout"
|
||||||
|
assert result["returncode"] == -1
|
||||||
assert "metrics" in result
|
assert "metrics" in result
|
||||||
|
|
||||||
|
|
||||||
@@ -1252,7 +1277,8 @@ def test_v2_runner_preserves_evidence_on_parse_failure(tmp_path: Path) -> None:
|
|||||||
run_fn=fake_run,
|
run_fn=fake_run,
|
||||||
)
|
)
|
||||||
|
|
||||||
assert result["task_outcome"] == "parse_error"
|
assert result["task_outcome"] == "runner_error"
|
||||||
|
assert result["returncode"] == -2
|
||||||
assert "parse_error" in result
|
assert "parse_error" in result
|
||||||
assert result["parse_error"]["type"] == "RuntimeError"
|
assert result["parse_error"]["type"] == "RuntimeError"
|
||||||
assert "subprocess exploded" in result["parse_error"]["message"]
|
assert "subprocess exploded" in result["parse_error"]["message"]
|
||||||
@@ -1595,6 +1621,7 @@ def test_runner_to_report_timeout(tmp_path: Path) -> None:
|
|||||||
)
|
)
|
||||||
|
|
||||||
assert result["task_outcome"] == "timeout"
|
assert result["task_outcome"] == "timeout"
|
||||||
|
assert result["returncode"] == -1
|
||||||
assert isinstance(result.get("workspace_path"), str)
|
assert isinstance(result.get("workspace_path"), str)
|
||||||
assert isinstance(result.get("report_paths"), dict)
|
assert isinstance(result.get("report_paths"), dict)
|
||||||
|
|
||||||
|
|||||||
@@ -25,6 +25,13 @@ def test_parse_model_profile_accepts_explicit_variant() -> None:
|
|||||||
assert parsed == ModelProfile("opencode/deepseek-v4-flash-free", "max")
|
assert parsed == ModelProfile("opencode/deepseek-v4-flash-free", "max")
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_model_profile_rejects_empty_variant() -> None:
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="variant cannot be empty"):
|
||||||
|
parse_model_profile("opencode/deepseek-v4-flash-free=")
|
||||||
|
|
||||||
|
|
||||||
def test_matrix_tasks_allocate_indices_across_profiles(tmp_path: Path) -> None:
|
def test_matrix_tasks_allocate_indices_across_profiles(tmp_path: Path) -> None:
|
||||||
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
|
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
|
||||||
(challenge.root / "results").mkdir()
|
(challenge.root / "results").mkdir()
|
||||||
@@ -49,6 +56,27 @@ def test_matrix_tasks_allocate_indices_across_profiles(tmp_path: Path) -> None:
|
|||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_matrix_tasks_allocate_indices_across_variants(tmp_path: Path) -> None:
|
||||||
|
challenge = load_challenge_manifest(_write_manifest(tmp_path / "challenge"))
|
||||||
|
|
||||||
|
tasks = build_matrix_tasks(
|
||||||
|
challenges=[challenge],
|
||||||
|
profiles=[InstructionProfile.NONE],
|
||||||
|
models=[
|
||||||
|
ModelProfile("opencode/mimo-v2.5-free", "high"),
|
||||||
|
ModelProfile("opencode/mimo-v2.5-free", "max"),
|
||||||
|
],
|
||||||
|
trials=2,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert [(task.variant, task.index) for task in tasks] == [
|
||||||
|
("high", 1),
|
||||||
|
("high", 2),
|
||||||
|
("max", 3),
|
||||||
|
("max", 4),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def test_run_trials_concurrency_invokes_all_indices(
|
def test_run_trials_concurrency_invokes_all_indices(
|
||||||
monkeypatch,
|
monkeypatch,
|
||||||
tmp_path: Path,
|
tmp_path: Path,
|
||||||
|
|||||||
@@ -138,6 +138,46 @@ def test_command_brief_supports_filepath(tmp_path: Path) -> None:
|
|||||||
assert cmd["detail"].endswith("app.py")
|
assert cmd["detail"].endswith("app.py")
|
||||||
|
|
||||||
|
|
||||||
|
def test_command_brief_supports_cmd_and_pattern(tmp_path: Path) -> None:
|
||||||
|
from examples.agent_challenges.report_models import build_trial_report
|
||||||
|
|
||||||
|
result = _raw_result(tmp_path)
|
||||||
|
metrics = result.get("metrics", {})
|
||||||
|
if isinstance(metrics, dict):
|
||||||
|
metrics["tool_calls"] = [
|
||||||
|
{
|
||||||
|
"ordinal": 1,
|
||||||
|
"call_id": "c3",
|
||||||
|
"tool": "grep",
|
||||||
|
"status": "success",
|
||||||
|
"title": "Search files",
|
||||||
|
"input": {"pattern": "*.json"},
|
||||||
|
"metadata": {},
|
||||||
|
"output_chars": 100,
|
||||||
|
"output_preview": "content",
|
||||||
|
"output_sha256": "xyz",
|
||||||
|
"failed": False,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ordinal": 2,
|
||||||
|
"call_id": "c4",
|
||||||
|
"tool": "shell",
|
||||||
|
"status": "success",
|
||||||
|
"title": "Run command",
|
||||||
|
"input": {"cmd": "uv run wf status"},
|
||||||
|
"metadata": {},
|
||||||
|
"output_chars": 100,
|
||||||
|
"output_preview": "content",
|
||||||
|
"output_sha256": "xyz",
|
||||||
|
"failed": False,
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
payload = build_trial_report(result, audit=None).model_dump(mode="json")
|
||||||
|
assert payload["commands_and_tools"][0]["detail"] == "*.json"
|
||||||
|
assert payload["commands_and_tools"][1]["detail"] == "uv run wf status"
|
||||||
|
|
||||||
|
|
||||||
def test_markdown_projection_has_stable_headings(tmp_path: Path) -> None:
|
def test_markdown_projection_has_stable_headings(tmp_path: Path) -> None:
|
||||||
from examples.agent_challenges.report_models import build_trial_report
|
from examples.agent_challenges.report_models import build_trial_report
|
||||||
from examples.agent_challenges.reports import render_trial_report_markdown
|
from examples.agent_challenges.reports import render_trial_report_markdown
|
||||||
@@ -224,7 +264,7 @@ def test_projections_write_both_files(tmp_path: Path) -> None:
|
|||||||
assert machine_content["schema_version"] == 1
|
assert machine_content["schema_version"] == 1
|
||||||
assert machine_content["manual_audit"]["status"] == "pending"
|
assert machine_content["manual_audit"]["status"] == "pending"
|
||||||
|
|
||||||
assert list(tmp_path.iterdir()) == [markdown_path, machine_path]
|
assert set(tmp_path.iterdir()) == {markdown_path, machine_path}
|
||||||
|
|
||||||
|
|
||||||
def test_projections_exclude_raw_outputs(tmp_path: Path) -> None:
|
def test_projections_exclude_raw_outputs(tmp_path: Path) -> None:
|
||||||
@@ -340,6 +380,30 @@ def test_manual_audit_invalid_outcome_raises_and_preserves_projections(
|
|||||||
assert results_md_path.read_bytes() == results_md_before
|
assert results_md_path.read_bytes() == results_md_before
|
||||||
|
|
||||||
|
|
||||||
|
def test_manual_audit_rejects_mismatched_result_path(tmp_path: Path) -> None:
|
||||||
|
from examples.agent_challenges.audit import save_v2_manual_audit
|
||||||
|
|
||||||
|
result_path = _write_v2_result(tmp_path)
|
||||||
|
payload = json.loads(result_path.read_text(encoding="utf-8"))
|
||||||
|
payload["result_path"] = str(tmp_path / "results" / "other.json")
|
||||||
|
result_path.write_text(json.dumps(payload), encoding="utf-8")
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="result_path field"):
|
||||||
|
save_v2_manual_audit(result_path, official_outcome="pass")
|
||||||
|
|
||||||
|
|
||||||
|
def test_manual_audit_rejects_workspace_escape(tmp_path: Path) -> None:
|
||||||
|
from examples.agent_challenges.audit import save_v2_manual_audit
|
||||||
|
|
||||||
|
result_path = _write_v2_result(tmp_path)
|
||||||
|
payload = json.loads(result_path.read_text(encoding="utf-8"))
|
||||||
|
payload["workspace_path"] = str(tmp_path.parent / "outside")
|
||||||
|
result_path.write_text(json.dumps(payload), encoding="utf-8")
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="workspace_path escapes"):
|
||||||
|
save_v2_manual_audit(result_path, official_outcome="pass")
|
||||||
|
|
||||||
|
|
||||||
def test_trial_report_bounds_agent_self_report_payload(tmp_path: Path) -> None:
|
def test_trial_report_bounds_agent_self_report_payload(tmp_path: Path) -> None:
|
||||||
from examples.agent_challenges.report_models import build_trial_report
|
from examples.agent_challenges.report_models import build_trial_report
|
||||||
|
|
||||||
|
|||||||
@@ -114,3 +114,20 @@ def test_report_workflow_read_notes_accepts_text_by_value() -> None:
|
|||||||
notes = _read_notes(ReadInput(**payload))
|
notes = _read_notes(ReadInput(**payload))
|
||||||
|
|
||||||
assert notes.text.startswith("# Weekly Project Update")
|
assert notes.text.startswith("# Weekly Project Update")
|
||||||
|
|
||||||
|
|
||||||
|
def test_report_workflow_read_input_requires_text_or_path() -> None:
|
||||||
|
with pytest.raises(ValueError, match="text or path"):
|
||||||
|
ReadInput.model_validate({})
|
||||||
|
|
||||||
|
|
||||||
|
def test_report_workflow_read_input_rejects_text_and_path_together() -> None:
|
||||||
|
with pytest.raises(ValueError, match="exactly one"):
|
||||||
|
ReadInput(text="hello", path="input.md")
|
||||||
|
|
||||||
|
|
||||||
|
def test_report_workflow_read_input_schema_requires_text_or_path() -> None:
|
||||||
|
schema = ReadInput.model_json_schema()
|
||||||
|
|
||||||
|
assert schema["oneOf"] == [{"required": ["text"]}, {"required": ["path"]}]
|
||||||
|
assert set(schema["properties"]) == {"text", "path"}
|
||||||
|
|||||||
@@ -116,6 +116,13 @@ def test_schema_unknown_name_fails_with_suggestion() -> None:
|
|||||||
assert "NodeUse" in result.output
|
assert "NodeUse" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
def test_schema_verbose_without_name_fails() -> None:
|
||||||
|
result = runner.invoke(app, ["schema", "--verbose"])
|
||||||
|
|
||||||
|
assert result.exit_code != 0
|
||||||
|
assert "--verbose requires a schema name" in result.output
|
||||||
|
|
||||||
|
|
||||||
def test_schema_catalog_resolves_aliases_and_components() -> None:
|
def test_schema_catalog_resolves_aliases_and_components() -> None:
|
||||||
catalog = schema_catalog()
|
catalog = schema_catalog()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user