docs: finalize thesis evaluation and front matter
This commit is contained in:
@@ -13,7 +13,8 @@ FIGURE_STEMS = (
|
||||
"agent-challenge-audited-outcomes-by-cell",
|
||||
"agent-challenge-automatic-vs-manual-outcomes",
|
||||
"agent-challenge-longitudinal-outcomes",
|
||||
"agent-challenge-duration-and-tokens",
|
||||
"agent-challenge-duration",
|
||||
"agent-challenge-token-volume",
|
||||
)
|
||||
_MANUAL_OUTCOMES = frozenset({"pass", "invalid", "fail"})
|
||||
_CHALLENGE_ORDER = {"browser": 0, "report": 1}
|
||||
@@ -210,17 +211,20 @@ def render_evaluation_markdown(cohort: EvaluationCohort) -> str:
|
||||
"## Audited Agent Challenge Campaign",
|
||||
"",
|
||||
(
|
||||
f"The primary campaign contains {len(cohort.trials)} audited trials: "
|
||||
f"{outcomes['pass']} passes, {outcomes['invalid']} invalid samples, "
|
||||
f"and {outcomes['fail']} failure."
|
||||
f"The primary campaign contains {len(cohort.trials)} manually audited "
|
||||
f"trials: {outcomes['pass']} clean product-path passes under the "
|
||||
f"campaign rules, {outcomes['invalid']} invalid evaluation samples, "
|
||||
f"and {outcomes['fail']} failure. These counts are not a "
|
||||
f"model-success-rate estimate."
|
||||
),
|
||||
"",
|
||||
(
|
||||
"The campaign crosses two challenges, two hosted models, three instruction "
|
||||
"profiles (`none`, `skills`, and `all`), and three repetitions per cell. "
|
||||
"The checked cohort snapshot records report hashes, prompt hashes, the "
|
||||
"repository commit, automatic metrics, and manual-audit outcomes; local "
|
||||
"raw report files are verified against those hashes when present."
|
||||
"The campaign crosses two challenges × two hosted models × three "
|
||||
"instruction profiles (`none`, `skills`, and `all`) = 12 cells, with "
|
||||
"three repetitions per cell (n=3). The checked cohort snapshot records "
|
||||
"report hashes, prompt hashes, the repository commit, automatic metrics, "
|
||||
"and manual-audit outcomes; local raw report files are verified against "
|
||||
"those hashes when present."
|
||||
),
|
||||
"",
|
||||
(
|
||||
@@ -228,6 +232,10 @@ def render_evaluation_markdown(cohort: EvaluationCohort) -> str:
|
||||
"this is longitudinal engineering evidence, not a controlled model comparison."
|
||||
),
|
||||
"",
|
||||
"> **Campaign validity note.** This campaign is a bounded longitudinal audit, "
|
||||
"not a controlled comparison. Each cell has n=3; waves changed product and "
|
||||
"prompt snapshots; all audits were performed by the author.",
|
||||
"",
|
||||
f"Selection rule: {cohort.selection_rule}",
|
||||
"",
|
||||
"| Challenge / model / profile | Pass | Invalid | Fail |",
|
||||
@@ -244,10 +252,12 @@ def render_evaluation_markdown(cohort: EvaluationCohort) -> str:
|
||||
"",
|
||||
(
|
||||
"A manual `pass` requires both successful product-path evidence and an "
|
||||
"acceptable audit trail. `Invalid` means the sample cannot support the "
|
||||
"clean benchmark claim, commonly because the agent read repository or "
|
||||
"example material outside its supplied workspace. `Fail` means the "
|
||||
"challenge contract itself was not established."
|
||||
"acceptable audit trail. It does not imply the agent avoided every "
|
||||
"exploratory read, only that no disqualifying read or bypass was found. "
|
||||
"`Invalid` means the sample cannot support the clean benchmark claim, "
|
||||
"commonly because the agent read repository or example material outside "
|
||||
"its supplied workspace. `Fail` means the challenge contract itself was "
|
||||
"not established."
|
||||
),
|
||||
"",
|
||||
"{#fig:agent-challenge-audited-outcomes-by-cell width=95%}",
|
||||
@@ -255,7 +265,10 @@ def render_evaluation_markdown(cohort: EvaluationCohort) -> str:
|
||||
(
|
||||
"[@fig:agent-challenge-audited-outcomes-by-cell] reports all three "
|
||||
"repetitions rather than hiding invalid samples. The profile labels are "
|
||||
"descriptive; this campaign does not isolate instruction-profile effects."
|
||||
"descriptive; this campaign does not isolate instruction-profile effects. "
|
||||
"Profile × wave is confounded because the base prompt changed before "
|
||||
"wave 3, so apparent differences may reflect prompt changes, model "
|
||||
"updates, or repository drift rather than instruction-layer effects."
|
||||
),
|
||||
"",
|
||||
"{#fig:agent-challenge-automatic-vs-manual-outcomes width=75%}",
|
||||
@@ -275,14 +288,21 @@ def render_evaluation_markdown(cohort: EvaluationCohort) -> str:
|
||||
"changed. They preserve the chronology needed to study those changes."
|
||||
),
|
||||
"",
|
||||
"{#fig:agent-challenge-duration-and-tokens width=95%}",
|
||||
"{#fig:agent-challenge-duration width=78%}",
|
||||
"",
|
||||
(
|
||||
"[@fig:agent-challenge-duration-and-tokens] separates each challenge and "
|
||||
"metric into its own panel. Circle and square markers redundantly identify "
|
||||
"the models without relying on color. Wall-clock duration includes hosted-service "
|
||||
"latency, and OpenCode token totals include cache-read accounting, so neither "
|
||||
"axis is a normalized model-efficiency metric."
|
||||
"[@fig:agent-challenge-duration] separates the two challenges. Circle "
|
||||
"and square markers redundantly identify the models without relying on "
|
||||
"color. Wall-clock duration includes hosted-service latency and is not a "
|
||||
"normalized model-efficiency metric."
|
||||
),
|
||||
"",
|
||||
"{#fig:agent-challenge-token-volume width=78%}",
|
||||
"",
|
||||
(
|
||||
"[@fig:agent-challenge-token-volume] reports OpenCode token totals, which "
|
||||
"include cache-read accounting. The figure records observed workload volume; "
|
||||
"it is not an efficiency comparison."
|
||||
),
|
||||
"",
|
||||
"### Campaign Limitations",
|
||||
|
||||
@@ -312,23 +312,24 @@ def _scatter_metric(
|
||||
axis.grid(axis="y")
|
||||
|
||||
|
||||
def _duration_and_tokens(cohort: EvaluationCohort, plt: Any) -> Figure:
|
||||
def _metric_by_challenge(
|
||||
cohort: EvaluationCohort,
|
||||
plt: Any,
|
||||
*,
|
||||
metric: str,
|
||||
) -> Figure:
|
||||
"""Render one readable metric panel per challenge."""
|
||||
from matplotlib.lines import Line2D
|
||||
|
||||
figure, axes = plt.subplots(2, 2, figsize=(9.6, 6.8), sharex=True)
|
||||
figure, axes = plt.subplots(2, 1, figsize=(7.4, 6.8), sharex=True)
|
||||
challenges = (("browser", "Browser click"), ("report", "Report workflow"))
|
||||
for row, (challenge, challenge_label) in enumerate(challenges):
|
||||
for axis, (challenge, challenge_label) in zip(axes, challenges, strict=True):
|
||||
trials = [trial for trial in cohort.trials if trial.challenge == challenge]
|
||||
duration_axis, token_axis = axes[row]
|
||||
_scatter_metric(duration_axis, trials, metric="duration")
|
||||
_scatter_metric(token_axis, trials, metric="tokens")
|
||||
duration_axis.set_ylabel(f"{challenge_label}\nMinutes")
|
||||
token_axis.set_ylabel(f"{challenge_label}\nMillion tokens")
|
||||
_scatter_metric(axis, trials, metric=metric)
|
||||
unit = "Minutes" if metric == "duration" else "Million tokens"
|
||||
axis.set_ylabel(f"{challenge_label}\n{unit}")
|
||||
|
||||
axes[0, 0].set_title("Wall-clock duration")
|
||||
axes[0, 1].set_title("Recorded token volume")
|
||||
axes[1, 0].set_xlabel("Instruction profile")
|
||||
axes[1, 1].set_xlabel("Instruction profile")
|
||||
axes[-1].set_xlabel("Instruction profile")
|
||||
legend_handles = [
|
||||
Line2D(
|
||||
[],
|
||||
@@ -342,8 +343,13 @@ def _duration_and_tokens(cohort: EvaluationCohort, plt: Any) -> Figure:
|
||||
)
|
||||
for model, style in _MODEL_STYLES.items()
|
||||
]
|
||||
title = (
|
||||
"Wall-clock duration by profile, model, and wave"
|
||||
if metric == "duration"
|
||||
else "Recorded token volume by profile, model, and wave"
|
||||
)
|
||||
figure.suptitle(
|
||||
"Runtime evidence by challenge, profile, model, and wave",
|
||||
title,
|
||||
fontsize=13,
|
||||
fontweight="bold",
|
||||
)
|
||||
@@ -357,7 +363,11 @@ def _duration_and_tokens(cohort: EvaluationCohort, plt: Any) -> Figure:
|
||||
figure.text(
|
||||
0.5,
|
||||
0.015,
|
||||
"Point labels 1–3 identify waves; token totals include OpenCode cache-read accounting.",
|
||||
(
|
||||
"Point labels 1–3 identify waves."
|
||||
if metric == "duration"
|
||||
else "Point labels 1–3 identify waves; totals include OpenCode cache-read accounting."
|
||||
),
|
||||
ha="center",
|
||||
color="#4C5961",
|
||||
fontsize=8,
|
||||
@@ -378,7 +388,14 @@ def render_evaluation_figures(
|
||||
_automatic_vs_manual(cohort, plt),
|
||||
),
|
||||
("agent-challenge-longitudinal-outcomes", _longitudinal_outcomes(cohort, plt)),
|
||||
("agent-challenge-duration-and-tokens", _duration_and_tokens(cohort, plt)),
|
||||
(
|
||||
"agent-challenge-duration",
|
||||
_metric_by_challenge(cohort, plt, metric="duration"),
|
||||
),
|
||||
(
|
||||
"agent-challenge-token-volume",
|
||||
_metric_by_challenge(cohort, plt, metric="tokens"),
|
||||
),
|
||||
)
|
||||
written: list[Path] = []
|
||||
for stem, figure in figures:
|
||||
|
||||
Reference in New Issue
Block a user