docs: publish thesis evaluation bundle
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
*.html
|
||||
*.pdf
|
||||
*.tex
|
||||
*.typ
|
||||
!figures/*.svg
|
||||
@@ -0,0 +1,625 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"cohort_id": "agent-operability-n3-2026-06-30",
|
||||
"title": "Agent-operability longitudinal n=3 campaign",
|
||||
"selection_rule": "The latest three completed, manually audited trials per challenge, model, and instruction profile as of 2026-06-30.",
|
||||
"limitations": [
|
||||
"The three waves span repository snapshots; they are longitudinal engineering evidence, not a controlled model comparison.",
|
||||
"The base prompt changed before wave 3 to require the challenge report inline.",
|
||||
"The models were free hosted OpenCode endpoints, so service load and latency were not controlled."
|
||||
],
|
||||
"runs": [
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-034.report.json",
|
||||
"report_sha256": "337cc7db84c1c95d58e8ab85525ed2739f8fab58acc69fcd6f74f041f7b463eb",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "none",
|
||||
"trial_index": 34,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 305.975,
|
||||
"tokens_total": 1267986,
|
||||
"audited_at": "2026-06-29T21:19:49Z",
|
||||
"audit_notes": "Valid product-path success. Draft authoring completed after repaired bindings and a deployment binding retry; no disqualifying reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-035.report.json",
|
||||
"report_sha256": "0ba1212eae4303d066577f170a40e0a5892b9945eeb80005be9873772e562348",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 35,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 255.076,
|
||||
"tokens_total": 1198504,
|
||||
"audited_at": "2026-06-29T21:19:49Z",
|
||||
"audit_notes": "Valid product-path success. Used supplied skills and trial workspace files only; no disqualifying reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-036.report.json",
|
||||
"report_sha256": "c523a041223ac83d68d4c72ec8b6d7668ee62dc3f8b84482d84858815a803dff",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "all",
|
||||
"trial_index": 36,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 223.147,
|
||||
"tokens_total": 1041287,
|
||||
"audited_at": "2026-06-29T21:19:50Z",
|
||||
"audit_notes": "Valid product-path success. All-profile reads were docs/skills/challenge context, not existing solution or implementation code."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-027.report.json",
|
||||
"report_sha256": "b287a707f1f34abf1e3ef3110f2815b576e75f86f3c130d8913ff152d9a39a00",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "none",
|
||||
"trial_index": 27,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "failed",
|
||||
"duration_seconds": 1103.134,
|
||||
"tokens_total": 1073828,
|
||||
"audited_at": "2026-06-30T09:39:44Z",
|
||||
"audit_notes": "Manual outcome pass with resume evidence. Automatic contamination came from broad repo/workspace reads, but sampled transcript did not show product-code, adjacent-solution, prior-store, or helper-script bypass."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-028.report.json",
|
||||
"report_sha256": "ea400f221a58fd13dc29432b5b1857eabdc730e314777d09a07e0b265621759e",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 28,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 164.709,
|
||||
"tokens_total": 564051,
|
||||
"audited_at": "2026-06-29T21:20:17Z",
|
||||
"audit_notes": "Valid product-path success. Used supplied skill docs and raw-plan path; no existing solution, product code, prior store, or adjacent attempt reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-029.report.json",
|
||||
"report_sha256": "a55e2658b71ef1dc17174a76a7a2304b4818becd7e8cef499616b8af6b8af3df",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "all",
|
||||
"trial_index": 29,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 197.079,
|
||||
"tokens_total": 641558,
|
||||
"audited_at": "2026-06-29T21:20:18Z",
|
||||
"audit_notes": "Valid product-path success. All-profile docs/skills reads did not include existing solution or implementation code."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-024.report.json",
|
||||
"report_sha256": "5e40f75599c291808e1c378962989e749f80f4daeeeabce08777dee8c08b1173",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "none",
|
||||
"trial_index": 24,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "failed",
|
||||
"duration_seconds": 934.753,
|
||||
"tokens_total": 326439,
|
||||
"audited_at": "2026-06-30T09:38:52Z",
|
||||
"audit_notes": "Manual outcome pass with resume evidence. Reads were limited to trial workspace fixtures and public CLI capability inspection in sampled report."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-025.report.json",
|
||||
"report_sha256": "d7c5be60ec253c42667154299127d960e2afce23914d6b0e3aac4d13932e389e",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 25,
|
||||
"repository_commit": "30ad99fca1e382d13fc1daacf9772ce5e709013f",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "failed",
|
||||
"duration_seconds": 854.959,
|
||||
"tokens_total": 56915,
|
||||
"audited_at": "2026-06-30T09:38:53Z",
|
||||
"audit_notes": "Manual outcome pass with resume evidence. Automatic contamination came from supplied skills/example-directory reads; sampled evidence did not show reading an existing solution or product implementation code."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-026.report.json",
|
||||
"report_sha256": "e08248b55ca9af2a6cf4163176cb29f22254261c1bbbe7257ea244956c53702d",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "all",
|
||||
"trial_index": 26,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 249.318,
|
||||
"tokens_total": 1346114,
|
||||
"audited_at": "2026-06-30T11:33:50Z",
|
||||
"audit_notes": "Invalid as clean benchmark. Product path completed, but the run used existing solution/example implementation as reference."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-024.report.json",
|
||||
"report_sha256": "1fb02a19d0a3e0c7baec01e2d15df11f2016224d56e872edc910b796e224d4be",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "none",
|
||||
"trial_index": 24,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 187.106,
|
||||
"tokens_total": 612379,
|
||||
"audited_at": "2026-06-30T11:33:49Z",
|
||||
"audit_notes": "Clean report-workflow pass. Agent used none profile, workspace files only, raw-plan path, no helper script or disallowed reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-023.report.json",
|
||||
"report_sha256": "a38ae6ced8eae200b713c23a4cf6ebd07ec3cf72f1991490cad52436a8df3cf2",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 23,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 253.113,
|
||||
"tokens_total": 823612,
|
||||
"audited_at": "2026-06-30T11:33:49Z",
|
||||
"audit_notes": "Clean report-workflow pass. Agent used supplied skills and workspace files only, completed through public wf commands, no helper script or disallowed reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 1,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-025.report.json",
|
||||
"report_sha256": "08ca446fca8e708c89ee426d5e174949f398953b708e4212684337258fe3d68c",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "all",
|
||||
"trial_index": 25,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 196.772,
|
||||
"tokens_total": 690458,
|
||||
"audited_at": "2026-06-30T11:33:49Z",
|
||||
"audit_notes": "Clean report-workflow pass. Agent used all profile resources without reading product code, existing solution, prior store, or adjacent attempts; raw-plan path succeeded on first attempt."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-037.report.json",
|
||||
"report_sha256": "52f429902bfd68b1b8893b5be03fa050d302968b2b6bb8906613e9ac5c6bde32",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "none",
|
||||
"trial_index": 37,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 257.747,
|
||||
"tokens_total": 1355428,
|
||||
"audited_at": "2026-06-30T11:32:57Z",
|
||||
"audit_notes": "Clean browser-click pass. Agent used draft path, public wf commands, no helper script, no product-code or existing-solution reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-038.report.json",
|
||||
"report_sha256": "dc643e38440d6c644c4149be49b400deb021dfe2eb68f5ff605c54ef8491a26a",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 38,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 210.851,
|
||||
"tokens_total": 1009244,
|
||||
"audited_at": "2026-06-30T11:32:57Z",
|
||||
"audit_notes": "Manual outcome pass with policy note. Useful evidence that skills profile solved browser via draft path."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-039.report.json",
|
||||
"report_sha256": "ab4cda9f59c9e101d551cbd3a3399e8f3b600b31bfa9f6a7e8321dbe1947ad88",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "all",
|
||||
"trial_index": 39,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 240.817,
|
||||
"tokens_total": 1271971,
|
||||
"audited_at": "2026-06-30T11:32:58Z",
|
||||
"audit_notes": "Clean browser-click pass. Agent used all-profile skills/workspace resources without disallowed reads."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-030.report.json",
|
||||
"report_sha256": "01e19f100066e1e69609268086924dfeed9d9a1f396e28283c0e5b571796eb8b",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "none",
|
||||
"trial_index": 30,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "fail",
|
||||
"task_outcome": "failed",
|
||||
"duration_seconds": 198.976,
|
||||
"tokens_total": 575081,
|
||||
"audited_at": "2026-06-30T11:32:58Z",
|
||||
"audit_notes": "Manual outcome fail for challenge contract. Treat as useful UX/process-cleanup evidence; could be false-positive Chrome/process detection, but no contrary evidence was captured."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-031.report.json",
|
||||
"report_sha256": "2de73f2b87182ae822a1a1cb0065c9f4fd243248c0293b7b7a0bacc8b523bcde",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 31,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "failed",
|
||||
"duration_seconds": 373.645,
|
||||
"tokens_total": 1659079,
|
||||
"audited_at": "2026-06-30T11:32:58Z",
|
||||
"audit_notes": "Invalid benchmark sample. Useful prompt finding: final self-report must be inline; prompt was updated after this run."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-032.report.json",
|
||||
"report_sha256": "8c758fb289a1dd41b82ab181dda33bc52e7163f48002b641913666ddc0f2cf7f",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "all",
|
||||
"trial_index": 32,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 283.632,
|
||||
"tokens_total": 1152542,
|
||||
"audited_at": "2026-06-30T11:32:59Z",
|
||||
"audit_notes": "Clean browser-click pass. Agent used raw-plan path, completed with before=false and after=true, no leftover processes."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-027.report.json",
|
||||
"report_sha256": "6e0fce88c196713a23fde3558df204ba5b8047b46c48f5e7b27249cd79d20a34",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "none",
|
||||
"trial_index": 27,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 314.767,
|
||||
"tokens_total": 1349666,
|
||||
"audited_at": "2026-06-30T10:42:07Z",
|
||||
"audit_notes": "Clean pass for report workflow. Agent used raw-plan path, public wf commands, no helper script, no product-code or existing-solution reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-028.report.json",
|
||||
"report_sha256": "7d1db2a8b2974d2815856faffd8e4d2898eb90d663e670024e1e52eb54928852",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 28,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 244.242,
|
||||
"tokens_total": 1087828,
|
||||
"audited_at": "2026-06-30T10:42:07Z",
|
||||
"audit_notes": "Manual outcome pass. Keep a policy note: not as clean as mimo skills because the transcript read challenge directory/prompt artifacts, but it did not appear to copy a ready-made answer."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-029.report.json",
|
||||
"report_sha256": "55c484089b471462132398fcb955afab8b36ea00723cd5ada19855ac6f7d0661",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "all",
|
||||
"trial_index": 29,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 187.826,
|
||||
"tokens_total": 749971,
|
||||
"audited_at": "2026-06-30T10:42:08Z",
|
||||
"audit_notes": "Invalid as clean benchmark. Product path completed, but the run used an existing solution/example implementation as reference."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-026.report.json",
|
||||
"report_sha256": "6423d2f6fb918e33fca2f826532ee1b34dd8a208cbb29cdb988e26fcda191a8b",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "none",
|
||||
"trial_index": 26,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 224.583,
|
||||
"tokens_total": 757043,
|
||||
"audited_at": "2026-06-30T10:42:08Z",
|
||||
"audit_notes": "Invalid as strict none-profile benchmark. Useful UX evidence: missed source binding, then missing end edge, then succeeded."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-027.report.json",
|
||||
"report_sha256": "97c9365feeb5401e91d0a0bd92d7a75e87745c08b1dbdaadd639e2a8a777136c",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 27,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 258.463,
|
||||
"tokens_total": 1043878,
|
||||
"audited_at": "2026-06-30T10:42:08Z",
|
||||
"audit_notes": "Clean pass. Agent used supplied skills and workspace files only, completed through public wf commands, no helper script or disallowed reads observed."
|
||||
},
|
||||
{
|
||||
"wave": 2,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-028.report.json",
|
||||
"report_sha256": "5305c841ea80b697b0f72eeedc693bb2fcb87623ceabfe88100606e46b65b79e",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "all",
|
||||
"trial_index": 28,
|
||||
"repository_commit": "04b1cf30f93f5946bfb7f0b001036319365aa237",
|
||||
"base_prompt_hash": "ae06f961d3c64d4d53da259cc0f9492290a3ccb6c6984433fb0986a4ad5643fb",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 223.048,
|
||||
"tokens_total": 899893,
|
||||
"audited_at": "2026-06-30T10:42:09Z",
|
||||
"audit_notes": "Clean pass. Agent used all profile resources without reading product code, existing solution, prior store, or adjacent attempt; raw-plan path succeeded on first attempt."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-040.report.json",
|
||||
"report_sha256": "fc0a9d519bc32bc9721b394bb2f4c3b3d9773996f89aed77a7932351040e51a2",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "none",
|
||||
"trial_index": 40,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 261.703,
|
||||
"tokens_total": 1211918,
|
||||
"audited_at": "2026-06-30T11:43:50Z",
|
||||
"audit_notes": "Invalid for clean benchmark. Product path completed through draft authoring and run output showed before.clicked=false and after.clicked=true, but there was a disallowed repository-root read."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-041.report.json",
|
||||
"report_sha256": "e4906b96c644040cf740485043b6e816b6d08975458ecb63cb33d20fdfe41abf",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 41,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 160.444,
|
||||
"tokens_total": 658537,
|
||||
"audited_at": "2026-06-30T11:44:09Z",
|
||||
"audit_notes": "Clean browser-click pass. Agent used supplied skills and workspace files, completed through public wf CLI draft/deploy/run path, returned inline report, and evidence shows before.clicked=false and after.clicked=true with no leftover process claim."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_deepseek-v4-flash-free-trial-042.report.json",
|
||||
"report_sha256": "706c3b226ee197649a90524919c4b62f447c15fba6396dcf86ac8f6ef61a2d52",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "all",
|
||||
"trial_index": 42,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 1663.355,
|
||||
"tokens_total": 1088332,
|
||||
"audited_at": "2026-06-30T12:06:41Z",
|
||||
"audit_notes": "Invalid for clean benchmark. Product path completed through draft authoring and run output showed before.clicked=false and after.clicked=true, but there was a repository-root read and a very long recovery path."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-033.report.json",
|
||||
"report_sha256": "ef4feb1fc5f6fe0324e4e74fd1ee19281832c11c1c1087888af4b8b858b3d54b",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "none",
|
||||
"trial_index": 33,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 211.796,
|
||||
"tokens_total": 759769,
|
||||
"audited_at": "2026-06-30T11:45:06Z",
|
||||
"audit_notes": "Browser-click pass with caveat. Product path completed via raw-plan import, deployment, and run; evidence shows before.clicked=false and after.clicked=true. Manifest read is recorded but not treated as answer leakage."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-034.report.json",
|
||||
"report_sha256": "d8e81e671df9cc2cf421118ccdb9139d6bfee9dc307cb312ecb4b50f5a911675",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 34,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 223.292,
|
||||
"tokens_total": 742181,
|
||||
"audited_at": "2026-06-30T11:47:29Z",
|
||||
"audit_notes": "Clean browser-click pass. Agent used supplied skills and workspace files, completed through public wf CLI draft/deploy/run path, and evidence shows before.clicked=false and after.clicked=true."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/browser_click_challenge/results/opencode_mimo-v2.5-free-trial-035.report.json",
|
||||
"report_sha256": "fd52890062cc309f82204e3b022d37a8494ec0a6bde880ad49b12f24bfbcf678",
|
||||
"challenge": "browser_click",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "all",
|
||||
"trial_index": 35,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 199.298,
|
||||
"tokens_total": 674210,
|
||||
"audited_at": "2026-06-30T11:47:45Z",
|
||||
"audit_notes": "Browser-click pass with caveat. Product path completed via raw-plan import, deployment, and run; evidence shows before.clicked=false and after.clicked=true. Challenge README read is recorded as docs/example context, not answer leakage."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-030.report.json",
|
||||
"report_sha256": "fd5f2c4f62c3997e749e233f15c70c1de48c4312ffaffd312e36666028824270",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "none",
|
||||
"trial_index": 30,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 402.334,
|
||||
"tokens_total": 2824556,
|
||||
"audited_at": "2026-06-30T12:39:21Z",
|
||||
"audit_notes": "Clean report-workflow pass. Agent used workspace files only, completed through public wf CLI draft/deploy/run path, and output markdown matched the expected title."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-031.report.json",
|
||||
"report_sha256": "b44fbd5c6cbbc3031a40384fe3ae7b861a363a21f1a69326f0b55a01b8898029",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 31,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 341.685,
|
||||
"tokens_total": 1650195,
|
||||
"audited_at": "2026-06-30T12:43:17Z",
|
||||
"audit_notes": "Invalid for clean benchmark. Product path completed and output was correct, but the agent read the source example directory outside the supplied trial workspace."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_deepseek-v4-flash-free-trial-032.report.json",
|
||||
"report_sha256": "a14f4467e4173b454ce139e3bf3d09d656e90f0ffc31ca66e83a1cc4015e6e8c",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/deepseek-v4-flash-free",
|
||||
"profile": "all",
|
||||
"trial_index": 32,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 217.551,
|
||||
"tokens_total": 844578,
|
||||
"audited_at": "2026-06-30T12:40:00Z",
|
||||
"audit_notes": "Report-workflow pass with caveat. Agent used raw-plan import and completed through public wf CLI. Repo-index read is recorded but not treated as answer leakage."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-029.report.json",
|
||||
"report_sha256": "ce2a905b9b9caca531cc93201c48bd4a85ceb5454321f4dc68da231d937af4b2",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "none",
|
||||
"trial_index": 29,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "invalid",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 470.566,
|
||||
"tokens_total": 2030961,
|
||||
"audited_at": "2026-06-30T12:40:16Z",
|
||||
"audit_notes": "Invalid for clean benchmark. Product path completed, but store/repo reads make the clean evidence unreliable."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-030.report.json",
|
||||
"report_sha256": "b116ef6ba104c2ad0451e7a4412c0b70c121c8ce1e170f596189a6cd66710e47",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "skills",
|
||||
"trial_index": 30,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 160.917,
|
||||
"tokens_total": 521020,
|
||||
"audited_at": "2026-06-30T12:40:35Z",
|
||||
"audit_notes": "Clean report-workflow pass. Agent used supplied skills and workspace files, completed through raw-plan import/deploy/run, and output markdown matched the expected title."
|
||||
},
|
||||
{
|
||||
"wave": 3,
|
||||
"report": "examples/agent_challenges/report_workflow_challenge/results/opencode_mimo-v2.5-free-trial-031.report.json",
|
||||
"report_sha256": "de86f4fec7577f2bfe3d99db8b8b82076d055ee66b13cb4ae8a419b6478c4001",
|
||||
"challenge": "report_workflow",
|
||||
"model": "opencode/mimo-v2.5-free",
|
||||
"profile": "all",
|
||||
"trial_index": 31,
|
||||
"repository_commit": "5e9b76eb946d94838f6d933e0fe99c8d6030072f",
|
||||
"base_prompt_hash": "b94659167706e5e5edc13429e1dd6c4f59a4f226bd9f926a4a4757c91becf160",
|
||||
"manual_outcome": "pass",
|
||||
"task_outcome": "success",
|
||||
"duration_seconds": 144.907,
|
||||
"tokens_total": 459046,
|
||||
"audited_at": "2026-06-30T12:40:53Z",
|
||||
"audit_notes": "Clean report-workflow pass. Agent used supplied skills/docs plus workspace files, completed through raw-plan import/deploy/run, and output markdown matched the expected title."
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
## Audited Agent Challenge Campaign
|
||||
|
||||
The primary campaign contains 36 audited trials: 27 passes, 8 invalid samples, and 1 failure.
|
||||
|
||||
The campaign crosses two challenges, two hosted models, three instruction profiles (`none`, `skills`, and `all`), and three repetitions per cell. The checked cohort snapshot records report hashes, prompt hashes, the repository commit, automatic metrics, and manual-audit outcomes; local raw report files are verified against those hashes when present.
|
||||
|
||||
Because repository snapshots and one prompt rule changed between waves, this is longitudinal engineering evidence, not a controlled model comparison.
|
||||
|
||||
Selection rule: The latest three completed, manually audited trials per challenge, model, and instruction profile as of 2026-06-30.
|
||||
|
||||
| Challenge / model / profile | Pass | Invalid | Fail |
|
||||
| --- | ---: | ---: | ---: |
|
||||
| browser / deepseek / none | 2 | 1 | 0 |
|
||||
| browser / deepseek / skills | 3 | 0 | 0 |
|
||||
| browser / deepseek / all | 2 | 1 | 0 |
|
||||
| browser / mimo / none | 2 | 0 | 1 |
|
||||
| browser / mimo / skills | 2 | 1 | 0 |
|
||||
| browser / mimo / all | 3 | 0 | 0 |
|
||||
| report / deepseek / none | 3 | 0 | 0 |
|
||||
| report / deepseek / skills | 2 | 1 | 0 |
|
||||
| report / deepseek / all | 1 | 2 | 0 |
|
||||
| report / mimo / none | 1 | 2 | 0 |
|
||||
| report / mimo / skills | 3 | 0 | 0 |
|
||||
| report / mimo / all | 3 | 0 | 0 |
|
||||
|
||||
: Audited outcomes by challenge, model, and instruction profile. {#tbl:agent-challenge-outcomes}
|
||||
|
||||
A manual `pass` requires both successful product-path evidence and an acceptable audit trail. `Invalid` means the sample cannot support the clean benchmark claim, commonly because the agent read repository or example material outside its supplied workspace. `Fail` means the challenge contract itself was not established.
|
||||
|
||||
{#fig:agent-challenge-audited-outcomes-by-cell width=95%}
|
||||
|
||||
[@fig:agent-challenge-audited-outcomes-by-cell] reports all three repetitions rather than hiding invalid samples. The profile labels are descriptive; this campaign does not isolate instruction-profile effects.
|
||||
|
||||
{#fig:agent-challenge-automatic-vs-manual-outcomes width=75%}
|
||||
|
||||
[@fig:agent-challenge-automatic-vs-manual-outcomes] shows why the manual layer matters. Seven automatically successful trials were invalid as clean evidence, while three automatically failed reports were accepted after their saved run evidence and report artifacts were manually audited.
|
||||
|
||||
{#fig:agent-challenge-longitudinal-outcomes width=75%}
|
||||
|
||||
The waves in [@fig:agent-challenge-longitudinal-outcomes] are not an improvement curve: product commits, prompt wording, and enforcement changed. They preserve the chronology needed to study those changes.
|
||||
|
||||
{#fig:agent-challenge-duration-and-tokens width=95%}
|
||||
|
||||
[@fig:agent-challenge-duration-and-tokens] separates each challenge and metric into its own panel. Circle and square markers redundantly identify the models without relying on color. Wall-clock duration includes hosted-service latency, and OpenCode token totals include cache-read accounting, so neither axis is a normalized model-efficiency metric.
|
||||
|
||||
### Campaign Limitations
|
||||
|
||||
- The three waves span repository snapshots; they are longitudinal engineering evidence, not a controlled model comparison.
|
||||
- The base prompt changed before wave 3 to require the challenge report inline.
|
||||
- The models were free hosted OpenCode endpoints, so service load and latency were not controlled.
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 275 KiB |
@@ -0,0 +1,152 @@
|
||||
# Diagram Scratchpad
|
||||
|
||||
This file is a working library of Mermaid diagrams for the thesis/report. Keep
|
||||
diagrams here while they are being shaped, then copy stable versions into the
|
||||
final document when the surrounding prose is ready.
|
||||
|
||||
## Main Architecture Spine
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Owner[Workflow Owner] --> Agent[External LLM Agent]
|
||||
Agent --> CLI[wf CLI]
|
||||
CLI --> Transport[JSON-RPC Transport]
|
||||
Transport --> Server[WorkflowServer]
|
||||
Server --> API[Workflow API Surface]
|
||||
API --> Core[Workflow Core]
|
||||
API --> Platform[Artifacts / Deployments / Runs]
|
||||
Server --> Sources[Source Providers]
|
||||
Sources --> Builtins[Platform Sources]
|
||||
Sources --> MCP[MCP Sources]
|
||||
Sources --> Python[Python Sources]
|
||||
```
|
||||
|
||||
## Layered Package Boundary
|
||||
|
||||
> Package ownership view, not a runtime call graph.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
CLI[wf_cli] --> Transport[wf_transport_rpc_http]
|
||||
Transport --> Server[wf_server]
|
||||
Server --> API[wf_api]
|
||||
API --> Artifacts[wf_artifacts]
|
||||
API --> Core[wf_core]
|
||||
API --> Platform[wf_platform]
|
||||
Server --> MCP[wf_sources_mcp]
|
||||
Server --> Python[wf_sources_python]
|
||||
MCP --> Platform
|
||||
Python --> Platform
|
||||
Artifacts --> Platform
|
||||
```
|
||||
|
||||
## Workflow Lifecycle
|
||||
|
||||
> Simplified view. For the detailed platform-domain version with source inventory
|
||||
> and diagnostics, see "Platform Domain" below.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Draft[Draft Workspace] --> ValidateDraft[Draft Validation]
|
||||
ValidateDraft --> Artifact[Workflow Artifact]
|
||||
Artifact --> Deployment[Workflow Deployment]
|
||||
Deployment --> ValidateDeploy[Deployment Validation]
|
||||
ValidateDeploy --> Run[Workflow Run]
|
||||
Run --> Trace[Run Trace]
|
||||
Run --> Inspect[Run Inspect / List]
|
||||
Run --> Resume[Resume If Interrupted]
|
||||
```
|
||||
|
||||
## Source Resolution
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Ref[Logical Source Requirement] --> PlatformCheck{Platform Source?}
|
||||
PlatformCheck -- yes --> Fixed[Fixed Source Id]
|
||||
PlatformCheck -- no --> Binding[Deployment Binding]
|
||||
Binding --> Concrete[Concrete Source]
|
||||
Fixed --> Runtime[Source Runtime / Handler]
|
||||
Concrete --> Runtime
|
||||
Runtime --> Capability[Workflow Capability]
|
||||
```
|
||||
|
||||
## Workflow Core
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Start[Prepare Run] --> ValidateInput[Validate Workflow Input]
|
||||
ValidateInput --> Select[Select Ready Frame]
|
||||
Select --> Step{Step Type}
|
||||
Step --> Node[NodeUse]
|
||||
Step --> Condition[Condition]
|
||||
Step --> Foreach[Foreach]
|
||||
Step --> Subgraph[Subgraph]
|
||||
Step --> Interrupt[Interrupt]
|
||||
Step --> Join[Join / Minimal Done Step]
|
||||
Step --> End[End]
|
||||
Node --> ResolveInput[Resolve Input Bindings]
|
||||
ResolveInput --> Handler[Invoke NodeDef Handler]
|
||||
Handler --> CheckOutcome[Check Declared Outcome]
|
||||
CheckOutcome --> StatePatch[Build Reducer-Aware State Patch]
|
||||
StatePatch --> Trace[Append Trace Frame]
|
||||
Condition --> Trace
|
||||
Foreach --> Trace
|
||||
Subgraph --> Trace
|
||||
Join --> Trace
|
||||
Trace --> Route[Route By Outcome Edge]
|
||||
Interrupt --> Stop[Persist Interrupt Request]
|
||||
End --> Finalize[Project Workflow Output]
|
||||
Route --> Select
|
||||
Stop --> Resume[Resume Payload + Outcome]
|
||||
Resume --> Route
|
||||
```
|
||||
|
||||
## Platform Domain
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Draft[Draft Workspace] --> DraftValidation[Draft Validation]
|
||||
DraftValidation --> Artifact[Workflow Artifact]
|
||||
Artifact --> Deployment[Workflow Deployment]
|
||||
SourceInventory[Source Inventory] --> DeploymentValidation[Deployment Validation]
|
||||
Deployment --> DeploymentValidation
|
||||
DeploymentValidation --> Run[Workflow Run]
|
||||
Run --> RunRecord[Run Record]
|
||||
Run --> Trace[Trace Slice]
|
||||
RunRecord --> Resume[Resume If Interrupted / Stopped]
|
||||
DeploymentValidation --> Diagnostics[Repairable Diagnostics]
|
||||
```
|
||||
|
||||
## Workflow API Lifecycle Cohesion
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Context[WorkflowOperationContext] --> API[WorkflowApi]
|
||||
API --> Caps[Capability API]
|
||||
API --> Drafts[Draft API]
|
||||
API --> Artifacts[Artifact API]
|
||||
API --> Deployments[Deployment API]
|
||||
API --> Runs[Run API]
|
||||
Caps --> Specs[WorkflowSpecProvider]
|
||||
Drafts --> DraftStore[Draft Store]
|
||||
Artifacts --> ArtifactStore[Artifact Store]
|
||||
Deployments --> ArtifactStore
|
||||
Runs --> RunStore[Run Store]
|
||||
Runs --> Runtime[WorkflowRuntimeRunner]
|
||||
Runs --> Live[Optional Live Source Checker]
|
||||
```
|
||||
|
||||
## Source Provider Boundary
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Config[Workflow Config Sources] --> Server[WorkflowServer Composition]
|
||||
Server --> Builtin[Platform Sources]
|
||||
Server --> MCP[MCP Source Provider]
|
||||
Server --> Python[Python Source Provider]
|
||||
Builtin --> Inventory[CapabilitySource Inventory]
|
||||
MCP --> Inventory
|
||||
Python --> Inventory
|
||||
Inventory --> API[Workflow API Surface]
|
||||
API --> Runtime[Workflow Runtime]
|
||||
```
|
||||
@@ -0,0 +1,14 @@
|
||||
# Thesis Evidence Index
|
||||
|
||||
The claim-to-evidence map now lives inline in
|
||||
[`system-design-implementation.md`](system-design-implementation.md), Appendix B:
|
||||
Evidence Index.
|
||||
|
||||
This file remains as a stable pointer for older roadmap and project-map links.
|
||||
|
||||
For the external-agent challenge evaluation workflow, including trial profiles,
|
||||
manual audits, and report interpretation, see
|
||||
[`../runbooks/agent-challenge-evaluation.md`](../runbooks/agent-challenge-evaluation.md).
|
||||
The primary checked cohort and generated aggregate are
|
||||
[`agent-challenge-cohort.json`](agent-challenge-cohort.json) and
|
||||
[`agent-challenge-results.md`](agent-challenge-results.md).
|
||||
@@ -0,0 +1,23 @@
|
||||
-- Keep one Markdown image reference while selecting print-native PDFs for XeLaTeX.
|
||||
local thesis_figure_format = "svg"
|
||||
|
||||
local function read_metadata(meta)
|
||||
if meta.thesisFigureFormat then
|
||||
thesis_figure_format = pandoc.utils.stringify(meta.thesisFigureFormat)
|
||||
end
|
||||
return meta
|
||||
end
|
||||
|
||||
local function select_figure_format(image)
|
||||
if thesis_figure_format == "pdf"
|
||||
and image.src:match("^figures/.*%.svg$") then
|
||||
image.src = image.src:gsub("%.svg$", ".pdf")
|
||||
end
|
||||
return image
|
||||
end
|
||||
|
||||
-- Separate passes guarantee metadata is available before image traversal.
|
||||
return {
|
||||
{ Meta = read_metadata },
|
||||
{ Image = select_figure_format },
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
|
After Width: | Height: | Size: 77 KiB |
File diff suppressed because it is too large
Load Diff
|
After Width: | Height: | Size: 40 KiB |
File diff suppressed because it is too large
Load Diff
|
After Width: | Height: | Size: 112 KiB |
File diff suppressed because it is too large
Load Diff
|
After Width: | Height: | Size: 40 KiB |
@@ -0,0 +1,106 @@
|
||||
# goal:
|
||||
# use pandoc -M to change the key: diagram:engine:mermaid:outputFormat to svg or pdf if output is html or pdf.
|
||||
# use the script at stuff/pandoc-diagram.ps1 to set the env vars and pass the filter to pandoc.
|
||||
param(
|
||||
[string]$type,
|
||||
[Parameter(ValueFromRemainingArguments = $true)]
|
||||
[string[]]$RemainingArgs
|
||||
)
|
||||
function New-PandocDiagramMetadata([string] $outputFormat) {
|
||||
return @{
|
||||
"diagram" = @{
|
||||
"engine" = @{
|
||||
"mermaid" = @{
|
||||
"outputFormat" = "$outputFormat"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#
|
||||
if ($type -eq "html") {
|
||||
$outputFormat = "svg"
|
||||
}
|
||||
elseif ($type -eq "pdf") {
|
||||
$outputFormat = "pdf"
|
||||
}
|
||||
else {
|
||||
Write-Error "Unsupported type: $type. Supported types are: html, pdf."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$metadata = New-PandocDiagramMetadata $outputFormat | ConvertTo-Json -Depth 10
|
||||
$pandoc_diagram = Join-Path $PSScriptRoot "../../stuff/pandoc-diagram.ps1"
|
||||
if (-not (Test-Path $pandoc_diagram)) {
|
||||
Write-Error "pandoc diagram wrapper not found at $pandoc_diagram. Make sure it exists, then rerun this command."
|
||||
exit 1
|
||||
}
|
||||
. $pandoc_diagram
|
||||
|
||||
$pandoc_crossref = Get-Command pandoc-crossref -ErrorAction SilentlyContinue
|
||||
if (-not $pandoc_crossref) {
|
||||
Write-Error "pandoc-crossref is required for Figure cross-references. Install it, then rerun this command."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$diagram_filter = Join-Path $PSScriptRoot "../../stuff/diagram.lua"
|
||||
if (-not (Test-Path $diagram_filter)) {
|
||||
Write-Error "diagram.lua filter not found at $diagram_filter. Make sure it exists, then rerun this command."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$figure_format_filter = Join-Path $PSScriptRoot "figure-format.lua"
|
||||
if (-not (Test-Path $figure_format_filter)) {
|
||||
Write-Error "figure-format.lua filter not found at $figure_format_filter. Make sure it exists, then rerun this command."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$include_markdown_filter = Join-Path $PSScriptRoot "include-markdown.lua"
|
||||
if (-not (Test-Path $include_markdown_filter)) {
|
||||
Write-Error "include-markdown.lua filter not found at $include_markdown_filter. Make sure it exists, then rerun this command."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$agent_results = Join-Path $PSScriptRoot "agent-challenge-results.md"
|
||||
if (-not (Test-Path $agent_results)) {
|
||||
Write-Error "agent-challenge-results.md is missing. Run generate_agent_challenge_evaluation.py first."
|
||||
exit 1
|
||||
}
|
||||
|
||||
$title_pages_header = Join-Path $PSScriptRoot "title-pages.tex"
|
||||
if (-not (Test-Path $title_pages_header)) {
|
||||
Write-Error "title-pages.tex is missing. Make sure the thesis front matter header exists."
|
||||
exit 1
|
||||
}
|
||||
$title_pages_args = @()
|
||||
if ($type -eq "pdf") {
|
||||
$title_pages_args = @("--include-in-header", $title_pages_header)
|
||||
}
|
||||
|
||||
|
||||
$metatempfile = New-TemporaryFile
|
||||
$pandocExitCode = 0
|
||||
|
||||
try {
|
||||
Set-Content -Path $metatempfile -Value $metadata
|
||||
pandoc `
|
||||
--lua-filter $include_markdown_filter `
|
||||
--lua-filter $diagram_filter `
|
||||
--lua-filter $figure_format_filter `
|
||||
--filter=pandoc-crossref `
|
||||
--pdf-engine=xelatex `
|
||||
@title_pages_args `
|
||||
--metadata thesisFigureFormat=$outputFormat `
|
||||
--metadata thesisAgentResults=$agent_results `
|
||||
--metadata-file=$metatempfile `
|
||||
--embed-resources --standalone --citeproc `
|
||||
@RemainingArgs
|
||||
$pandocExitCode = $LASTEXITCODE
|
||||
}
|
||||
finally {
|
||||
Remove-Item $metatempfile
|
||||
}
|
||||
|
||||
if ($pandocExitCode -ne 0) {
|
||||
throw "pandoc failed with exit code $pandocExitCode"
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
THESIS_DIR = Path(__file__).resolve().parent
|
||||
ROOT = THESIS_DIR.parents[1]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from examples.agent_challenges.evaluation import ( # noqa: E402
|
||||
load_evaluation_cohort,
|
||||
render_evaluation_figures,
|
||||
render_evaluation_markdown,
|
||||
)
|
||||
|
||||
|
||||
def _write_text_atomic(path: Path, content: str) -> None:
|
||||
"""Replace generated Markdown without exposing a partially written file."""
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(content, encoding="utf-8", newline="\n")
|
||||
temporary.replace(path)
|
||||
|
||||
|
||||
def generate(*, manifest_path: Path, output_dir: Path) -> tuple[Path, ...]:
|
||||
"""Generate the audited Markdown rollup plus SVG/PDF figure pairs."""
|
||||
cohort = load_evaluation_cohort(manifest_path, repository_root=ROOT)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
markdown_path = output_dir / "agent-challenge-results.md"
|
||||
_write_text_atomic(markdown_path, render_evaluation_markdown(cohort))
|
||||
figure_paths = render_evaluation_figures(cohort, output_dir / "figures")
|
||||
return (markdown_path, *figure_paths)
|
||||
|
||||
|
||||
def _parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Regenerate thesis agent-challenge results and figures."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--manifest",
|
||||
type=Path,
|
||||
default=THESIS_DIR / "agent-challenge-cohort.json",
|
||||
help="Explicit audited cohort manifest.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output-dir",
|
||||
type=Path,
|
||||
default=THESIS_DIR,
|
||||
help="Directory receiving Markdown and figures/ outputs.",
|
||||
)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _parse_args(argv)
|
||||
for path in generate(
|
||||
manifest_path=args.manifest.resolve(), output_dir=args.output_dir.resolve()
|
||||
):
|
||||
print(path.relative_to(ROOT))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,39 @@
|
||||
# Generate both HTML and PDF outputs for one Markdown file.
|
||||
param(
|
||||
[string]$file = $(throw "File is required."),
|
||||
[Parameter(ValueFromRemainingArguments = $true)]
|
||||
[string[]]$RemainingArgs = @()
|
||||
)
|
||||
$DebugPreference = "Continue"
|
||||
$ErrorActionPreference = "Stop"
|
||||
|
||||
# file exists?
|
||||
if (-not (Test-Path $file)) {
|
||||
Write-Error "File not found: $file"
|
||||
exit 1
|
||||
}
|
||||
|
||||
$file = (Resolve-Path $file).Path
|
||||
$resourcePath = [System.IO.Path]::GetDirectoryName($file)
|
||||
|
||||
$evaluationGenerator = Join-Path $PSScriptRoot "generate_agent_challenge_evaluation.py"
|
||||
& uv run python $evaluationGenerator
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
throw "agent challenge evaluation generation failed with exit code $LASTEXITCODE"
|
||||
}
|
||||
|
||||
# name without extension
|
||||
function Get-OutputFilenames([string] $file, [string] $type) {
|
||||
$parentdir = [System.IO.Path]::GetDirectoryName($file)
|
||||
if (-not $parentdir) { $parentdir = '.' }
|
||||
$filename = [System.IO.Path]::GetFileNameWithoutExtension($file)
|
||||
$newfilename = "${filename}.${type}"
|
||||
return Join-Path $parentdir $newfilename
|
||||
}
|
||||
|
||||
Write-Host "Generating HTML and PDF for $file..."
|
||||
Write-Debug "Remaining args: $($RemainingArgs | Format-List)"
|
||||
|
||||
& $PSScriptRoot\generate.ps1 -type html -- -i $file -o (Get-OutputFilenames $file "html") --resource-path $resourcePath @RemainingArgs
|
||||
|
||||
& $PSScriptRoot\generate.ps1 -type pdf -- -i $file -o (Get-OutputFilenames $file "pdf") --resource-path $resourcePath @RemainingArgs
|
||||
@@ -0,0 +1,32 @@
|
||||
-- Replace explicit include markers with parsed Markdown before cross-reference processing.
|
||||
local agent_results_path = nil
|
||||
|
||||
local function read_metadata(meta)
|
||||
if meta.thesisAgentResults then
|
||||
agent_results_path = pandoc.utils.stringify(meta.thesisAgentResults)
|
||||
end
|
||||
return meta
|
||||
end
|
||||
|
||||
local function include_results(div)
|
||||
if div.identifier ~= "include-agent-challenge-results" then
|
||||
return nil
|
||||
end
|
||||
if not agent_results_path or agent_results_path == "" then
|
||||
error("thesisAgentResults metadata is required for the evaluation include")
|
||||
end
|
||||
|
||||
local file, open_error = io.open(agent_results_path, "r")
|
||||
if not file then
|
||||
error("cannot open agent challenge results: " .. tostring(open_error))
|
||||
end
|
||||
local content = file:read("*a")
|
||||
file:close()
|
||||
return pandoc.read(content, "markdown").blocks
|
||||
end
|
||||
|
||||
-- Separate passes guarantee metadata is available before block traversal.
|
||||
return {
|
||||
{ Meta = read_metadata },
|
||||
{ Div = include_results },
|
||||
}
|
||||
@@ -0,0 +1,107 @@
|
||||
@online{mcp-tools-2025,
|
||||
title = {Tools},
|
||||
author = {{Model Context Protocol}},
|
||||
year = {2025},
|
||||
url = {https://modelcontextprotocol.io/specification/2025-06-18/server/tools},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{mcp-lifecycle-2025,
|
||||
title = {Lifecycle},
|
||||
author = {{Model Context Protocol}},
|
||||
year = {2025},
|
||||
url = {https://modelcontextprotocol.io/specification/2025-03-26/basic/lifecycle},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{langgraph-overview-2026,
|
||||
title = {LangGraph Overview},
|
||||
author = {{LangChain}},
|
||||
year = {2026},
|
||||
url = {https://docs.langchain.com/oss/python/langgraph/overview},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{langgraph-persistence-2026,
|
||||
title = {Persistence},
|
||||
author = {{LangChain}},
|
||||
year = {2026},
|
||||
url = {https://docs.langchain.com/oss/python/langgraph/persistence},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{zapier-operating-constraints,
|
||||
title = {Zapier Operating Constraints},
|
||||
author = {{Zapier}},
|
||||
year = {2026},
|
||||
url = {https://docs.zapier.com/integrations/build/operating-constraints},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{zapier-zap-limits,
|
||||
title = {Zap Limits},
|
||||
author = {{Zapier}},
|
||||
year = {2026},
|
||||
url = {https://help.zapier.com/hc/en-us/articles/8496181445261-Zap-limits},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{nist-agent-cheating-2025,
|
||||
title = {Examples of Cheating in CAISI's Agent Evaluations},
|
||||
author = {{National Institute of Standards and Technology}},
|
||||
year = {2025},
|
||||
url = {https://www.nist.gov/caisi/cheating-ai-agent-evaluations/2-examples-cheating-caisis-agent-evaluations},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{openai-swebench-audit-2026,
|
||||
title = {Why SWE-bench Verified No Longer Measures Frontier Coding Capabilities},
|
||||
author = {{OpenAI}},
|
||||
year = {2026},
|
||||
url = {https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@article{react-2022,
|
||||
title = {ReAct: Synergizing Reasoning and Acting in Language Models},
|
||||
author = {Yao, Shunyu and Zhao, Jeffrey and Yu, Dian and Du, Nan and Shafran, Izhak and Narasimhan, Karthik and Cao, Yuan},
|
||||
year = {2022},
|
||||
journal = {arXiv preprint arXiv:2210.03629},
|
||||
url = {https://arxiv.org/abs/2210.03629},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@article{toolformer-2023,
|
||||
title = {Toolformer: Language Models Can Teach Themselves to Use Tools},
|
||||
author = {Schick, Timo and Dwivedi-Yu, Jane and Dessi, Roberto and Raileanu, Roberta and Lomeli, Maria and Zettlemoyer, Luke and Cancedda, Nicola and Scialom, Thomas},
|
||||
year = {2023},
|
||||
journal = {arXiv preprint arXiv:2302.04761},
|
||||
url = {https://arxiv.org/abs/2302.04761},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@online{openai-structured-outputs-2024,
|
||||
title = {Introducing Structured Outputs in the API},
|
||||
author = {{OpenAI}},
|
||||
year = {2024},
|
||||
url = {https://openai.com/index/introducing-structured-outputs-in-the-api/},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@article{auditable-agents-2026,
|
||||
title = {Auditable Agents},
|
||||
author = {Nian, Yi and Yuan, Aojie and Zhang, Haiyue and Li, Jiate and Zhao, Yue},
|
||||
year = {2026},
|
||||
journal = {arXiv preprint arXiv:2604.05485},
|
||||
url = {https://arxiv.org/abs/2604.05485},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
|
||||
@article{audit-trails-llm-2026,
|
||||
title = {Audit Trails for Accountability in Large Language Models},
|
||||
author = {Ojewale, Victor and Suresh, Harini and Venkatasubramanian, Suresh},
|
||||
year = {2026},
|
||||
journal = {arXiv preprint arXiv:2601.20727},
|
||||
url = {https://arxiv.org/abs/2601.20727},
|
||||
urldate = {2026-06-16}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,688 @@
|
||||
# Thesis Outline
|
||||
|
||||
> **Note:** All sections below have been transferred to the formal Big Doc at
|
||||
> [`system-design-implementation.md`](system-design-implementation.md). This
|
||||
> outline is now historical scaffolding; maintain the Big Doc directly.
|
||||
|
||||
This is a writing scaffold for a thesis/report about the workflow platform. It
|
||||
should guide the argument; it is not a changelog.
|
||||
|
||||
The final document should read like a formal system design and implementation
|
||||
report. Keep detailed command transcripts and long CLI outputs in appendices or
|
||||
linked runbooks; inline chapters should show only the commands/results needed to
|
||||
support the argument.
|
||||
|
||||
## Core Argument
|
||||
|
||||
External LLM agents are useful workflow authors and operators, but durable
|
||||
workspace automation needs a typed execution substrate. Artifacts, deployments,
|
||||
source bindings, validation, runs, traces, and resumability should be owned by
|
||||
the platform, not improvised through raw tool-call loops.
|
||||
|
||||
The short version:
|
||||
|
||||
> The LLM plans. The runtime executes. Source providers expose capabilities.
|
||||
> Stores preserve durable workflow state.
|
||||
|
||||
## Research Question
|
||||
|
||||
Primary research question:
|
||||
|
||||
> How can an AI-agent-facing workflow platform represent, validate, execute, and
|
||||
> persist reusable workspace automations while keeping planning separate from
|
||||
> deterministic execution?
|
||||
|
||||
Product motivation:
|
||||
|
||||
> How can external AI agents help workspace operators create reusable
|
||||
> automations without requiring them to write scripts, while preserving
|
||||
> validation, inspection, and durable execution?
|
||||
|
||||
## Main Contribution
|
||||
|
||||
The thesis contribution is a platform architecture, not a new foundation model:
|
||||
|
||||
1. typed workflow artifact/deployment/run lifecycle
|
||||
2. source-provider boundary for MCP, Python, and future OpenAPI sources
|
||||
3. durable server/API/CLI surface that external agents can drive
|
||||
4. validation and inspection mechanisms that reduce planner trial-and-error
|
||||
5. next-action guidance that points an agent toward useful lifecycle operations
|
||||
without replacing validation
|
||||
|
||||
## Evidence Strategy
|
||||
|
||||
Use one representative workspace case study as the narrative spine: a
|
||||
document/report preparation workflow backed by local fixtures and trusted Python
|
||||
sources. The case study should read or receive a small document-like input,
|
||||
extract structured information such as action items or sections, normalize the
|
||||
result, and produce report-shaped JSON or Markdown. MCP-backed sources can be
|
||||
secondary evidence, but the thesis-critical demo should not depend on remote
|
||||
MCP auth, quota, or provider availability.
|
||||
|
||||
The case study should exist as a runnable example, not only prose. Target shape:
|
||||
`examples/report_workflow/ops.py`, `input.md`, `wf.config.json`, a short
|
||||
`README.md`, and commands for config validation, server startup, capability
|
||||
calls, draft/artifact/deployment creation, run, inspect, and trace.
|
||||
The runnable evidence bundle for this case study lives at
|
||||
`examples/report_workflow/`.
|
||||
|
||||
|
||||
Keep the thesis-critical path deterministic. Do not require an LLM call inside
|
||||
the case-study workflow. LLM nodes can be discussed as future work or an
|
||||
optional variant, but the evidence path should be reproducible without model
|
||||
credentials, cost, or output variance.
|
||||
|
||||
Prefer typed report JSON as the primary output, with Markdown rendering optional
|
||||
later. A useful output contract is:
|
||||
|
||||
```json
|
||||
{
|
||||
"title": "Weekly Project Update",
|
||||
"summary": "...",
|
||||
"action_items": [
|
||||
{"owner": "Alice", "task": "Prepare demo config", "due": "Friday"}
|
||||
],
|
||||
"risks": ["..."],
|
||||
"followups": ["..."]
|
||||
}
|
||||
```
|
||||
|
||||
This makes schemas and validation visible in the case study.
|
||||
|
||||
Use CLI lifecycle commands for the thesis narrative because they demonstrate the
|
||||
agent-operable surface. Tests may seed `RawWorkflowPlan` objects directly when
|
||||
that makes assertions tighter, but the case-study runbook should show config
|
||||
validation, server startup, capability inspection/call, draft or artifact
|
||||
creation, deployment save/validate, run start, run inspect, run trace, and run
|
||||
list.
|
||||
|
||||
Support that case study with platform evidence: automated tests, CLI/server
|
||||
smoke runs, source-provider examples, run persistence/resume checks, source
|
||||
drift producing unrunnable deployments, stateful MCP session reuse, and a small
|
||||
failed-attempt case study showing how validation/diagnostics reduced blind
|
||||
retries.
|
||||
|
||||
Do not frame the evaluation as a broad user study unless that study actually
|
||||
exists. The evidence claim is that the prototype demonstrates the architecture
|
||||
and workflow lifecycle under controlled examples.
|
||||
|
||||
Do not make runtime throughput a central claim. The project targets planner
|
||||
efficiency and operational clarity: fewer blind retries through typed contracts,
|
||||
validation, diagnostics, compact outputs, and traces. Performance optimization
|
||||
is future work unless backed by explicit measurements.
|
||||
|
||||
## Working Title
|
||||
|
||||
Safer current title:
|
||||
|
||||
> Design and Implementation of lda.chat: Infrastructure for AI Agents to Author
|
||||
> and Execute Workspace Workflows
|
||||
|
||||
Aspirational product title:
|
||||
|
||||
> Design and Implementation of lda.chat: An AI Agent Platform for Authoring and
|
||||
> Executing Workspace Workflows
|
||||
|
||||
Use the safer title if the thesis must describe exactly what exists today:
|
||||
external agents such as Claude Desktop and OpenCode can drive the platform, but
|
||||
`lda.chat` does not yet bundle its own autonomous agent brain. Use the
|
||||
aspirational title only if the thesis explicitly frames `lda.chat` as the
|
||||
platform intended to host or serve AI agents, not as a completed built-in agent.
|
||||
|
||||
## 1. Problem Statement
|
||||
|
||||
Current agent/tool systems often let the LLM directly orchestrate side effects
|
||||
through ad hoc tool calls. This creates practical problems:
|
||||
|
||||
- weak validation before execution
|
||||
- poor resumability after interruption or process restart
|
||||
- hard-to-audit tool-call traces
|
||||
- limited reuse of successful procedures
|
||||
- unclear boundaries between planning, execution, and provider-specific state
|
||||
|
||||
The thesis should frame the platform as a response to those pressures.
|
||||
|
||||
The automation target is reusable workspace procedures, not arbitrary office
|
||||
work end-to-end. Examples include document transformation, data collection,
|
||||
tool/API calls, report preparation, monitoring checks, and scheduled workspace
|
||||
operations.
|
||||
|
||||
## 2. Thesis
|
||||
|
||||
The proposed model is a prototype platform for AI-assisted work:
|
||||
|
||||
- workflows are typed graphs
|
||||
- source capabilities are exposed through explicit contracts
|
||||
- deployments bind logical source requirements to concrete sources
|
||||
- runs are durable stopped records
|
||||
- clients interact through stable APIs/transports rather than direct runtime
|
||||
internals
|
||||
|
||||
Natural language belongs in the authoring loop: a workspace operator can express
|
||||
intent to an external LLM agent, but the reusable output should be a typed
|
||||
workflow artifact and deployment rather than an opaque prompt transcript.
|
||||
|
||||
Be explicit about actors:
|
||||
|
||||
- the workflow owner wants to see useful workflow runs and outputs
|
||||
- the external LLM agent may be the one driving CLI/API operations
|
||||
- the developer/operator configures sources, secrets, server processes, and
|
||||
trusted Python code
|
||||
|
||||
The CLI should be described as agent-operable first and human-usable second. Its
|
||||
structured output, status/inspect/list commands, validation commands, compact
|
||||
summaries, and guarded destructive actions make it a practical surface for
|
||||
external agents.
|
||||
|
||||
This is not a claim that LLMs cannot operate tools. It is a claim that durable,
|
||||
inspectable, reusable work benefits from a separate execution substrate.
|
||||
|
||||
Use “prototype platform” deliberately. The implementation proves the core
|
||||
lifecycle and architecture, but it is not yet a finished product with scheduling,
|
||||
visual workflow editing, production secrets, general fork/gather, and broad
|
||||
real-world evaluation.
|
||||
|
||||
Next actions and repairable validation failures should be framed as
|
||||
machine-client UX. Human interfaces use buttons and affordances; an agent-facing
|
||||
API needs compact structured hints, stable diagnostic fields, and suggested next
|
||||
operations so an LLM agent can recover without blind probing.
|
||||
|
||||
## 3. Design Goals
|
||||
|
||||
The design goals should be stated early and then revisited in evaluation:
|
||||
|
||||
- deterministic execution
|
||||
- validation-centered lifecycle for LLM-authored workflows
|
||||
- typed inputs, state, outputs, and node payloads
|
||||
- explicit source binding
|
||||
- platform sources with fixed process-provided identities, separate from
|
||||
configured workspace/account sources
|
||||
- scoped workflow portability through artifact requirements and deployment
|
||||
binding contracts
|
||||
- durable artifacts, deployments, and stopped runs
|
||||
- inspectable trace slices
|
||||
- reviewable lifecycle points for drafts, deployments, runs, diagnostics, and
|
||||
guarded destructive actions
|
||||
- resumability after interruption
|
||||
- transport neutrality for CLI, server, and future UI/MCP clients
|
||||
- source-provider extensibility for MCP, Python, OpenAPI, and future families
|
||||
- source-provider correctness, especially for external systems whose tools,
|
||||
resources, prompts, or authentication depend on initialized stateful sessions
|
||||
|
||||
Auth should be described as prototype source-readiness plumbing, not a completed
|
||||
production secret system. The implementation includes typed auth records, OAuth
|
||||
refresh-token support, source auth diagnostics, and MCP auth binding. This is
|
||||
enough to show how credentials participate in source readiness and diagnostics,
|
||||
but encrypted-at-rest storage, production secret-manager integration, and broad
|
||||
provider verification remain future work.
|
||||
|
||||
## 4. Positioning And Related Systems
|
||||
|
||||
Keep this section short and category-oriented. The goal is to position the
|
||||
system, not to claim full feature parity with mature platforms.
|
||||
|
||||
Compare against:
|
||||
|
||||
- direct LLM tool orchestration: flexible but weak durability and validation
|
||||
- generated scripts: simple and maintainable for some tasks, but lifecycle
|
||||
affordances are manual
|
||||
- Zapier/RPA/workflow automation platforms: mature integrations and scheduling,
|
||||
but less agent-native typed authoring/repair flow in this prototype's terms
|
||||
- LangGraph-style agent graphs/durable agents: adjacent durability ideas, but a
|
||||
different emphasis from source-provider-backed reusable workspace workflows
|
||||
- MCP: useful protocol for tools/resources/prompts, but not itself the workflow
|
||||
artifact/deployment/run lifecycle
|
||||
|
||||
## 5. Architecture
|
||||
|
||||
Explain the active package boundaries:
|
||||
|
||||
```text
|
||||
wf_cli
|
||||
-> wf_transport_rpc_http
|
||||
-> wf_server
|
||||
-> wf_api
|
||||
-> wf_core / wf_artifacts / wf_sources_*
|
||||
```
|
||||
|
||||
Important layers:
|
||||
|
||||
- `wf_core`: deterministic workflow kernel
|
||||
- `wf_authoring`: authoring support for `NodeSpec`, drafts, wrappers, API
|
||||
surfaces, source providers, and MCP/admin tools
|
||||
- `wf_api`: application surface for capabilities, drafts, artifacts,
|
||||
deployments, runs, and admin/source operations
|
||||
- `wf_server`: durable server composition boundary
|
||||
- `wf_transport_rpc_http`: JSON-RPC-over-HTTP transport
|
||||
- `wf_sources_mcp`: MCP upstream source implementation and persistent runtime
|
||||
- `wf_sources_python`: trusted in-process Python source loading
|
||||
- `wf_mcp`: legacy/special-purpose MCP compatibility package
|
||||
|
||||
Use package names as implementation evidence, not as the main argument. The
|
||||
conceptual architecture should lead: workflow core, platform domain, workflow
|
||||
API surface, server/transport composition, and source providers. Package names
|
||||
then show how those concepts were implemented in this codebase.
|
||||
|
||||
The thesis should explain why the old “everything in MCP” shape was split:
|
||||
transport, source provider, workflow API, and runtime concerns are different.
|
||||
|
||||
Architecture spine:
|
||||
|
||||
- workflow core: deterministic execution semantics for graph, state, outcomes,
|
||||
trace, and resume rules
|
||||
- platform domain: artifacts, deployments, runs, stores, sources, binding
|
||||
contracts, validation, and admin concepts
|
||||
- workflow API surface: lifecycle operations exposed to clients
|
||||
- server/transport composition: concrete stores, sources, runtimes, and
|
||||
communication mechanisms
|
||||
|
||||
JSON-RPC is an implementation of the Workflow API Surface. It should not be
|
||||
presented as the product boundary or as the place where workflow semantics live.
|
||||
`wf_server` is composition: it assembles concrete stores, sources, runtimes, and
|
||||
admin surfaces into a long-lived service. It should not own workflow semantics.
|
||||
`wf_authoring` is support infrastructure used by drafts, API surfaces, source
|
||||
providers, and MCP tools; it is not a fifth runtime/product layer.
|
||||
If MCP is discussed, distinguish upstream MCP sources from a future client-facing
|
||||
MCP frontend. The former exists as a source family; the latter should not be
|
||||
claimed as a completed clean platform surface.
|
||||
|
||||
MCP is an important source-provider case study, not the product identity. It
|
||||
demonstrates why source-provider correctness matters: a source may require
|
||||
persistent sessions, auth context, catalog refresh, resources, and prompt
|
||||
inventory. The platform treats MCP as one source family behind the workflow
|
||||
boundary, not as the whole architecture.
|
||||
|
||||
## 6. Workflow Model
|
||||
|
||||
Describe workflows as typed graphs:
|
||||
|
||||
- `input_schema`: validates run input
|
||||
- `state_schema`: defines workflow memory and reducer behavior
|
||||
- `output_schema`: defines final result shape
|
||||
- `NodeUse`: invokes named `NodeSpec`
|
||||
- edges: route by declared outcomes
|
||||
- reducers: merge concurrent or repeated writes safely
|
||||
- interrupts: represent typed external input points
|
||||
- subgraphs: compose workflows as nodes
|
||||
|
||||
Separate lifecycle objects:
|
||||
|
||||
- draft workspace: mutable authoring state for agent/user iteration
|
||||
- workflow artifact: immutable versioned workflow definition
|
||||
- deployment: binding contract from artifact version to concrete source/runtime
|
||||
context
|
||||
- run: execution record with status, diagnostics, output, trace, and resumable
|
||||
stopped/interrupted state where applicable
|
||||
|
||||
Key distinction:
|
||||
|
||||
- outcome controls routing
|
||||
- output carries business data
|
||||
|
||||
The graph model improves the safety posture by making automation structure
|
||||
explicit. Node contracts, source requirements, state writes, outcomes,
|
||||
validation gates, and trace records are visible before and after execution. It
|
||||
does not guarantee safe behavior from provider code, credentials, or external
|
||||
side effects.
|
||||
|
||||
Use generated scripts as a serious baseline, not a strawman. Scripts can be
|
||||
simpler and maintainable for many tasks. The platform argument is that reusable
|
||||
workspace automation benefits from lifecycle affordances that scripts do not
|
||||
automatically provide: typed validation, source binding, artifact/deployment
|
||||
separation, run records, resumability, trace inspection, and repairable
|
||||
diagnostics.
|
||||
|
||||
Code ends at the source-provider boundary. A workflow can call trusted Python,
|
||||
Playwright, API, MCP, or future LLM capabilities, but those should appear as
|
||||
typed source capabilities. The workflow itself remains an orchestration artifact,
|
||||
not an embedded code blob.
|
||||
|
||||
Durability is a contract over time, not just storage. Artifacts, deployments,
|
||||
bindings, run records, and traces preserve workflow intent. Validation against
|
||||
the current source catalog determines whether that intent is still runnable. If
|
||||
a source changes incompatibly and a deployment becomes `unrunnable`, the system
|
||||
has preserved the contract instead of silently drifting.
|
||||
|
||||
Trace claims should be grounded in the current code: run summaries expose
|
||||
`trace_count`, and clients can request caller-bounded trace slices for debugging.
|
||||
Do not overstate this as production observability, distributed tracing, metrics,
|
||||
or OpenTelemetry support.
|
||||
|
||||
## 7. Source Model
|
||||
|
||||
The common boundary is `CapabilitySource`.
|
||||
|
||||
Source families today:
|
||||
|
||||
| Source | Kind | Role |
|
||||
| --- | --- | --- |
|
||||
| `wf.std` | `system` | built-in workflow nodes and reducers |
|
||||
| `wf.recipes` | `system` | first-party workflow recipes |
|
||||
| MCP sources | `connection` | upstream MCP tools/resources/prompts |
|
||||
| Python sources | `python` | trusted project-local `NodeSpec` registries |
|
||||
|
||||
The thesis should stress that the runtime does not care where a `NodeSpec` came
|
||||
from. Source-specific behavior belongs in provider packages and server
|
||||
composition.
|
||||
|
||||
Use precise vocabulary:
|
||||
|
||||
- tools are provider-native operations, such as MCP tools
|
||||
- workflow capabilities are `NodeSpec` contracts callable from graphs
|
||||
- resources are source-owned addressable content; a URI is meaningful only with
|
||||
its owning source
|
||||
- prompts are source-owned prompt/template inventory; rendering may be stateful
|
||||
|
||||
Platform sources such as `wf.std` and `wf.source` are process-provided and do
|
||||
not require deployment self-bindings. Configured sources such as MCP, Python,
|
||||
and future OpenAPI sources remain explicit server/operator choices.
|
||||
|
||||
`wf.source.read_resource` is the current explicit dereference helper: workflows
|
||||
pass inert resource refs by value, then the helper resolves the logical source
|
||||
through runtime/platform context and returns bounded text. Prompt rendering is
|
||||
deliberately not a workflow helper yet; keep it in future work unless the thesis
|
||||
adds a concrete graph use case, argument schema, and bounded output policy.
|
||||
|
||||
MCP should be presented as one source family and a useful stress test for
|
||||
source-provider correctness, not as the platform identity.
|
||||
|
||||
Current provider seam:
|
||||
|
||||
```python
|
||||
class WorkflowSourceProvider(Protocol):
|
||||
def load_sources(self) -> Mapping[str, CapabilitySource]: ...
|
||||
```
|
||||
|
||||
This seam is intentionally narrow: it covers static inventory, not runtime
|
||||
pools, admin/apply, auth, or live health checks.
|
||||
|
||||
For MCP, source-provider correctness includes stateful runtime behavior. A
|
||||
workflow capability call should not silently turn a stateful external provider
|
||||
into a fresh one-off client call when provider state is part of correctness.
|
||||
|
||||
## 8. Implementation Vertical Slice
|
||||
|
||||
Use the working product path as evidence:
|
||||
|
||||
```text
|
||||
wf config validate
|
||||
-> wf-rpc-server --config
|
||||
-> wf status
|
||||
-> wf source list
|
||||
-> wf source resources / prompts
|
||||
-> wf cap list / inspect / call
|
||||
-> wf draft create --capability
|
||||
-> wf draft save
|
||||
-> wf deploy save / validate
|
||||
-> wf run start
|
||||
-> wf run inspect / trace / list
|
||||
```
|
||||
|
||||
A strong demonstration is the Python source flow:
|
||||
|
||||
1. write `ops.py` with `@node`
|
||||
2. configure `kind: "python"` source
|
||||
3. validate config
|
||||
4. start server
|
||||
5. call `local.ops.echo`
|
||||
6. create draft/artifact/deployment
|
||||
7. run deployment successfully
|
||||
|
||||
This shows the source abstraction is not MCP-only.
|
||||
|
||||
Use diagrams as first-class explanation, especially Mermaid diagrams that can be
|
||||
rendered by the existing document generation flow. Each major part should have
|
||||
at least one diagram that explains its role and boundaries before code excerpts
|
||||
or package names appear. Prefer diagrams for:
|
||||
|
||||
- main architecture spine:
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
Owner[Workflow Owner] --> Agent[External LLM Agent]
|
||||
Agent --> CLI[wf CLI]
|
||||
CLI --> Transport[JSON-RPC Transport]
|
||||
Transport --> Server[WorkflowServer]
|
||||
Server --> API[Workflow API Surface]
|
||||
API --> Core[Workflow Core]
|
||||
API --> Platform[Artifacts / Deployments / Runs]
|
||||
Server --> Sources[Source Providers]
|
||||
Sources --> Builtins[Platform Sources]
|
||||
Sources --> MCP[MCP Sources]
|
||||
Sources --> Python[Python Sources]
|
||||
```
|
||||
|
||||
- layer architecture: CLI/transport/server/API/core/source providers
|
||||
- workflow core: schemas, nodes, outcomes, reducers, trace, interrupts/resume
|
||||
- platform domain: draft workspaces, artifacts, deployments, source inventory,
|
||||
validation diagnostics, run records
|
||||
- lifecycle: draft -> artifact -> deployment -> run -> trace/list/resume
|
||||
- source resolution: logical source -> deployment binding/platform context ->
|
||||
concrete source/runtime
|
||||
- source-provider comparison: built-in, MCP, Python, future OpenAPI
|
||||
- runtime call path: cap call/run -> Workflow API -> source runtime/client
|
||||
|
||||
Use source excerpts sparingly. Include small snippets for key seams such as the
|
||||
artifact/deployment/run lifecycle shape, `CapabilitySource`, the
|
||||
`WorkflowSourceProvider` protocol, a compact Python source `@node` example, and
|
||||
selected CLI/JSON responses. Avoid long file listings; the implementation
|
||||
chapter should explain the architecture, not reproduce the repository.
|
||||
|
||||
Frame Python sources as trusted developer extensibility. They are useful because
|
||||
project-local code can become typed workflow capabilities quickly, but they are
|
||||
not sandboxed non-programmer plugins yet.
|
||||
|
||||
## 9. Evaluation
|
||||
|
||||
Evaluation should use concrete evidence:
|
||||
|
||||
- automated tests for workflow core, API, transports, source providers, and CLI
|
||||
- live smoke test against `wf-rpc-server`
|
||||
- durable run/resume tests
|
||||
- stateful MCP session reuse tests
|
||||
- MCP source-provider correctness tests covering tools, resources, prompts, and
|
||||
session reuse through the same server path
|
||||
- Python source workflow-run integration test
|
||||
- bounded source inventory and resource-read tests, especially to avoid raw
|
||||
provider payload spam
|
||||
- OAuth refresh-token/auth-binding tests for HTTP MCP sources, with Google Drive
|
||||
MCP treated as manual smoke coverage rather than a regression fixture
|
||||
- config validation catching import/path errors before server startup
|
||||
- planner-efficiency checks: validation, source catalogs, compact output, and
|
||||
inspectable errors should reduce repeated blind LLM attempts
|
||||
- next-action guidance should reduce planner uncertainty before and after
|
||||
validation calls
|
||||
- attempt-count comparison on representative tasks, for example old interaction
|
||||
traces with many failed attempts versus the current structured lifecycle
|
||||
- draft-validation and run-failure analysis, especially cases where old session
|
||||
or source assumptions caused repeated failed runs
|
||||
- source-drift cases where old deployments become unrunnable with diagnostics
|
||||
instead of silently executing against incompatible capabilities
|
||||
- current agent/tool evaluation with a small repeat count. A practical prototype
|
||||
target is five end-to-end attempts where free or commodity LLM agents try to
|
||||
use the CLI/API surface to complete the deterministic case study. Report
|
||||
pass/fail counts, failure categories, and whether diagnostics were actionable;
|
||||
do not present this as a statistical reliability benchmark.
|
||||
|
||||
For the agent/tool evaluation, success means end-to-end completion: the agent
|
||||
creates or selects a valid workflow artifact/deployment and completes a run whose
|
||||
output matches the expected typed report schema and key content. Merely calling
|
||||
a capability, producing a draft, or returning freeform text outside the schema is
|
||||
not success.
|
||||
|
||||
Count failures explicitly. Useful categories include:
|
||||
|
||||
- config/setup failure
|
||||
- source discovery or source binding failure
|
||||
- draft validation failure
|
||||
- deployment validation failure
|
||||
- run failure
|
||||
- output schema/content mismatch
|
||||
- excessive manual intervention
|
||||
- agent gave up or looped without progress
|
||||
|
||||
Track autonomous and assisted success separately. Autonomous success means the
|
||||
agent completes the task with only the initial prompt/runbook. Assisted success
|
||||
means completion after limited documented help such as confirming that the
|
||||
server is running or pointing at the intended config path. Manual artifact edits,
|
||||
code fixes, or changing the expected output should count as failures for
|
||||
autonomous evaluation.
|
||||
|
||||
Avoid vague claims such as “robust” or “production-ready” unless backed by
|
||||
specific checks.
|
||||
|
||||
Evidence package:
|
||||
|
||||
- architecture/code walkthrough tied to the four-layer model
|
||||
- automated tests for lifecycle, validation, source providers, persistence,
|
||||
resume, stateful MCP reuse, and Python source integration
|
||||
- live CLI/server smoke run
|
||||
- before/after failed-attempt case study from old ad-hoc interaction to
|
||||
structured workflow lifecycle
|
||||
- explicit limitations and future work
|
||||
|
||||
Recommended case study:
|
||||
|
||||
- document/report preparation, not an echo demo
|
||||
- deterministic current sources first, such as Python source text transforms
|
||||
- optional/future LLM summarization as a typed source capability, not required
|
||||
- output should be a structured report or Markdown/JSON artifact that a workflow
|
||||
owner would plausibly want
|
||||
- package the example with fixture input, Python source code, workflow config,
|
||||
store/environment setup, and CLI/server commands
|
||||
|
||||
Possible evaluation questions:
|
||||
|
||||
- Can a source capability be discovered, called, saved into a workflow, deployed,
|
||||
and run?
|
||||
- Can an interrupted run survive process restart and resume?
|
||||
- Can the same server be used through CLI and JSON-RPC transport?
|
||||
- Can a new source family be added without changing `wf_core`?
|
||||
- Are large/raw provider payloads bounded in CLI output?
|
||||
- Can platform sources such as `wf.std` be used without self-bindings while
|
||||
configured sources still require explicit bindings?
|
||||
- Can source resources be referenced by logical source and dereferenced only
|
||||
through an explicit bounded helper?
|
||||
- Can an external LLM agent converge on a valid workflow without spending most
|
||||
of the interaction on tool-output spam and trial-and-error?
|
||||
- Does the structured surface reduce failed attempts before success compared to
|
||||
earlier ad-hoc agent/tool interaction traces?
|
||||
- Do `wf draft validate`, deployment validation, and run inspection catch or
|
||||
explain the kinds of issues that previously caused repeated failed runs?
|
||||
- Does deployment validation surface source drift as runnable/unrunnable state
|
||||
with diagnostics rather than silent behavior changes?
|
||||
|
||||
## 9.1 Positioning Against Existing Automation Platforms
|
||||
|
||||
The thesis should discuss the space it fits into through multiple baselines:
|
||||
direct LLM tool use, manual scripts, Zapier-style automation platforms, RPA
|
||||
tools, and workflow engines. The goal is not to claim feature parity with mature
|
||||
products. The goal is to explain the trade-off this prototype explores.
|
||||
|
||||
Zapier and similar platforms are stronger today at:
|
||||
|
||||
- polished non-programmer UI
|
||||
- large integration catalogs
|
||||
- hosted scheduling and triggers
|
||||
- operational maturity
|
||||
|
||||
Manual scripts are powerful and often faster for technical users, so the thesis
|
||||
should not dismiss them. The fair comparison is accessibility and adaptability:
|
||||
how much skill and maintenance effort is required before a workspace operator or
|
||||
external agent can turn a repeated task into a reusable workflow?
|
||||
|
||||
This prototype explores a different center of gravity:
|
||||
|
||||
- external AI agents can drive the authoring/execution lifecycle directly
|
||||
- workflows are typed graphs with explicit schemas and source bindings
|
||||
- local Python, MCP, and future OpenAPI sources can share one workflow surface
|
||||
- runs, traces, artifacts, and deployments are first-class inspectable records
|
||||
|
||||
Use the comparison to position the work, not as a claim that the prototype
|
||||
outperforms existing automation products.
|
||||
|
||||
## 10. Limitations
|
||||
|
||||
State limitations explicitly:
|
||||
|
||||
- Python sources are trusted in-process code; no sandbox yet.
|
||||
- Python sources are static at server startup; no hot reload yet.
|
||||
- Source provider lifecycle is early, especially for non-MCP mutable sources.
|
||||
- Workflow portability is scoped; local Python code, MCP catalogs, auth records,
|
||||
and source stores can differ between environments.
|
||||
- The prototype has not been evaluated against a broad external provider catalog
|
||||
or a large user study.
|
||||
- File-backed stores are the current implementation proof for durable lifecycle;
|
||||
durability itself should not be framed as filesystem-specific.
|
||||
- Auth records/admin surfaces exist as prototype plumbing, but end-to-end
|
||||
production credential handling is not verified as a core thesis claim.
|
||||
- Run deletion is not implemented.
|
||||
- MCP widget/resource proxying is not supported; upstream interactive widgets
|
||||
are not carried through the durable workflow path.
|
||||
- Crash recovery is at stopped boundaries, not arbitrary mid-node checkpoints.
|
||||
- Offline scheduling is not implemented yet.
|
||||
- There is no visual workflow editor yet.
|
||||
- There is no bundled autonomous agent brain.
|
||||
- General fork/gather workflow control is future work.
|
||||
- There is no full approval, roles, policy, or multi-user review system.
|
||||
|
||||
Limitations make the thesis more credible. They also motivate future work.
|
||||
|
||||
## 11. Future Work
|
||||
|
||||
Likely future-work sections:
|
||||
|
||||
- provider lifecycle: add/update/remove/apply/reload for multiple source families
|
||||
- OpenAPI or fetch-style source provider for broader HTTP integration
|
||||
- Python development reload
|
||||
- LLM nodes as typed source capabilities
|
||||
- production auth/secret stores
|
||||
- SQL/transactional stores
|
||||
- scheduler/server daemon operations
|
||||
- offline scheduling for deployments
|
||||
- fork/gather workflow control
|
||||
- richer run rewind/time-travel debugging beyond stopped/interrupted resume
|
||||
- UI/admin dashboard
|
||||
- first-party workflow UI for listing, inspecting, and editing workflows
|
||||
- richer evaluation with real workflows and larger source catalogs
|
||||
|
||||
## 12. Conclusion
|
||||
|
||||
Restate the core argument: external LLM agents can author and operate workflows,
|
||||
but the durable workflow lifecycle should live in a typed platform substrate.
|
||||
Summarize what the implementation proves: artifacts, deployments, runs,
|
||||
validation, source providers, server/API/CLI surfaces, and reproducible evidence
|
||||
across built-in, MCP, and Python sources.
|
||||
|
||||
## Appendices
|
||||
|
||||
Keep long operational material out of the main argument:
|
||||
|
||||
- reproducible command transcript for the case study
|
||||
- smoke-test commands and abbreviated outputs
|
||||
- selected config files
|
||||
- generated workflow/artifact/deployment examples
|
||||
- test/evidence index
|
||||
|
||||
## What Not To Do
|
||||
|
||||
Do not make the thesis a commit history. The reader does not need every
|
||||
refactor.
|
||||
|
||||
Do not over-center MCP. MCP is one source family and one compatibility/frontend
|
||||
area, not the whole platform.
|
||||
|
||||
Do not claim unimplemented production properties:
|
||||
|
||||
- no Python sandbox
|
||||
- no general provider hot reload
|
||||
- no production secret manager
|
||||
- no full MCP widget passthrough
|
||||
- no unmeasured performance or production-readiness claims
|
||||
|
||||
The strongest version is an honest systems argument:
|
||||
|
||||
1. Direct LLM tool orchestration has durability and validation problems.
|
||||
2. Typed workflows address those problems.
|
||||
3. The implementation proves the model across local, MCP, and Python sources.
|
||||
4. Remaining work is clear and bounded.
|
||||
Reference in New Issue
Block a user