@tea-agent/loop-agent 0.13.0-alpha.0 → 0.13.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (262) hide show
  1. package/AGENTS.md +155 -153
  2. package/CHANGELOG.md +326 -301
  3. package/README.md +345 -326
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/generate-task-dag.js +28 -58
  7. package/dist/application/evaluation/candidate-hash.js +75 -0
  8. package/dist/application/evaluation/candidate.js +52 -0
  9. package/dist/application/evaluation/replay.js +289 -0
  10. package/dist/application/evaluation/types.js +130 -0
  11. package/dist/cli/command-definitions.js +17 -4
  12. package/dist/cli/program.js +8 -4
  13. package/dist/commands/cursor-prompt.js +6 -6
  14. package/dist/commands/eval.js +235 -0
  15. package/dist/commands/init.js +544 -506
  16. package/dist/commands/loop-benchmark.js +11 -11
  17. package/dist/commands/pi-reuse-benchmark.js +16 -16
  18. package/dist/executors/pi-sdk-executor.js +38 -24
  19. package/dist/executors/shell-executor.js +34 -2
  20. package/dist/executors/shell-presets.js +20 -0
  21. package/dist/executors/shell-verification.js +7 -0
  22. package/dist/governance/manifest-types.js +1 -0
  23. package/dist/infrastructure/evaluation/candidate-store.js +435 -0
  24. package/dist/infrastructure/evaluation/store.js +40 -0
  25. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  26. package/dist/task/config-types.js +23 -0
  27. package/dist/task/runtime.js +27 -27
  28. package/dist/worker/observe/routes.js +18 -3
  29. package/dist/worker/observe/spec-evidence.js +1 -1
  30. package/dist/worker/observe/static/api.js +46 -46
  31. package/dist/worker/observe/static/app.js +150 -150
  32. package/dist/worker/observe/static/constants.js +148 -148
  33. package/dist/worker/observe/static/copy.js +67 -67
  34. package/dist/worker/observe/static/dag-helpers.js +172 -172
  35. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  36. package/dist/worker/observe/static/dag-layout.js +83 -83
  37. package/dist/worker/observe/static/dag-model.js +72 -72
  38. package/dist/worker/observe/static/dom.js +61 -61
  39. package/dist/worker/observe/static/format-pool.js +67 -67
  40. package/dist/worker/observe/static/format.js +292 -292
  41. package/dist/worker/observe/static/index.html +308 -308
  42. package/dist/worker/observe/static/kpi.js +94 -94
  43. package/dist/worker/observe/static/relations.js +133 -133
  44. package/dist/worker/observe/static/router.js +93 -93
  45. package/dist/worker/observe/static/run-processing.js +148 -148
  46. package/dist/worker/observe/static/shell-chrome.js +68 -68
  47. package/dist/worker/observe/static/state.js +253 -253
  48. package/dist/worker/observe/static/styles.css +1902 -1902
  49. package/dist/worker/observe/static/views/batch.js +227 -227
  50. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  51. package/dist/worker/observe/static/views/dag-inspector.js +607 -596
  52. package/dist/worker/observe/static/views/dag.js +362 -362
  53. package/dist/worker/observe/static/views/dashboard.js +445 -445
  54. package/dist/worker/observe/static/views/failures.js +143 -143
  55. package/dist/worker/observe/static/views/feature.js +492 -492
  56. package/dist/worker/observe/static/views/pool.js +350 -350
  57. package/dist/worker/observe/static/views/run.js +453 -453
  58. package/dist/worker/observe/static/views/session-timeline.js +205 -205
  59. package/dist/worker/observe/static/views/shell.js +7 -7
  60. package/dist/worker/observe/static/views/task.js +314 -314
  61. package/dist/worker/observe/static/views/timeline.js +163 -163
  62. package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
  63. package/dist/workflows/dag/canvas-observer.js +275 -275
  64. package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
  65. package/dist/workflows/dag/init-hybrid.js +1415 -200
  66. package/dist/workflows/dag/node-execution.js +9 -0
  67. package/dist/workflows/dag/prompt.js +9 -0
  68. package/dist/workflows/dag/report.js +35 -1
  69. package/dist/workflows/dag/runner.js +28 -2
  70. package/dist/workflows/dag/task-demand-routing.js +383 -0
  71. package/dist/workflows/dag/types.js +50 -13
  72. package/dist/workflows/dag/upstream-artifacts.js +1 -0
  73. package/dist/workflows/dag/validate.js +59 -1
  74. package/docs/README.md +106 -104
  75. package/docs/agent-dag-recovery-playbook.md +195 -193
  76. package/docs/agent-dag-runner.md +67 -67
  77. package/docs/architecture/README.md +26 -26
  78. package/docs/architecture/dag-execution.md +140 -140
  79. package/docs/architecture/evolution.md +54 -54
  80. package/docs/architecture/facts-and-state.md +71 -71
  81. package/docs/architecture/runtime-boundaries.md +191 -191
  82. package/docs/architecture/system-overview.md +93 -93
  83. package/docs/architecture/worker-and-feature.md +85 -85
  84. package/docs/cursor-prompt-sidecar.md +36 -36
  85. package/docs/decisions/README.md +18 -18
  86. package/docs/design/README.md +167 -85
  87. package/docs/development-principles.md +73 -73
  88. package/docs/exec-plans/README.md +6 -6
  89. package/docs/exec-plans/active/README.md +15 -11
  90. package/docs/exec-plans/completed/README.md +85 -74
  91. package/docs/feature-workflow.md +389 -339
  92. package/docs/harness-methodology-debugging.md +153 -153
  93. package/docs/harness-methodology-tdd.md +130 -130
  94. package/docs/harness-methodology-verification.md +27 -27
  95. package/docs/init-surface.manifest.json +289 -280
  96. package/docs/loop-agent-harness.md +142 -141
  97. package/docs/production-readiness.md +96 -96
  98. package/docs/progress/README.md +64 -58
  99. package/docs/reports/README.md +117 -100
  100. package/docs/skills/README.md +7 -7
  101. package/docs/skills/vetted-skill-registry.md +29 -27
  102. package/docs/templates/adr.md +60 -60
  103. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  104. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  105. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  106. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  107. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  108. package/docs/templates/agent-dag-report.schema.json +473 -473
  109. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  110. package/docs/templates/agent-dag.base.json +190 -190
  111. package/docs/templates/agent-dag.final-verification.json +185 -185
  112. package/docs/templates/agent-dag.schema.json +411 -383
  113. package/docs/templates/agent-dag.supervised-implementation.json +501 -501
  114. package/docs/templates/backend-test-analysis.schema.json +44 -0
  115. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
  116. package/docs/templates/backend-test-dag.json +311 -288
  117. package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
  118. package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
  119. package/docs/templates/exec-plan.md +64 -64
  120. package/docs/templates/feature-spec.md +53 -53
  121. package/docs/templates/frontend-design-contract.md +42 -33
  122. package/docs/templates/frontend-task-constraints.md +35 -25
  123. package/docs/templates/frontend-task-requirement.md +70 -61
  124. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
  125. package/docs/templates/frontend-test-dag.json +23 -0
  126. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
  127. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
  128. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
  129. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
  130. package/docs/templates/harness.schema.json +221 -221
  131. package/docs/templates/hybrid-dag.json +188 -188
  132. package/docs/templates/init-evolution-review.md +35 -35
  133. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  134. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  135. package/docs/templates/knowledge-sync-dag.json +178 -177
  136. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  137. package/docs/templates/product-line/AGENTS.md +8 -8
  138. package/docs/templates/product-line/README.md +9 -9
  139. package/docs/templates/product-line/acceptance.yaml +14 -14
  140. package/docs/templates/product-line/closeout.yaml +9 -9
  141. package/docs/templates/product-line/design.md +13 -13
  142. package/docs/templates/product-line/links.md +10 -10
  143. package/docs/templates/product-line/requirement.md +17 -17
  144. package/docs/templates/product-line/task-graph.yaml +15 -15
  145. package/docs/templates/product-line/task.yaml +64 -64
  146. package/docs/templates/product-line/test-plan.md +7 -7
  147. package/docs/templates/production-readiness-checklist.md +57 -57
  148. package/docs/templates/progress-log.md +17 -17
  149. package/docs/templates/project-start-checklist.md +9 -9
  150. package/docs/templates/qa-report.md +48 -48
  151. package/docs/templates/sprint-contract.md +29 -29
  152. package/docs/templates/worker-dogfood-evidence.md +80 -80
  153. package/docs/templates/worker-dogfood-setup.md +68 -68
  154. package/docs/verification-matrix.md +70 -67
  155. package/examples/decision-gate-agent-dag.json +177 -177
  156. package/examples/example-dag.json +46 -46
  157. package/examples/hybrid-loop-agent-dag.json +189 -189
  158. package/harness.json +66 -66
  159. package/package.json +88 -52
  160. package/scripts/check-product-line-docs.sh +29 -29
  161. package/scripts/check-task-pool-root.sh +32 -32
  162. package/scripts/kb-bootstrap-init-skeleton.sh +240 -239
  163. package/scripts/kb-graph-incremental-prepare.mjs +386 -372
  164. package/scripts/kb-graph-incremental-prepare.sh +5 -5
  165. package/scripts/kb-graph-materialize.mjs +105 -105
  166. package/scripts/kb-graph-materialize.sh +4 -4
  167. package/scripts/kb-graph-promote.mjs +164 -153
  168. package/scripts/kb-graph-promote.sh +4 -4
  169. package/scripts/kb-query.mjs +554 -554
  170. package/scripts/kb-query.sh +5 -5
  171. package/skills/agent-worker/SKILL.md +39 -39
  172. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  173. package/skills/ai-engineering-context/SKILL.md +48 -48
  174. package/skills/analyze-product-dependencies/SKILL.md +67 -0
  175. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
  176. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
  177. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
  178. package/skills/analyze-product-dependencies/references/example.md +76 -0
  179. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
  180. package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
  181. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
  182. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
  183. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
  184. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
  185. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
  186. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
  187. package/skills/analyze-product-requirements/SKILL.md +90 -0
  188. package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
  189. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
  190. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
  191. package/skills/analyze-product-requirements/references/example.md +86 -0
  192. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
  193. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
  194. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
  195. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
  196. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
  197. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
  198. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
  199. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
  200. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
  201. package/skills/code-review-core/SKILL.md +20 -20
  202. package/skills/codebase-scout/SKILL.md +19 -19
  203. package/skills/frontend-design-review/SKILL.md +66 -61
  204. package/skills/frontend-design-review/references/review-checklist.md +58 -37
  205. package/skills/frontend-implementation/SKILL.md +45 -52
  206. package/skills/frontend-implementation/references/code-standards.md +32 -34
  207. package/skills/frontend-implementation/references/design-spec.md +46 -46
  208. package/skills/frontend-implementation/references/node-contracts.md +76 -63
  209. package/skills/frontend-review/SKILL.md +59 -53
  210. package/skills/frontend-review/references/review-findings.md +47 -42
  211. package/skills/frontend-verification/SKILL.md +53 -40
  212. package/skills/frontend-verification/references/verification-checklist.md +68 -56
  213. package/skills/grill-me/SKILL.md +10 -10
  214. package/skills/grill-with-docs/SKILL.md +88 -88
  215. package/skills/grill-with-docs/adr-format.md +47 -47
  216. package/skills/grill-with-docs/context-format.md +60 -60
  217. package/skills/init-capability-evolution/SKILL.md +70 -70
  218. package/skills/loop-agent/SKILL.md +151 -151
  219. package/skills/loop-agent/references/README.md +67 -67
  220. package/skills/loop-agent/references/command-reference.md +505 -453
  221. package/skills/loop-agent/references/docs-converge.md +126 -126
  222. package/skills/loop-agent/references/harness-policy.md +263 -263
  223. package/skills/loop-agent/references/hybrid-dag.md +238 -233
  224. package/skills/loop-agent/references/learned/README.md +21 -21
  225. package/skills/loop-agent/references/long-running-loop.md +57 -57
  226. package/skills/loop-agent/references/model-routing.md +36 -36
  227. package/skills/loop-agent/references/multi-worktree.md +54 -54
  228. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  229. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  230. package/skills/loop-agent/references/pi-prompt.md +23 -23
  231. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  232. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  233. package/skills/loop-agent/references/task-workflow.md +89 -89
  234. package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
  235. package/skills/playwright-cli/SKILL.md +420 -0
  236. package/skills/playwright-cli/references/element-attributes.md +23 -0
  237. package/skills/playwright-cli/references/playwright-tests.md +39 -0
  238. package/skills/playwright-cli/references/request-mocking.md +87 -0
  239. package/skills/playwright-cli/references/running-code.md +241 -0
  240. package/skills/playwright-cli/references/session-management.md +225 -0
  241. package/skills/playwright-cli/references/storage-state.md +275 -0
  242. package/skills/playwright-cli/references/test-generation.md +433 -0
  243. package/skills/playwright-cli/references/tracing.md +139 -0
  244. package/skills/playwright-cli/references/video-recording.md +143 -0
  245. package/skills/playwright-cli-case-generator/SKILL.md +74 -0
  246. package/skills/requesting-code-review/SKILL.md +101 -101
  247. package/skills/requesting-code-review/code-reviewer.md +168 -168
  248. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  249. package/skills/systematic-debugging/SKILL.md +296 -296
  250. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  251. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  252. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  253. package/skills/systematic-debugging/find-polluter.sh +63 -63
  254. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  255. package/skills/systematic-debugging/test-academic.md +14 -14
  256. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  257. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  258. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  259. package/skills/test-driven-development/SKILL.md +20 -20
  260. package/skills/using-git-worktrees/SKILL.md +215 -215
  261. package/skills/verification-before-completion/SKILL.md +154 -154
  262. package/skills/webapp-testing/SKILL.md +19 -19
@@ -0,0 +1,130 @@
1
+ import { z } from "zod";
2
+ export const evalSplitSchema = z.enum(["public", "private", "held_out"]);
3
+ export const replayEvidenceRefSchema = z
4
+ .object({
5
+ candidateId: z.string().min(1),
6
+ taskRef: z.string().min(1),
7
+ seed: z.number().int().nonnegative(),
8
+ split: evalSplitSchema,
9
+ runId: z
10
+ .string()
11
+ .regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/, "runId must be one safe path segment"),
12
+ stateSha256: z.string().regex(/^[a-f0-9]{64}$/),
13
+ runSha256: z.string().regex(/^[a-f0-9]{64}$/),
14
+ })
15
+ .strict();
16
+ export const replaySpecSchema = z
17
+ .object({
18
+ schemaVersion: z.literal(1),
19
+ replayId: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/),
20
+ incumbentCandidateId: z.string().min(1),
21
+ challengerCandidateId: z.string().min(1),
22
+ evidence: z.array(replayEvidenceRefSchema).min(1),
23
+ })
24
+ .strict()
25
+ .superRefine((spec, ctx) => {
26
+ if (spec.incumbentCandidateId === spec.challengerCandidateId) {
27
+ ctx.addIssue({
28
+ code: z.ZodIssueCode.custom,
29
+ message: "incumbentCandidateId and challengerCandidateId must differ",
30
+ path: ["challengerCandidateId"],
31
+ });
32
+ }
33
+ const allowed = new Set([
34
+ spec.incumbentCandidateId,
35
+ spec.challengerCandidateId,
36
+ ]);
37
+ const pairKeys = new Set();
38
+ for (let i = 0; i < spec.evidence.length; i += 1) {
39
+ const evidence = spec.evidence[i];
40
+ if (!allowed.has(evidence.candidateId)) {
41
+ ctx.addIssue({
42
+ code: z.ZodIssueCode.custom,
43
+ message: "evidence candidateId must match incumbent or challenger",
44
+ path: ["evidence", i, "candidateId"],
45
+ });
46
+ }
47
+ const key = `${evidence.candidateId}\u0000${evidence.split}\u0000${evidence.taskRef}\u0000${evidence.seed}`;
48
+ if (pairKeys.has(key)) {
49
+ ctx.addIssue({
50
+ code: z.ZodIssueCode.custom,
51
+ message: "duplicate evidence for candidateId + split + taskRef + seed",
52
+ path: ["evidence", i],
53
+ });
54
+ }
55
+ pairKeys.add(key);
56
+ }
57
+ });
58
+ // --- Candidate Registry (M2 W2.1–W2.2) ---
59
+ export const candidateIdSchema = z
60
+ .string()
61
+ .regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/, "candidateId must be one safe path segment");
62
+ export const candidateKindSchema = z.enum([
63
+ "prompt",
64
+ "skill",
65
+ "context_policy",
66
+ "model_routing",
67
+ "profile",
68
+ "composite",
69
+ ]);
70
+ export const candidateContentRefSchema = z
71
+ .object({
72
+ path: z
73
+ .string()
74
+ .min(1)
75
+ .refine((value) => !pathIsAbsoluteLike(value), "content ref path must be repo-relative"),
76
+ sha256: z
77
+ .string()
78
+ .regex(/^(sha256:)?[a-f0-9]{64}$/i, "sha256 must be 64 hex digits"),
79
+ })
80
+ .strict();
81
+ function pathIsAbsoluteLike(value) {
82
+ if (value.startsWith("/") || value.startsWith("\\"))
83
+ return true;
84
+ if (/^[A-Za-z]:[\\/]/.test(value))
85
+ return true;
86
+ return false;
87
+ }
88
+ export const candidateManifestInputSchema = z
89
+ .object({
90
+ schemaVersion: z.literal(1),
91
+ candidateId: candidateIdSchema,
92
+ parentCandidateId: candidateIdSchema.nullable().optional(),
93
+ candidateKind: candidateKindSchema,
94
+ createdAt: z.string().min(1),
95
+ description: z.string().optional(),
96
+ contentRefs: z.array(candidateContentRefSchema).min(1),
97
+ // optional on input; always computed/verified on register/read
98
+ bundleHash: z
99
+ .string()
100
+ .regex(/^(sha256:)?[a-f0-9]{64}$/i)
101
+ .optional(),
102
+ })
103
+ .strict();
104
+ export const candidateManifestSchema = candidateManifestInputSchema
105
+ .extend({
106
+ parentCandidateId: candidateIdSchema.nullable(),
107
+ bundleHash: z.string().regex(/^sha256:[a-f0-9]{64}$/),
108
+ })
109
+ .strict();
110
+ export const lifecycleStateSchema = z.enum([
111
+ "proposed",
112
+ "eligible",
113
+ "experimenting",
114
+ "accepted",
115
+ "rejected",
116
+ "invalid",
117
+ "retired",
118
+ ]);
119
+ export const lifecycleEventSchema = z
120
+ .object({
121
+ schemaVersion: z.literal(1),
122
+ seq: z.number().int().positive(),
123
+ from: lifecycleStateSchema.nullable(),
124
+ to: lifecycleStateSchema,
125
+ reason: z.string().min(1),
126
+ at: z.string().min(1),
127
+ previousEventHash: z.string().regex(/^[a-f0-9]{64}$/),
128
+ eventHash: z.string().regex(/^[a-f0-9]{64}$/),
129
+ })
130
+ .strict();
@@ -1,6 +1,7 @@
1
1
  import { runDoctor } from "../commands/doctor.js";
2
2
  import { runDocsArchive } from "../commands/docs-archive.js";
3
3
  import { runDocsAudit } from "../commands/docs-audit.js";
4
+ import { runEval } from "../commands/eval.js";
4
5
  import { runCoverageAudit } from "../commands/coverage-audit.js";
5
6
  import { runExamples } from "../commands/examples.js";
6
7
  import { runCloseout } from "../commands/closeout.js";
@@ -10,7 +11,7 @@ import { runInstructions } from "../commands/instructions.js";
10
11
  import { runNewTask } from "../commands/new-task.js";
11
12
  import { runImportPrd } from "../commands/import-prd.js";
12
13
  import { runPlanList } from "../commands/plan-list.js";
13
- import { runPlanCheck, runPlanComplete, runPlanCreate } from "../commands/plan.js";
14
+ import { runPlanCheck, runPlanComplete, runPlanCreate, } from "../commands/plan.js";
14
15
  import { runPromoteRun } from "../commands/promote-run.js";
15
16
  import { runSpine } from "../commands/spine.js";
16
17
  import { runStats } from "../commands/stats.js";
@@ -76,6 +77,7 @@ const INIT_SUBCOMMANDS = [
76
77
  "update",
77
78
  ];
78
79
  const EXAMPLES_SUBCOMMANDS = ["list", "show", "copy"];
80
+ const EVAL_SUBCOMMANDS = ["replay", "report", "candidate"];
79
81
  const CLOSEOUT_SUBCOMMANDS = ["task"];
80
82
  const PLAN_SUBCOMMANDS = ["list", "create", "complete", "check"];
81
83
  const SPINE_SUBCOMMANDS = ["audit"];
@@ -194,6 +196,17 @@ export const COMMAND_DEFINITIONS = [
194
196
  await runExamples(repoRoot, [subcommand, ...rest].filter(Boolean));
195
197
  },
196
198
  },
199
+ {
200
+ name: "eval",
201
+ adapter: "required",
202
+ tier: "operator",
203
+ intent: "Replay completed DAG evidence and manage immutable Candidate Registry lifecycle without live model execution or promotion.",
204
+ usage: "eval <replay|report|candidate> ...; candidate <register|show|list|transition> [--json|--markdown]",
205
+ subcommands: [...EVAL_SUBCOMMANDS],
206
+ handler: async ({ repoRoot, subcommand, rest }) => {
207
+ await runEval(repoRoot, [subcommand, ...rest].filter((arg) => Boolean(arg)));
208
+ },
209
+ },
197
210
  {
198
211
  name: "new-task",
199
212
  adapter: "required",
@@ -286,17 +299,17 @@ export const COMMAND_DEFINITIONS = [
286
299
  if (subcommand === "create") {
287
300
  const [planId, ...titleParts] = rest;
288
301
  if (!planId)
289
- throw new Error("usage: plan create <plan-id> \"<title>\"");
302
+ throw new Error('usage: plan create <plan-id> "<title>"');
290
303
  await runPlanCreate(repoRoot, planId, titleParts.join(" "));
291
304
  return;
292
305
  }
293
306
  if (subcommand === "complete") {
294
307
  const [planId, ...summaryParts] = rest;
295
308
  if (!planId)
296
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
309
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
297
310
  const summary = parseSummaryFlag(summaryParts);
298
311
  if (!summary)
299
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
312
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
300
313
  await runPlanComplete(repoRoot, planId, { summary });
301
314
  return;
302
315
  }
@@ -22,6 +22,7 @@ import { runDagWorkflowValidate } from "../commands/dag-workflow-validate.js";
22
22
  import { runDelegate } from "../commands/delegate.js";
23
23
  import { runDocsArchive } from "../commands/docs-archive.js";
24
24
  import { runDocsAudit } from "../commands/docs-audit.js";
25
+ import { runEval } from "../commands/eval.js";
25
26
  import { runDoctor } from "../commands/doctor.js";
26
27
  import { runExamples } from "../commands/examples.js";
27
28
  import { runGoal } from "../commands/goal.js";
@@ -38,7 +39,7 @@ import { runImportPrd } from "../commands/import-prd.js";
38
39
  import { runPiReuseBenchmark } from "../commands/pi-reuse-benchmark.js";
39
40
  import { parsePiPromptArgs, printPiPromptUsage, runPiPrompt, } from "../commands/pi-prompt.js";
40
41
  import { runPlanList } from "../commands/plan-list.js";
41
- import { runPlanCheck, runPlanComplete, runPlanCreate } from "../commands/plan.js";
42
+ import { runPlanCheck, runPlanComplete, runPlanCreate, } from "../commands/plan.js";
42
43
  import { runPromoteRun } from "../commands/promote-run.js";
43
44
  import { runReferenceIndex } from "../commands/reference-index.js";
44
45
  import { runRunDag } from "../commands/run-dag.js";
@@ -163,17 +164,17 @@ async function runCommanderAction(ctx, command, subcommand, rest) {
163
164
  if (subcommand === "create") {
164
165
  const [planId, ...titleParts] = rest;
165
166
  if (!planId)
166
- throw new Error("usage: plan create <plan-id> \"<title>\"");
167
+ throw new Error('usage: plan create <plan-id> "<title>"');
167
168
  await runPlanCreate(ctx.repoRoot, planId, titleParts.join(" "));
168
169
  return;
169
170
  }
170
171
  if (subcommand === "complete") {
171
172
  const [planId, ...summaryParts] = rest;
172
173
  if (!planId)
173
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
174
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
174
175
  const summary = parsePlanSummaryFlag(summaryParts);
175
176
  if (!summary)
176
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
177
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
177
178
  await runPlanComplete(ctx.repoRoot, planId, { summary });
178
179
  return;
179
180
  }
@@ -244,6 +245,9 @@ async function runCommanderAction(ctx, command, subcommand, rest) {
244
245
  case "knowledge":
245
246
  await runKnowledge(ctx.repoRoot, compactArgs([subcommand, ...rest]));
246
247
  return;
248
+ case "eval":
249
+ await runEval(ctx.repoRoot, compactArgs([subcommand, ...rest]));
250
+ return;
247
251
  case "dag":
248
252
  await runDagAction(ctx.repoRoot, subcommand, rest);
249
253
  return;
@@ -141,7 +141,7 @@ async function runCursorPromptBatch(task, cwd, model, timeoutMs) {
141
141
  }
142
142
  async function runCursorPromptStreaming(task, cwd, model, timeoutMs) {
143
143
  const startedAt = Date.now();
144
- process.stderr.write(`[cursor-prompt] streaming (model=${model}, cwd=${cwd})
144
+ process.stderr.write(`[cursor-prompt] streaming (model=${model}, cwd=${cwd})
145
145
  `);
146
146
  const runDir = (await computeRunDir(cwd, task)) ?? undefined;
147
147
  const result = await executeCursorPromptStream({
@@ -164,15 +164,15 @@ async function runCursorPromptStreaming(task, cwd, model, timeoutMs) {
164
164
  });
165
165
  const elapsed = ((Date.now() - startedAt) / 1000).toFixed(1);
166
166
  if (result.ok) {
167
- process.stderr.write(`
168
- [cursor-prompt] done in ${elapsed}s, status=${result.status}
167
+ process.stderr.write(`
168
+ [cursor-prompt] done in ${elapsed}s, status=${result.status}
169
169
  `);
170
170
  }
171
171
  else {
172
172
  const stderr = result.stderr || "(no output)";
173
- process.stderr.write(`
174
- [cursor-prompt] FAILED in ${elapsed}s (${result.failureCategory}):
175
- ${stderr}
173
+ process.stderr.write(`
174
+ [cursor-prompt] FAILED in ${elapsed}s (${result.failureCategory}):
175
+ ${stderr}
176
176
  `);
177
177
  process.exit(1);
178
178
  }
@@ -0,0 +1,235 @@
1
+ import path from "node:path";
2
+ import { writeTextAtomic } from "../infrastructure/harness/atomic-write.js";
3
+ import { readReplayScorecard } from "../infrastructure/evaluation/store.js";
4
+ import { formatReplayMarkdown, replayEvaluation, } from "../application/evaluation/replay.js";
5
+ import { formatCandidateMarkdown, listCandidates, registerCandidate, showCandidate, transitionCandidate, } from "../application/evaluation/candidate.js";
6
+ import { lifecycleStateSchema, } from "../application/evaluation/types.js";
7
+ const USAGE = "usage: eval <replay|report|candidate> ...; candidate <register|show|list|transition> [--json|--markdown]";
8
+ function parseFormatFlags(args) {
9
+ const json = args.includes("--json");
10
+ const markdown = args.includes("--markdown");
11
+ if (json && markdown) {
12
+ throw new Error("eval accepts only one of --json or --markdown");
13
+ }
14
+ return { json: json || !markdown, markdown };
15
+ }
16
+ function flagValue(args, flag) {
17
+ const index = args.indexOf(flag);
18
+ if (index >= 0) {
19
+ const value = args[index + 1];
20
+ if (!value || value.startsWith("-")) {
21
+ throw new Error(`${flag} requires a value`);
22
+ }
23
+ return value;
24
+ }
25
+ const prefix = `${flag}=`;
26
+ return args.find((arg) => arg.startsWith(prefix))?.slice(prefix.length);
27
+ }
28
+ function assertKnownFlags(args, allowed) {
29
+ for (let i = 0; i < args.length; i += 1) {
30
+ const arg = args[i];
31
+ if (!arg.startsWith("-"))
32
+ continue;
33
+ const key = arg.split("=", 1)[0];
34
+ if (!allowed.includes(key)) {
35
+ throw new Error(`unknown eval argument: ${arg}`);
36
+ }
37
+ if ([
38
+ "--spec",
39
+ "--output",
40
+ "--replay-id",
41
+ "--manifest",
42
+ "--candidate-id",
43
+ "--to",
44
+ "--reason",
45
+ ].includes(key) &&
46
+ !arg.includes("=")) {
47
+ i += 1;
48
+ }
49
+ }
50
+ }
51
+ function parseCandidateArgs(rest) {
52
+ const [action, ...tail] = rest;
53
+ if (action === "register") {
54
+ assertKnownFlags(tail, ["--manifest", "--json", "--markdown"]);
55
+ const manifestPath = flagValue(tail, "--manifest");
56
+ if (!manifestPath) {
57
+ throw new Error("eval candidate register requires --manifest <path>");
58
+ }
59
+ return {
60
+ command: "candidate",
61
+ action: "register",
62
+ manifestPath,
63
+ ...parseFormatFlags(tail),
64
+ };
65
+ }
66
+ if (action === "show") {
67
+ assertKnownFlags(tail, ["--candidate-id", "--json", "--markdown"]);
68
+ const candidateId = flagValue(tail, "--candidate-id");
69
+ if (!candidateId) {
70
+ throw new Error("eval candidate show requires --candidate-id <id>");
71
+ }
72
+ return {
73
+ command: "candidate",
74
+ action: "show",
75
+ candidateId,
76
+ ...parseFormatFlags(tail),
77
+ };
78
+ }
79
+ if (action === "list") {
80
+ assertKnownFlags(tail, ["--json", "--markdown"]);
81
+ return {
82
+ command: "candidate",
83
+ action: "list",
84
+ ...parseFormatFlags(tail),
85
+ };
86
+ }
87
+ if (action === "transition") {
88
+ assertKnownFlags(tail, [
89
+ "--candidate-id",
90
+ "--to",
91
+ "--reason",
92
+ "--json",
93
+ "--markdown",
94
+ ]);
95
+ const candidateId = flagValue(tail, "--candidate-id");
96
+ const toRaw = flagValue(tail, "--to");
97
+ const reason = flagValue(tail, "--reason");
98
+ if (!candidateId) {
99
+ throw new Error("eval candidate transition requires --candidate-id <id>");
100
+ }
101
+ if (!toRaw) {
102
+ throw new Error("eval candidate transition requires --to <state>");
103
+ }
104
+ if (!reason) {
105
+ throw new Error("eval candidate transition requires --reason <text>");
106
+ }
107
+ const to = lifecycleStateSchema.parse(toRaw);
108
+ return {
109
+ command: "candidate",
110
+ action: "transition",
111
+ candidateId,
112
+ to,
113
+ reason,
114
+ ...parseFormatFlags(tail),
115
+ };
116
+ }
117
+ throw new Error("usage: eval candidate <register|show|list|transition> ...");
118
+ }
119
+ export function parseEvalArgs(args) {
120
+ const [command, ...rest] = args;
121
+ if (command === "replay") {
122
+ assertKnownFlags(rest, ["--spec", "--output", "--json", "--markdown"]);
123
+ const specPath = flagValue(rest, "--spec");
124
+ if (!specPath)
125
+ throw new Error("eval replay requires --spec <path>");
126
+ return {
127
+ command,
128
+ specPath,
129
+ outputPath: flagValue(rest, "--output"),
130
+ ...parseFormatFlags(rest),
131
+ };
132
+ }
133
+ if (command === "report") {
134
+ assertKnownFlags(rest, ["--replay-id", "--json", "--markdown"]);
135
+ const replayId = flagValue(rest, "--replay-id");
136
+ if (!replayId)
137
+ throw new Error("eval report requires --replay-id <id>");
138
+ return { command, replayId, ...parseFormatFlags(rest) };
139
+ }
140
+ if (command === "candidate") {
141
+ return parseCandidateArgs(rest);
142
+ }
143
+ throw new Error(USAGE);
144
+ }
145
+ function printScorecard(input) {
146
+ if (input.json) {
147
+ console.log(JSON.stringify(input.scorecard, null, 2));
148
+ return;
149
+ }
150
+ process.stdout.write(input.markdown);
151
+ }
152
+ function printCandidate(input) {
153
+ if (input.json) {
154
+ console.log(JSON.stringify(input.record, null, 2));
155
+ return;
156
+ }
157
+ process.stdout.write(formatCandidateMarkdown(input.record));
158
+ }
159
+ export async function runEval(repoRoot, args) {
160
+ const parsed = parseEvalArgs(args);
161
+ if (parsed.command === "replay") {
162
+ const result = await replayEvaluation({
163
+ repoRoot,
164
+ specPath: parsed.specPath,
165
+ });
166
+ if (parsed.outputPath) {
167
+ const outputPath = path.resolve(repoRoot, parsed.outputPath);
168
+ await writeTextAtomic(outputPath, parsed.markdown
169
+ ? result.markdown
170
+ : `${JSON.stringify(result.scorecard, null, 2)}\n`, { repoRoot });
171
+ }
172
+ printScorecard({
173
+ scorecard: result.scorecard,
174
+ markdown: result.markdown,
175
+ json: parsed.json,
176
+ });
177
+ return;
178
+ }
179
+ if (parsed.command === "report") {
180
+ const scorecard = (await readReplayScorecard(repoRoot, parsed.replayId));
181
+ const markdown = formatReplayMarkdown(scorecard);
182
+ printScorecard({ scorecard, markdown, json: parsed.json });
183
+ return;
184
+ }
185
+ // candidate subcommands
186
+ if (parsed.action === "register") {
187
+ const result = await registerCandidate({
188
+ repoRoot,
189
+ manifestPath: parsed.manifestPath,
190
+ });
191
+ if (parsed.json) {
192
+ console.log(JSON.stringify({
193
+ idempotent: result.idempotent,
194
+ manifestPath: result.manifestPath,
195
+ lifecyclePath: result.lifecyclePath,
196
+ record: result.record,
197
+ }, null, 2));
198
+ return;
199
+ }
200
+ process.stdout.write(`${result.idempotent ? "idempotent " : ""}registered ${result.record.manifest.candidateId}\n${formatCandidateMarkdown(result.record)}`);
201
+ return;
202
+ }
203
+ if (parsed.action === "show") {
204
+ const record = await showCandidate({
205
+ repoRoot,
206
+ candidateId: parsed.candidateId,
207
+ });
208
+ printCandidate({ record, json: parsed.json });
209
+ return;
210
+ }
211
+ if (parsed.action === "list") {
212
+ const rows = await listCandidates({ repoRoot });
213
+ if (parsed.json) {
214
+ console.log(JSON.stringify(rows, null, 2));
215
+ return;
216
+ }
217
+ const lines = [
218
+ "# Candidates",
219
+ "",
220
+ ...rows.map((row) => `- \`${row.candidateId}\` status=\`${row.status}\` hash=\`${row.bundleHash}\` promotionApplied=false`),
221
+ "",
222
+ ];
223
+ process.stdout.write(`${lines.join("\n")}\n`);
224
+ return;
225
+ }
226
+ if (parsed.action === "transition") {
227
+ const record = await transitionCandidate({
228
+ repoRoot,
229
+ candidateId: parsed.candidateId,
230
+ to: parsed.to,
231
+ reason: parsed.reason,
232
+ });
233
+ printCandidate({ record, json: parsed.json });
234
+ }
235
+ }