@tea-agent/loop-agent 0.12.0 → 0.13.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (284) hide show
  1. package/AGENTS.md +155 -153
  2. package/CHANGELOG.md +338 -265
  3. package/README.md +345 -298
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/generate-task-dag.js +28 -28
  7. package/dist/application/evaluation/candidate-hash.js +75 -0
  8. package/dist/application/evaluation/candidate.js +52 -0
  9. package/dist/application/evaluation/replay.js +289 -0
  10. package/dist/application/evaluation/types.js +130 -0
  11. package/dist/cli/command-definitions.js +27 -7
  12. package/dist/cli/program.js +8 -4
  13. package/dist/commands/cursor-prompt.js +6 -6
  14. package/dist/commands/eval.js +235 -0
  15. package/dist/commands/init.js +544 -506
  16. package/dist/commands/knowledge.js +129 -31
  17. package/dist/commands/loop-benchmark.js +11 -11
  18. package/dist/commands/pi-reuse-benchmark.js +16 -16
  19. package/dist/executors/pi-sdk-executor.js +38 -24
  20. package/dist/executors/shell-executor.js +34 -2
  21. package/dist/executors/shell-presets.js +20 -0
  22. package/dist/executors/shell-verification.js +7 -0
  23. package/dist/governance/manifest-types.js +4 -0
  24. package/dist/infrastructure/evaluation/candidate-store.js +435 -0
  25. package/dist/infrastructure/evaluation/store.js +40 -0
  26. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  27. package/dist/task/config-types.js +28 -1
  28. package/dist/task/runtime.js +27 -27
  29. package/dist/worker/cli.js +96 -1
  30. package/dist/worker/delivery/package.js +3 -3
  31. package/dist/worker/feature/decision-loader.js +37 -6
  32. package/dist/worker/feature/next-action.js +10 -2
  33. package/dist/worker/feature/ready-plan-projection.js +81 -0
  34. package/dist/worker/feature/reducer.js +2 -1
  35. package/dist/worker/feature/review.js +19 -2
  36. package/dist/worker/feature/run.js +27 -2
  37. package/dist/worker/follow-up/approve.js +5 -2
  38. package/dist/worker/follow-up/factory.js +1 -1
  39. package/dist/worker/observability/read-model.js +246 -41
  40. package/dist/worker/observe/routes.js +173 -15
  41. package/dist/worker/observe/spec-evidence.js +281 -0
  42. package/dist/worker/observe/static/api.js +46 -27
  43. package/dist/worker/observe/static/app.js +150 -150
  44. package/dist/worker/observe/static/constants.js +148 -148
  45. package/dist/worker/observe/static/copy.js +67 -67
  46. package/dist/worker/observe/static/dag-helpers.js +172 -172
  47. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  48. package/dist/worker/observe/static/dag-layout.js +83 -83
  49. package/dist/worker/observe/static/dag-model.js +72 -72
  50. package/dist/worker/observe/static/dom.js +61 -61
  51. package/dist/worker/observe/static/format-pool.js +67 -67
  52. package/dist/worker/observe/static/format.js +292 -292
  53. package/dist/worker/observe/static/index.html +308 -308
  54. package/dist/worker/observe/static/kpi.js +94 -94
  55. package/dist/worker/observe/static/relations.js +133 -128
  56. package/dist/worker/observe/static/router.js +93 -85
  57. package/dist/worker/observe/static/run-processing.js +148 -148
  58. package/dist/worker/observe/static/shell-chrome.js +68 -68
  59. package/dist/worker/observe/static/state.js +253 -253
  60. package/dist/worker/observe/static/styles.css +1902 -1890
  61. package/dist/worker/observe/static/views/batch.js +227 -226
  62. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  63. package/dist/worker/observe/static/views/dag-inspector.js +607 -477
  64. package/dist/worker/observe/static/views/dag.js +362 -362
  65. package/dist/worker/observe/static/views/dashboard.js +445 -442
  66. package/dist/worker/observe/static/views/failures.js +143 -143
  67. package/dist/worker/observe/static/views/feature.js +492 -453
  68. package/dist/worker/observe/static/views/pool.js +350 -347
  69. package/dist/worker/observe/static/views/run.js +453 -453
  70. package/dist/worker/observe/static/views/session-timeline.js +205 -205
  71. package/dist/worker/observe/static/views/shell.js +7 -7
  72. package/dist/worker/observe/static/views/task.js +314 -260
  73. package/dist/worker/observe/static/views/timeline.js +163 -163
  74. package/dist/worker/pool/doctor.js +165 -0
  75. package/dist/worker/pool/migrate-state.js +303 -0
  76. package/dist/worker/pool/run-store.js +205 -17
  77. package/dist/worker/pool/types.js +17 -1
  78. package/dist/worker/pool/validation.js +100 -15
  79. package/dist/worker/report/morning-report.js +12 -2
  80. package/dist/worker/runner/run-ready.js +41 -26
  81. package/dist/worker/task-graph/ready-planner.js +136 -0
  82. package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
  83. package/dist/workflows/dag/canvas-observer.js +275 -275
  84. package/dist/workflows/dag/convergence/controller.js +16 -8
  85. package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
  86. package/dist/workflows/dag/failure-routing.js +12 -1
  87. package/dist/workflows/dag/init-hybrid.js +2404 -360
  88. package/dist/workflows/dag/node-execution.js +9 -0
  89. package/dist/workflows/dag/prompt.js +9 -0
  90. package/dist/workflows/dag/report.js +35 -1
  91. package/dist/workflows/dag/runner.js +28 -2
  92. package/dist/workflows/dag/task-demand-routing.js +383 -0
  93. package/dist/workflows/dag/types.js +51 -13
  94. package/dist/workflows/dag/upstream-artifacts.js +1 -0
  95. package/dist/workflows/dag/validate.js +59 -1
  96. package/docs/README.md +106 -104
  97. package/docs/agent-dag-recovery-playbook.md +195 -184
  98. package/docs/agent-dag-runner.md +67 -67
  99. package/docs/architecture/README.md +26 -26
  100. package/docs/architecture/dag-execution.md +140 -140
  101. package/docs/architecture/evolution.md +54 -53
  102. package/docs/architecture/facts-and-state.md +71 -58
  103. package/docs/architecture/runtime-boundaries.md +191 -191
  104. package/docs/architecture/system-overview.md +93 -93
  105. package/docs/architecture/worker-and-feature.md +85 -81
  106. package/docs/cursor-prompt-sidecar.md +36 -36
  107. package/docs/decisions/README.md +18 -15
  108. package/docs/design/README.md +167 -77
  109. package/docs/development-principles.md +73 -73
  110. package/docs/exec-plans/README.md +6 -6
  111. package/docs/exec-plans/active/README.md +15 -9
  112. package/docs/exec-plans/completed/README.md +85 -73
  113. package/docs/feature-workflow.md +389 -261
  114. package/docs/harness-methodology-debugging.md +153 -153
  115. package/docs/harness-methodology-tdd.md +130 -130
  116. package/docs/harness-methodology-verification.md +27 -27
  117. package/docs/init-surface.manifest.json +289 -280
  118. package/docs/loop-agent-harness.md +142 -130
  119. package/docs/production-readiness.md +96 -96
  120. package/docs/progress/README.md +64 -54
  121. package/docs/reports/README.md +117 -94
  122. package/docs/skills/README.md +7 -7
  123. package/docs/skills/vetted-skill-registry.md +29 -27
  124. package/docs/templates/adr.md +60 -60
  125. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  126. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  127. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  128. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  129. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  130. package/docs/templates/agent-dag-report.schema.json +473 -473
  131. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  132. package/docs/templates/agent-dag.base.json +190 -190
  133. package/docs/templates/agent-dag.final-verification.json +185 -185
  134. package/docs/templates/agent-dag.schema.json +411 -383
  135. package/docs/templates/agent-dag.supervised-implementation.json +501 -501
  136. package/docs/templates/backend-test-analysis.schema.json +44 -0
  137. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
  138. package/docs/templates/backend-test-dag.json +311 -276
  139. package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
  140. package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
  141. package/docs/templates/exec-plan.md +64 -64
  142. package/docs/templates/feature-spec.md +53 -53
  143. package/docs/templates/frontend-design-contract.md +42 -33
  144. package/docs/templates/frontend-task-constraints.md +35 -25
  145. package/docs/templates/frontend-task-requirement.md +70 -61
  146. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
  147. package/docs/templates/frontend-test-dag.json +23 -0
  148. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
  149. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
  150. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
  151. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
  152. package/docs/templates/harness.schema.json +221 -221
  153. package/docs/templates/hybrid-dag.json +188 -188
  154. package/docs/templates/init-evolution-review.md +35 -35
  155. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  156. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -0
  157. package/docs/templates/knowledge-sync-dag.json +178 -0
  158. package/docs/templates/knowledge-sync-draft.schema.json +71 -0
  159. package/docs/templates/product-line/AGENTS.md +8 -8
  160. package/docs/templates/product-line/README.md +9 -9
  161. package/docs/templates/product-line/acceptance.yaml +14 -14
  162. package/docs/templates/product-line/closeout.yaml +9 -9
  163. package/docs/templates/product-line/design.md +13 -13
  164. package/docs/templates/product-line/links.md +10 -10
  165. package/docs/templates/product-line/requirement.md +17 -17
  166. package/docs/templates/product-line/task-graph.yaml +15 -15
  167. package/docs/templates/product-line/task.yaml +64 -64
  168. package/docs/templates/product-line/test-plan.md +7 -7
  169. package/docs/templates/production-readiness-checklist.md +57 -57
  170. package/docs/templates/progress-log.md +17 -17
  171. package/docs/templates/project-start-checklist.md +9 -9
  172. package/docs/templates/qa-report.md +48 -48
  173. package/docs/templates/sprint-contract.md +29 -29
  174. package/docs/templates/worker-dogfood-evidence.md +80 -80
  175. package/docs/templates/worker-dogfood-setup.md +68 -68
  176. package/docs/verification-matrix.md +70 -66
  177. package/examples/decision-gate-agent-dag.json +177 -177
  178. package/examples/example-dag.json +46 -46
  179. package/examples/hybrid-loop-agent-dag.json +189 -189
  180. package/harness.json +66 -66
  181. package/package.json +88 -46
  182. package/scripts/check-product-line-docs.sh +29 -29
  183. package/scripts/check-task-pool-root.sh +32 -32
  184. package/scripts/kb-bootstrap-init-skeleton.sh +240 -0
  185. package/scripts/kb-graph-incremental-prepare.mjs +386 -0
  186. package/scripts/kb-graph-incremental-prepare.sh +5 -0
  187. package/scripts/kb-graph-materialize.mjs +105 -0
  188. package/scripts/kb-graph-materialize.sh +4 -0
  189. package/scripts/kb-graph-promote.mjs +164 -0
  190. package/scripts/kb-graph-promote.sh +4 -0
  191. package/scripts/kb-query.mjs +554 -0
  192. package/scripts/kb-query.sh +5 -0
  193. package/skills/agent-worker/SKILL.md +39 -37
  194. package/skills/agent-worker/references/agent-worker-operator.md +60 -43
  195. package/skills/ai-engineering-context/SKILL.md +48 -48
  196. package/skills/analyze-product-dependencies/SKILL.md +67 -0
  197. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
  198. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
  199. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
  200. package/skills/analyze-product-dependencies/references/example.md +76 -0
  201. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
  202. package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
  203. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
  204. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
  205. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
  206. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
  207. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
  208. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
  209. package/skills/analyze-product-requirements/SKILL.md +90 -0
  210. package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
  211. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
  212. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
  213. package/skills/analyze-product-requirements/references/example.md +86 -0
  214. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
  215. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
  216. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
  217. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
  218. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
  219. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
  220. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
  221. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
  222. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
  223. package/skills/code-review-core/SKILL.md +20 -20
  224. package/skills/codebase-scout/SKILL.md +19 -19
  225. package/skills/frontend-design-review/SKILL.md +66 -59
  226. package/skills/frontend-design-review/references/review-checklist.md +58 -37
  227. package/skills/frontend-implementation/SKILL.md +47 -51
  228. package/skills/frontend-implementation/references/code-standards.md +32 -34
  229. package/skills/frontend-implementation/references/design-spec.md +46 -46
  230. package/skills/frontend-implementation/references/node-contracts.md +76 -32
  231. package/skills/frontend-review/SKILL.md +59 -53
  232. package/skills/frontend-review/references/review-findings.md +47 -42
  233. package/skills/frontend-verification/SKILL.md +53 -40
  234. package/skills/frontend-verification/references/verification-checklist.md +68 -56
  235. package/skills/grill-me/SKILL.md +10 -10
  236. package/skills/grill-with-docs/SKILL.md +88 -88
  237. package/skills/grill-with-docs/adr-format.md +47 -47
  238. package/skills/grill-with-docs/context-format.md +60 -60
  239. package/skills/init-capability-evolution/SKILL.md +70 -70
  240. package/skills/loop-agent/SKILL.md +151 -151
  241. package/skills/loop-agent/references/README.md +67 -67
  242. package/skills/loop-agent/references/command-reference.md +505 -452
  243. package/skills/loop-agent/references/docs-converge.md +126 -126
  244. package/skills/loop-agent/references/harness-policy.md +263 -263
  245. package/skills/loop-agent/references/hybrid-dag.md +238 -233
  246. package/skills/loop-agent/references/learned/README.md +21 -21
  247. package/skills/loop-agent/references/long-running-loop.md +57 -57
  248. package/skills/loop-agent/references/model-routing.md +36 -36
  249. package/skills/loop-agent/references/multi-worktree.md +54 -54
  250. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  251. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  252. package/skills/loop-agent/references/pi-prompt.md +23 -23
  253. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  254. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  255. package/skills/loop-agent/references/task-workflow.md +89 -89
  256. package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
  257. package/skills/playwright-cli/SKILL.md +420 -0
  258. package/skills/playwright-cli/references/element-attributes.md +23 -0
  259. package/skills/playwright-cli/references/playwright-tests.md +39 -0
  260. package/skills/playwright-cli/references/request-mocking.md +87 -0
  261. package/skills/playwright-cli/references/running-code.md +241 -0
  262. package/skills/playwright-cli/references/session-management.md +225 -0
  263. package/skills/playwright-cli/references/storage-state.md +275 -0
  264. package/skills/playwright-cli/references/test-generation.md +433 -0
  265. package/skills/playwright-cli/references/tracing.md +139 -0
  266. package/skills/playwright-cli/references/video-recording.md +143 -0
  267. package/skills/playwright-cli-case-generator/SKILL.md +74 -0
  268. package/skills/requesting-code-review/SKILL.md +101 -101
  269. package/skills/requesting-code-review/code-reviewer.md +168 -168
  270. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  271. package/skills/systematic-debugging/SKILL.md +296 -296
  272. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  273. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  274. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  275. package/skills/systematic-debugging/find-polluter.sh +63 -63
  276. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  277. package/skills/systematic-debugging/test-academic.md +14 -14
  278. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  279. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  280. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  281. package/skills/test-driven-development/SKILL.md +20 -20
  282. package/skills/using-git-worktrees/SKILL.md +215 -215
  283. package/skills/verification-before-completion/SKILL.md +154 -154
  284. package/skills/webapp-testing/SKILL.md +19 -19
@@ -0,0 +1,130 @@
1
+ import { z } from "zod";
2
+ export const evalSplitSchema = z.enum(["public", "private", "held_out"]);
3
+ export const replayEvidenceRefSchema = z
4
+ .object({
5
+ candidateId: z.string().min(1),
6
+ taskRef: z.string().min(1),
7
+ seed: z.number().int().nonnegative(),
8
+ split: evalSplitSchema,
9
+ runId: z
10
+ .string()
11
+ .regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/, "runId must be one safe path segment"),
12
+ stateSha256: z.string().regex(/^[a-f0-9]{64}$/),
13
+ runSha256: z.string().regex(/^[a-f0-9]{64}$/),
14
+ })
15
+ .strict();
16
+ export const replaySpecSchema = z
17
+ .object({
18
+ schemaVersion: z.literal(1),
19
+ replayId: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/),
20
+ incumbentCandidateId: z.string().min(1),
21
+ challengerCandidateId: z.string().min(1),
22
+ evidence: z.array(replayEvidenceRefSchema).min(1),
23
+ })
24
+ .strict()
25
+ .superRefine((spec, ctx) => {
26
+ if (spec.incumbentCandidateId === spec.challengerCandidateId) {
27
+ ctx.addIssue({
28
+ code: z.ZodIssueCode.custom,
29
+ message: "incumbentCandidateId and challengerCandidateId must differ",
30
+ path: ["challengerCandidateId"],
31
+ });
32
+ }
33
+ const allowed = new Set([
34
+ spec.incumbentCandidateId,
35
+ spec.challengerCandidateId,
36
+ ]);
37
+ const pairKeys = new Set();
38
+ for (let i = 0; i < spec.evidence.length; i += 1) {
39
+ const evidence = spec.evidence[i];
40
+ if (!allowed.has(evidence.candidateId)) {
41
+ ctx.addIssue({
42
+ code: z.ZodIssueCode.custom,
43
+ message: "evidence candidateId must match incumbent or challenger",
44
+ path: ["evidence", i, "candidateId"],
45
+ });
46
+ }
47
+ const key = `${evidence.candidateId}\u0000${evidence.split}\u0000${evidence.taskRef}\u0000${evidence.seed}`;
48
+ if (pairKeys.has(key)) {
49
+ ctx.addIssue({
50
+ code: z.ZodIssueCode.custom,
51
+ message: "duplicate evidence for candidateId + split + taskRef + seed",
52
+ path: ["evidence", i],
53
+ });
54
+ }
55
+ pairKeys.add(key);
56
+ }
57
+ });
58
+ // --- Candidate Registry (M2 W2.1–W2.2) ---
59
+ export const candidateIdSchema = z
60
+ .string()
61
+ .regex(/^[A-Za-z0-9][A-Za-z0-9._-]*$/, "candidateId must be one safe path segment");
62
+ export const candidateKindSchema = z.enum([
63
+ "prompt",
64
+ "skill",
65
+ "context_policy",
66
+ "model_routing",
67
+ "profile",
68
+ "composite",
69
+ ]);
70
+ export const candidateContentRefSchema = z
71
+ .object({
72
+ path: z
73
+ .string()
74
+ .min(1)
75
+ .refine((value) => !pathIsAbsoluteLike(value), "content ref path must be repo-relative"),
76
+ sha256: z
77
+ .string()
78
+ .regex(/^(sha256:)?[a-f0-9]{64}$/i, "sha256 must be 64 hex digits"),
79
+ })
80
+ .strict();
81
+ function pathIsAbsoluteLike(value) {
82
+ if (value.startsWith("/") || value.startsWith("\\"))
83
+ return true;
84
+ if (/^[A-Za-z]:[\\/]/.test(value))
85
+ return true;
86
+ return false;
87
+ }
88
+ export const candidateManifestInputSchema = z
89
+ .object({
90
+ schemaVersion: z.literal(1),
91
+ candidateId: candidateIdSchema,
92
+ parentCandidateId: candidateIdSchema.nullable().optional(),
93
+ candidateKind: candidateKindSchema,
94
+ createdAt: z.string().min(1),
95
+ description: z.string().optional(),
96
+ contentRefs: z.array(candidateContentRefSchema).min(1),
97
+ // optional on input; always computed/verified on register/read
98
+ bundleHash: z
99
+ .string()
100
+ .regex(/^(sha256:)?[a-f0-9]{64}$/i)
101
+ .optional(),
102
+ })
103
+ .strict();
104
+ export const candidateManifestSchema = candidateManifestInputSchema
105
+ .extend({
106
+ parentCandidateId: candidateIdSchema.nullable(),
107
+ bundleHash: z.string().regex(/^sha256:[a-f0-9]{64}$/),
108
+ })
109
+ .strict();
110
+ export const lifecycleStateSchema = z.enum([
111
+ "proposed",
112
+ "eligible",
113
+ "experimenting",
114
+ "accepted",
115
+ "rejected",
116
+ "invalid",
117
+ "retired",
118
+ ]);
119
+ export const lifecycleEventSchema = z
120
+ .object({
121
+ schemaVersion: z.literal(1),
122
+ seq: z.number().int().positive(),
123
+ from: lifecycleStateSchema.nullable(),
124
+ to: lifecycleStateSchema,
125
+ reason: z.string().min(1),
126
+ at: z.string().min(1),
127
+ previousEventHash: z.string().regex(/^[a-f0-9]{64}$/),
128
+ eventHash: z.string().regex(/^[a-f0-9]{64}$/),
129
+ })
130
+ .strict();
@@ -1,6 +1,7 @@
1
1
  import { runDoctor } from "../commands/doctor.js";
2
2
  import { runDocsArchive } from "../commands/docs-archive.js";
3
3
  import { runDocsAudit } from "../commands/docs-audit.js";
4
+ import { runEval } from "../commands/eval.js";
4
5
  import { runCoverageAudit } from "../commands/coverage-audit.js";
5
6
  import { runExamples } from "../commands/examples.js";
6
7
  import { runCloseout } from "../commands/closeout.js";
@@ -10,7 +11,7 @@ import { runInstructions } from "../commands/instructions.js";
10
11
  import { runNewTask } from "../commands/new-task.js";
11
12
  import { runImportPrd } from "../commands/import-prd.js";
12
13
  import { runPlanList } from "../commands/plan-list.js";
13
- import { runPlanCheck, runPlanComplete, runPlanCreate } from "../commands/plan.js";
14
+ import { runPlanCheck, runPlanComplete, runPlanCreate, } from "../commands/plan.js";
14
15
  import { runPromoteRun } from "../commands/promote-run.js";
15
16
  import { runSpine } from "../commands/spine.js";
16
17
  import { runStats } from "../commands/stats.js";
@@ -76,6 +77,7 @@ const INIT_SUBCOMMANDS = [
76
77
  "update",
77
78
  ];
78
79
  const EXAMPLES_SUBCOMMANDS = ["list", "show", "copy"];
80
+ const EVAL_SUBCOMMANDS = ["replay", "report", "candidate"];
79
81
  const CLOSEOUT_SUBCOMMANDS = ["task"];
80
82
  const PLAN_SUBCOMMANDS = ["list", "create", "complete", "check"];
81
83
  const SPINE_SUBCOMMANDS = ["audit"];
@@ -84,7 +86,14 @@ const HANDOFF_USAGE = "handoff <check|coverage> [taskId]";
84
86
  const REFERENCE_SUBCOMMANDS = ["index"];
85
87
  const STUDY_SUBCOMMANDS = ["init"];
86
88
  const WORKTREE_SUBCOMMANDS = ["create", "list", "remove"];
87
- const KNOWLEDGE_SUBCOMMANDS = ["curate"];
89
+ const KNOWLEDGE_SUBCOMMANDS = [
90
+ "curate",
91
+ "query",
92
+ "graph-init",
93
+ "graph-materialize",
94
+ "graph-promote",
95
+ "graph-incremental-prepare",
96
+ ];
88
97
  const DAG_SUBCOMMANDS = [
89
98
  "init-hybrid",
90
99
  "run-task",
@@ -187,6 +196,17 @@ export const COMMAND_DEFINITIONS = [
187
196
  await runExamples(repoRoot, [subcommand, ...rest].filter(Boolean));
188
197
  },
189
198
  },
199
+ {
200
+ name: "eval",
201
+ adapter: "required",
202
+ tier: "operator",
203
+ intent: "Replay completed DAG evidence and manage immutable Candidate Registry lifecycle without live model execution or promotion.",
204
+ usage: "eval <replay|report|candidate> ...; candidate <register|show|list|transition> [--json|--markdown]",
205
+ subcommands: [...EVAL_SUBCOMMANDS],
206
+ handler: async ({ repoRoot, subcommand, rest }) => {
207
+ await runEval(repoRoot, [subcommand, ...rest].filter((arg) => Boolean(arg)));
208
+ },
209
+ },
190
210
  {
191
211
  name: "new-task",
192
212
  adapter: "required",
@@ -279,17 +299,17 @@ export const COMMAND_DEFINITIONS = [
279
299
  if (subcommand === "create") {
280
300
  const [planId, ...titleParts] = rest;
281
301
  if (!planId)
282
- throw new Error("usage: plan create <plan-id> \"<title>\"");
302
+ throw new Error('usage: plan create <plan-id> "<title>"');
283
303
  await runPlanCreate(repoRoot, planId, titleParts.join(" "));
284
304
  return;
285
305
  }
286
306
  if (subcommand === "complete") {
287
307
  const [planId, ...summaryParts] = rest;
288
308
  if (!planId)
289
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
309
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
290
310
  const summary = parseSummaryFlag(summaryParts);
291
311
  if (!summary)
292
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
312
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
293
313
  await runPlanComplete(repoRoot, planId, { summary });
294
314
  return;
295
315
  }
@@ -435,8 +455,8 @@ export const COMMAND_DEFINITIONS = [
435
455
  name: "knowledge",
436
456
  adapter: "required",
437
457
  tier: "operator",
438
- intent: "Curate completed convergence patterns into human-gated learned guidance proposals.",
439
- usage: "knowledge curate [--json|--markdown] [--output <path>]",
458
+ intent: "Curate repair guidance proposals, query the structured knowledge graph, and run graph bootstrap helpers (init/materialize/promote/incremental).",
459
+ usage: "knowledge <curate|query|graph-init|graph-materialize|graph-promote|graph-incremental-prepare> ...",
440
460
  subcommands: [...KNOWLEDGE_SUBCOMMANDS],
441
461
  handler: async ({ repoRoot, subcommand, rest }) => {
442
462
  await runKnowledge(repoRoot, [subcommand, ...rest].filter((arg) => Boolean(arg)));
@@ -22,6 +22,7 @@ import { runDagWorkflowValidate } from "../commands/dag-workflow-validate.js";
22
22
  import { runDelegate } from "../commands/delegate.js";
23
23
  import { runDocsArchive } from "../commands/docs-archive.js";
24
24
  import { runDocsAudit } from "../commands/docs-audit.js";
25
+ import { runEval } from "../commands/eval.js";
25
26
  import { runDoctor } from "../commands/doctor.js";
26
27
  import { runExamples } from "../commands/examples.js";
27
28
  import { runGoal } from "../commands/goal.js";
@@ -38,7 +39,7 @@ import { runImportPrd } from "../commands/import-prd.js";
38
39
  import { runPiReuseBenchmark } from "../commands/pi-reuse-benchmark.js";
39
40
  import { parsePiPromptArgs, printPiPromptUsage, runPiPrompt, } from "../commands/pi-prompt.js";
40
41
  import { runPlanList } from "../commands/plan-list.js";
41
- import { runPlanCheck, runPlanComplete, runPlanCreate } from "../commands/plan.js";
42
+ import { runPlanCheck, runPlanComplete, runPlanCreate, } from "../commands/plan.js";
42
43
  import { runPromoteRun } from "../commands/promote-run.js";
43
44
  import { runReferenceIndex } from "../commands/reference-index.js";
44
45
  import { runRunDag } from "../commands/run-dag.js";
@@ -163,17 +164,17 @@ async function runCommanderAction(ctx, command, subcommand, rest) {
163
164
  if (subcommand === "create") {
164
165
  const [planId, ...titleParts] = rest;
165
166
  if (!planId)
166
- throw new Error("usage: plan create <plan-id> \"<title>\"");
167
+ throw new Error('usage: plan create <plan-id> "<title>"');
167
168
  await runPlanCreate(ctx.repoRoot, planId, titleParts.join(" "));
168
169
  return;
169
170
  }
170
171
  if (subcommand === "complete") {
171
172
  const [planId, ...summaryParts] = rest;
172
173
  if (!planId)
173
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
174
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
174
175
  const summary = parsePlanSummaryFlag(summaryParts);
175
176
  if (!summary)
176
- throw new Error("usage: plan complete <plan-id> --summary \"<summary>\"");
177
+ throw new Error('usage: plan complete <plan-id> --summary "<summary>"');
177
178
  await runPlanComplete(ctx.repoRoot, planId, { summary });
178
179
  return;
179
180
  }
@@ -244,6 +245,9 @@ async function runCommanderAction(ctx, command, subcommand, rest) {
244
245
  case "knowledge":
245
246
  await runKnowledge(ctx.repoRoot, compactArgs([subcommand, ...rest]));
246
247
  return;
248
+ case "eval":
249
+ await runEval(ctx.repoRoot, compactArgs([subcommand, ...rest]));
250
+ return;
247
251
  case "dag":
248
252
  await runDagAction(ctx.repoRoot, subcommand, rest);
249
253
  return;
@@ -141,7 +141,7 @@ async function runCursorPromptBatch(task, cwd, model, timeoutMs) {
141
141
  }
142
142
  async function runCursorPromptStreaming(task, cwd, model, timeoutMs) {
143
143
  const startedAt = Date.now();
144
- process.stderr.write(`[cursor-prompt] streaming (model=${model}, cwd=${cwd})
144
+ process.stderr.write(`[cursor-prompt] streaming (model=${model}, cwd=${cwd})
145
145
  `);
146
146
  const runDir = (await computeRunDir(cwd, task)) ?? undefined;
147
147
  const result = await executeCursorPromptStream({
@@ -164,15 +164,15 @@ async function runCursorPromptStreaming(task, cwd, model, timeoutMs) {
164
164
  });
165
165
  const elapsed = ((Date.now() - startedAt) / 1000).toFixed(1);
166
166
  if (result.ok) {
167
- process.stderr.write(`
168
- [cursor-prompt] done in ${elapsed}s, status=${result.status}
167
+ process.stderr.write(`
168
+ [cursor-prompt] done in ${elapsed}s, status=${result.status}
169
169
  `);
170
170
  }
171
171
  else {
172
172
  const stderr = result.stderr || "(no output)";
173
- process.stderr.write(`
174
- [cursor-prompt] FAILED in ${elapsed}s (${result.failureCategory}):
175
- ${stderr}
173
+ process.stderr.write(`
174
+ [cursor-prompt] FAILED in ${elapsed}s (${result.failureCategory}):
175
+ ${stderr}
176
176
  `);
177
177
  process.exit(1);
178
178
  }
@@ -0,0 +1,235 @@
1
+ import path from "node:path";
2
+ import { writeTextAtomic } from "../infrastructure/harness/atomic-write.js";
3
+ import { readReplayScorecard } from "../infrastructure/evaluation/store.js";
4
+ import { formatReplayMarkdown, replayEvaluation, } from "../application/evaluation/replay.js";
5
+ import { formatCandidateMarkdown, listCandidates, registerCandidate, showCandidate, transitionCandidate, } from "../application/evaluation/candidate.js";
6
+ import { lifecycleStateSchema, } from "../application/evaluation/types.js";
7
+ const USAGE = "usage: eval <replay|report|candidate> ...; candidate <register|show|list|transition> [--json|--markdown]";
8
+ function parseFormatFlags(args) {
9
+ const json = args.includes("--json");
10
+ const markdown = args.includes("--markdown");
11
+ if (json && markdown) {
12
+ throw new Error("eval accepts only one of --json or --markdown");
13
+ }
14
+ return { json: json || !markdown, markdown };
15
+ }
16
+ function flagValue(args, flag) {
17
+ const index = args.indexOf(flag);
18
+ if (index >= 0) {
19
+ const value = args[index + 1];
20
+ if (!value || value.startsWith("-")) {
21
+ throw new Error(`${flag} requires a value`);
22
+ }
23
+ return value;
24
+ }
25
+ const prefix = `${flag}=`;
26
+ return args.find((arg) => arg.startsWith(prefix))?.slice(prefix.length);
27
+ }
28
+ function assertKnownFlags(args, allowed) {
29
+ for (let i = 0; i < args.length; i += 1) {
30
+ const arg = args[i];
31
+ if (!arg.startsWith("-"))
32
+ continue;
33
+ const key = arg.split("=", 1)[0];
34
+ if (!allowed.includes(key)) {
35
+ throw new Error(`unknown eval argument: ${arg}`);
36
+ }
37
+ if ([
38
+ "--spec",
39
+ "--output",
40
+ "--replay-id",
41
+ "--manifest",
42
+ "--candidate-id",
43
+ "--to",
44
+ "--reason",
45
+ ].includes(key) &&
46
+ !arg.includes("=")) {
47
+ i += 1;
48
+ }
49
+ }
50
+ }
51
+ function parseCandidateArgs(rest) {
52
+ const [action, ...tail] = rest;
53
+ if (action === "register") {
54
+ assertKnownFlags(tail, ["--manifest", "--json", "--markdown"]);
55
+ const manifestPath = flagValue(tail, "--manifest");
56
+ if (!manifestPath) {
57
+ throw new Error("eval candidate register requires --manifest <path>");
58
+ }
59
+ return {
60
+ command: "candidate",
61
+ action: "register",
62
+ manifestPath,
63
+ ...parseFormatFlags(tail),
64
+ };
65
+ }
66
+ if (action === "show") {
67
+ assertKnownFlags(tail, ["--candidate-id", "--json", "--markdown"]);
68
+ const candidateId = flagValue(tail, "--candidate-id");
69
+ if (!candidateId) {
70
+ throw new Error("eval candidate show requires --candidate-id <id>");
71
+ }
72
+ return {
73
+ command: "candidate",
74
+ action: "show",
75
+ candidateId,
76
+ ...parseFormatFlags(tail),
77
+ };
78
+ }
79
+ if (action === "list") {
80
+ assertKnownFlags(tail, ["--json", "--markdown"]);
81
+ return {
82
+ command: "candidate",
83
+ action: "list",
84
+ ...parseFormatFlags(tail),
85
+ };
86
+ }
87
+ if (action === "transition") {
88
+ assertKnownFlags(tail, [
89
+ "--candidate-id",
90
+ "--to",
91
+ "--reason",
92
+ "--json",
93
+ "--markdown",
94
+ ]);
95
+ const candidateId = flagValue(tail, "--candidate-id");
96
+ const toRaw = flagValue(tail, "--to");
97
+ const reason = flagValue(tail, "--reason");
98
+ if (!candidateId) {
99
+ throw new Error("eval candidate transition requires --candidate-id <id>");
100
+ }
101
+ if (!toRaw) {
102
+ throw new Error("eval candidate transition requires --to <state>");
103
+ }
104
+ if (!reason) {
105
+ throw new Error("eval candidate transition requires --reason <text>");
106
+ }
107
+ const to = lifecycleStateSchema.parse(toRaw);
108
+ return {
109
+ command: "candidate",
110
+ action: "transition",
111
+ candidateId,
112
+ to,
113
+ reason,
114
+ ...parseFormatFlags(tail),
115
+ };
116
+ }
117
+ throw new Error("usage: eval candidate <register|show|list|transition> ...");
118
+ }
119
+ export function parseEvalArgs(args) {
120
+ const [command, ...rest] = args;
121
+ if (command === "replay") {
122
+ assertKnownFlags(rest, ["--spec", "--output", "--json", "--markdown"]);
123
+ const specPath = flagValue(rest, "--spec");
124
+ if (!specPath)
125
+ throw new Error("eval replay requires --spec <path>");
126
+ return {
127
+ command,
128
+ specPath,
129
+ outputPath: flagValue(rest, "--output"),
130
+ ...parseFormatFlags(rest),
131
+ };
132
+ }
133
+ if (command === "report") {
134
+ assertKnownFlags(rest, ["--replay-id", "--json", "--markdown"]);
135
+ const replayId = flagValue(rest, "--replay-id");
136
+ if (!replayId)
137
+ throw new Error("eval report requires --replay-id <id>");
138
+ return { command, replayId, ...parseFormatFlags(rest) };
139
+ }
140
+ if (command === "candidate") {
141
+ return parseCandidateArgs(rest);
142
+ }
143
+ throw new Error(USAGE);
144
+ }
145
+ function printScorecard(input) {
146
+ if (input.json) {
147
+ console.log(JSON.stringify(input.scorecard, null, 2));
148
+ return;
149
+ }
150
+ process.stdout.write(input.markdown);
151
+ }
152
+ function printCandidate(input) {
153
+ if (input.json) {
154
+ console.log(JSON.stringify(input.record, null, 2));
155
+ return;
156
+ }
157
+ process.stdout.write(formatCandidateMarkdown(input.record));
158
+ }
159
+ export async function runEval(repoRoot, args) {
160
+ const parsed = parseEvalArgs(args);
161
+ if (parsed.command === "replay") {
162
+ const result = await replayEvaluation({
163
+ repoRoot,
164
+ specPath: parsed.specPath,
165
+ });
166
+ if (parsed.outputPath) {
167
+ const outputPath = path.resolve(repoRoot, parsed.outputPath);
168
+ await writeTextAtomic(outputPath, parsed.markdown
169
+ ? result.markdown
170
+ : `${JSON.stringify(result.scorecard, null, 2)}\n`, { repoRoot });
171
+ }
172
+ printScorecard({
173
+ scorecard: result.scorecard,
174
+ markdown: result.markdown,
175
+ json: parsed.json,
176
+ });
177
+ return;
178
+ }
179
+ if (parsed.command === "report") {
180
+ const scorecard = (await readReplayScorecard(repoRoot, parsed.replayId));
181
+ const markdown = formatReplayMarkdown(scorecard);
182
+ printScorecard({ scorecard, markdown, json: parsed.json });
183
+ return;
184
+ }
185
+ // candidate subcommands
186
+ if (parsed.action === "register") {
187
+ const result = await registerCandidate({
188
+ repoRoot,
189
+ manifestPath: parsed.manifestPath,
190
+ });
191
+ if (parsed.json) {
192
+ console.log(JSON.stringify({
193
+ idempotent: result.idempotent,
194
+ manifestPath: result.manifestPath,
195
+ lifecyclePath: result.lifecyclePath,
196
+ record: result.record,
197
+ }, null, 2));
198
+ return;
199
+ }
200
+ process.stdout.write(`${result.idempotent ? "idempotent " : ""}registered ${result.record.manifest.candidateId}\n${formatCandidateMarkdown(result.record)}`);
201
+ return;
202
+ }
203
+ if (parsed.action === "show") {
204
+ const record = await showCandidate({
205
+ repoRoot,
206
+ candidateId: parsed.candidateId,
207
+ });
208
+ printCandidate({ record, json: parsed.json });
209
+ return;
210
+ }
211
+ if (parsed.action === "list") {
212
+ const rows = await listCandidates({ repoRoot });
213
+ if (parsed.json) {
214
+ console.log(JSON.stringify(rows, null, 2));
215
+ return;
216
+ }
217
+ const lines = [
218
+ "# Candidates",
219
+ "",
220
+ ...rows.map((row) => `- \`${row.candidateId}\` status=\`${row.status}\` hash=\`${row.bundleHash}\` promotionApplied=false`),
221
+ "",
222
+ ];
223
+ process.stdout.write(`${lines.join("\n")}\n`);
224
+ return;
225
+ }
226
+ if (parsed.action === "transition") {
227
+ const record = await transitionCandidate({
228
+ repoRoot,
229
+ candidateId: parsed.candidateId,
230
+ to: parsed.to,
231
+ reason: parsed.reason,
232
+ });
233
+ printCandidate({ record, json: parsed.json });
234
+ }
235
+ }