@tea-agent/loop-agent 0.12.0 → 0.13.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (284) hide show
  1. package/AGENTS.md +155 -153
  2. package/CHANGELOG.md +338 -265
  3. package/README.md +345 -298
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/generate-task-dag.js +28 -28
  7. package/dist/application/evaluation/candidate-hash.js +75 -0
  8. package/dist/application/evaluation/candidate.js +52 -0
  9. package/dist/application/evaluation/replay.js +289 -0
  10. package/dist/application/evaluation/types.js +130 -0
  11. package/dist/cli/command-definitions.js +27 -7
  12. package/dist/cli/program.js +8 -4
  13. package/dist/commands/cursor-prompt.js +6 -6
  14. package/dist/commands/eval.js +235 -0
  15. package/dist/commands/init.js +544 -506
  16. package/dist/commands/knowledge.js +129 -31
  17. package/dist/commands/loop-benchmark.js +11 -11
  18. package/dist/commands/pi-reuse-benchmark.js +16 -16
  19. package/dist/executors/pi-sdk-executor.js +38 -24
  20. package/dist/executors/shell-executor.js +34 -2
  21. package/dist/executors/shell-presets.js +20 -0
  22. package/dist/executors/shell-verification.js +7 -0
  23. package/dist/governance/manifest-types.js +4 -0
  24. package/dist/infrastructure/evaluation/candidate-store.js +435 -0
  25. package/dist/infrastructure/evaluation/store.js +40 -0
  26. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  27. package/dist/task/config-types.js +28 -1
  28. package/dist/task/runtime.js +27 -27
  29. package/dist/worker/cli.js +96 -1
  30. package/dist/worker/delivery/package.js +3 -3
  31. package/dist/worker/feature/decision-loader.js +37 -6
  32. package/dist/worker/feature/next-action.js +10 -2
  33. package/dist/worker/feature/ready-plan-projection.js +81 -0
  34. package/dist/worker/feature/reducer.js +2 -1
  35. package/dist/worker/feature/review.js +19 -2
  36. package/dist/worker/feature/run.js +27 -2
  37. package/dist/worker/follow-up/approve.js +5 -2
  38. package/dist/worker/follow-up/factory.js +1 -1
  39. package/dist/worker/observability/read-model.js +246 -41
  40. package/dist/worker/observe/routes.js +173 -15
  41. package/dist/worker/observe/spec-evidence.js +281 -0
  42. package/dist/worker/observe/static/api.js +46 -27
  43. package/dist/worker/observe/static/app.js +150 -150
  44. package/dist/worker/observe/static/constants.js +148 -148
  45. package/dist/worker/observe/static/copy.js +67 -67
  46. package/dist/worker/observe/static/dag-helpers.js +172 -172
  47. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  48. package/dist/worker/observe/static/dag-layout.js +83 -83
  49. package/dist/worker/observe/static/dag-model.js +72 -72
  50. package/dist/worker/observe/static/dom.js +61 -61
  51. package/dist/worker/observe/static/format-pool.js +67 -67
  52. package/dist/worker/observe/static/format.js +292 -292
  53. package/dist/worker/observe/static/index.html +308 -308
  54. package/dist/worker/observe/static/kpi.js +94 -94
  55. package/dist/worker/observe/static/relations.js +133 -128
  56. package/dist/worker/observe/static/router.js +93 -85
  57. package/dist/worker/observe/static/run-processing.js +148 -148
  58. package/dist/worker/observe/static/shell-chrome.js +68 -68
  59. package/dist/worker/observe/static/state.js +253 -253
  60. package/dist/worker/observe/static/styles.css +1902 -1890
  61. package/dist/worker/observe/static/views/batch.js +227 -226
  62. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  63. package/dist/worker/observe/static/views/dag-inspector.js +607 -477
  64. package/dist/worker/observe/static/views/dag.js +362 -362
  65. package/dist/worker/observe/static/views/dashboard.js +445 -442
  66. package/dist/worker/observe/static/views/failures.js +143 -143
  67. package/dist/worker/observe/static/views/feature.js +492 -453
  68. package/dist/worker/observe/static/views/pool.js +350 -347
  69. package/dist/worker/observe/static/views/run.js +453 -453
  70. package/dist/worker/observe/static/views/session-timeline.js +205 -205
  71. package/dist/worker/observe/static/views/shell.js +7 -7
  72. package/dist/worker/observe/static/views/task.js +314 -260
  73. package/dist/worker/observe/static/views/timeline.js +163 -163
  74. package/dist/worker/pool/doctor.js +165 -0
  75. package/dist/worker/pool/migrate-state.js +303 -0
  76. package/dist/worker/pool/run-store.js +205 -17
  77. package/dist/worker/pool/types.js +17 -1
  78. package/dist/worker/pool/validation.js +100 -15
  79. package/dist/worker/report/morning-report.js +12 -2
  80. package/dist/worker/runner/run-ready.js +41 -26
  81. package/dist/worker/task-graph/ready-planner.js +136 -0
  82. package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
  83. package/dist/workflows/dag/canvas-observer.js +275 -275
  84. package/dist/workflows/dag/convergence/controller.js +16 -8
  85. package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
  86. package/dist/workflows/dag/failure-routing.js +12 -1
  87. package/dist/workflows/dag/init-hybrid.js +2404 -360
  88. package/dist/workflows/dag/node-execution.js +9 -0
  89. package/dist/workflows/dag/prompt.js +9 -0
  90. package/dist/workflows/dag/report.js +35 -1
  91. package/dist/workflows/dag/runner.js +28 -2
  92. package/dist/workflows/dag/task-demand-routing.js +383 -0
  93. package/dist/workflows/dag/types.js +51 -13
  94. package/dist/workflows/dag/upstream-artifacts.js +1 -0
  95. package/dist/workflows/dag/validate.js +59 -1
  96. package/docs/README.md +106 -104
  97. package/docs/agent-dag-recovery-playbook.md +195 -184
  98. package/docs/agent-dag-runner.md +67 -67
  99. package/docs/architecture/README.md +26 -26
  100. package/docs/architecture/dag-execution.md +140 -140
  101. package/docs/architecture/evolution.md +54 -53
  102. package/docs/architecture/facts-and-state.md +71 -58
  103. package/docs/architecture/runtime-boundaries.md +191 -191
  104. package/docs/architecture/system-overview.md +93 -93
  105. package/docs/architecture/worker-and-feature.md +85 -81
  106. package/docs/cursor-prompt-sidecar.md +36 -36
  107. package/docs/decisions/README.md +18 -15
  108. package/docs/design/README.md +167 -77
  109. package/docs/development-principles.md +73 -73
  110. package/docs/exec-plans/README.md +6 -6
  111. package/docs/exec-plans/active/README.md +15 -9
  112. package/docs/exec-plans/completed/README.md +85 -73
  113. package/docs/feature-workflow.md +389 -261
  114. package/docs/harness-methodology-debugging.md +153 -153
  115. package/docs/harness-methodology-tdd.md +130 -130
  116. package/docs/harness-methodology-verification.md +27 -27
  117. package/docs/init-surface.manifest.json +289 -280
  118. package/docs/loop-agent-harness.md +142 -130
  119. package/docs/production-readiness.md +96 -96
  120. package/docs/progress/README.md +64 -54
  121. package/docs/reports/README.md +117 -94
  122. package/docs/skills/README.md +7 -7
  123. package/docs/skills/vetted-skill-registry.md +29 -27
  124. package/docs/templates/adr.md +60 -60
  125. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  126. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  127. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  128. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  129. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  130. package/docs/templates/agent-dag-report.schema.json +473 -473
  131. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  132. package/docs/templates/agent-dag.base.json +190 -190
  133. package/docs/templates/agent-dag.final-verification.json +185 -185
  134. package/docs/templates/agent-dag.schema.json +411 -383
  135. package/docs/templates/agent-dag.supervised-implementation.json +501 -501
  136. package/docs/templates/backend-test-analysis.schema.json +44 -0
  137. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
  138. package/docs/templates/backend-test-dag.json +311 -276
  139. package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
  140. package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
  141. package/docs/templates/exec-plan.md +64 -64
  142. package/docs/templates/feature-spec.md +53 -53
  143. package/docs/templates/frontend-design-contract.md +42 -33
  144. package/docs/templates/frontend-task-constraints.md +35 -25
  145. package/docs/templates/frontend-task-requirement.md +70 -61
  146. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
  147. package/docs/templates/frontend-test-dag.json +23 -0
  148. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
  149. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
  150. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
  151. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
  152. package/docs/templates/harness.schema.json +221 -221
  153. package/docs/templates/hybrid-dag.json +188 -188
  154. package/docs/templates/init-evolution-review.md +35 -35
  155. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  156. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -0
  157. package/docs/templates/knowledge-sync-dag.json +178 -0
  158. package/docs/templates/knowledge-sync-draft.schema.json +71 -0
  159. package/docs/templates/product-line/AGENTS.md +8 -8
  160. package/docs/templates/product-line/README.md +9 -9
  161. package/docs/templates/product-line/acceptance.yaml +14 -14
  162. package/docs/templates/product-line/closeout.yaml +9 -9
  163. package/docs/templates/product-line/design.md +13 -13
  164. package/docs/templates/product-line/links.md +10 -10
  165. package/docs/templates/product-line/requirement.md +17 -17
  166. package/docs/templates/product-line/task-graph.yaml +15 -15
  167. package/docs/templates/product-line/task.yaml +64 -64
  168. package/docs/templates/product-line/test-plan.md +7 -7
  169. package/docs/templates/production-readiness-checklist.md +57 -57
  170. package/docs/templates/progress-log.md +17 -17
  171. package/docs/templates/project-start-checklist.md +9 -9
  172. package/docs/templates/qa-report.md +48 -48
  173. package/docs/templates/sprint-contract.md +29 -29
  174. package/docs/templates/worker-dogfood-evidence.md +80 -80
  175. package/docs/templates/worker-dogfood-setup.md +68 -68
  176. package/docs/verification-matrix.md +70 -66
  177. package/examples/decision-gate-agent-dag.json +177 -177
  178. package/examples/example-dag.json +46 -46
  179. package/examples/hybrid-loop-agent-dag.json +189 -189
  180. package/harness.json +66 -66
  181. package/package.json +88 -46
  182. package/scripts/check-product-line-docs.sh +29 -29
  183. package/scripts/check-task-pool-root.sh +32 -32
  184. package/scripts/kb-bootstrap-init-skeleton.sh +240 -0
  185. package/scripts/kb-graph-incremental-prepare.mjs +386 -0
  186. package/scripts/kb-graph-incremental-prepare.sh +5 -0
  187. package/scripts/kb-graph-materialize.mjs +105 -0
  188. package/scripts/kb-graph-materialize.sh +4 -0
  189. package/scripts/kb-graph-promote.mjs +164 -0
  190. package/scripts/kb-graph-promote.sh +4 -0
  191. package/scripts/kb-query.mjs +554 -0
  192. package/scripts/kb-query.sh +5 -0
  193. package/skills/agent-worker/SKILL.md +39 -37
  194. package/skills/agent-worker/references/agent-worker-operator.md +60 -43
  195. package/skills/ai-engineering-context/SKILL.md +48 -48
  196. package/skills/analyze-product-dependencies/SKILL.md +67 -0
  197. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
  198. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
  199. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
  200. package/skills/analyze-product-dependencies/references/example.md +76 -0
  201. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
  202. package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
  203. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
  204. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
  205. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
  206. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
  207. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
  208. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
  209. package/skills/analyze-product-requirements/SKILL.md +90 -0
  210. package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
  211. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
  212. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
  213. package/skills/analyze-product-requirements/references/example.md +86 -0
  214. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
  215. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
  216. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
  217. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
  218. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
  219. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
  220. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
  221. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
  222. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
  223. package/skills/code-review-core/SKILL.md +20 -20
  224. package/skills/codebase-scout/SKILL.md +19 -19
  225. package/skills/frontend-design-review/SKILL.md +66 -59
  226. package/skills/frontend-design-review/references/review-checklist.md +58 -37
  227. package/skills/frontend-implementation/SKILL.md +47 -51
  228. package/skills/frontend-implementation/references/code-standards.md +32 -34
  229. package/skills/frontend-implementation/references/design-spec.md +46 -46
  230. package/skills/frontend-implementation/references/node-contracts.md +76 -32
  231. package/skills/frontend-review/SKILL.md +59 -53
  232. package/skills/frontend-review/references/review-findings.md +47 -42
  233. package/skills/frontend-verification/SKILL.md +53 -40
  234. package/skills/frontend-verification/references/verification-checklist.md +68 -56
  235. package/skills/grill-me/SKILL.md +10 -10
  236. package/skills/grill-with-docs/SKILL.md +88 -88
  237. package/skills/grill-with-docs/adr-format.md +47 -47
  238. package/skills/grill-with-docs/context-format.md +60 -60
  239. package/skills/init-capability-evolution/SKILL.md +70 -70
  240. package/skills/loop-agent/SKILL.md +151 -151
  241. package/skills/loop-agent/references/README.md +67 -67
  242. package/skills/loop-agent/references/command-reference.md +505 -452
  243. package/skills/loop-agent/references/docs-converge.md +126 -126
  244. package/skills/loop-agent/references/harness-policy.md +263 -263
  245. package/skills/loop-agent/references/hybrid-dag.md +238 -233
  246. package/skills/loop-agent/references/learned/README.md +21 -21
  247. package/skills/loop-agent/references/long-running-loop.md +57 -57
  248. package/skills/loop-agent/references/model-routing.md +36 -36
  249. package/skills/loop-agent/references/multi-worktree.md +54 -54
  250. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  251. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  252. package/skills/loop-agent/references/pi-prompt.md +23 -23
  253. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  254. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  255. package/skills/loop-agent/references/task-workflow.md +89 -89
  256. package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
  257. package/skills/playwright-cli/SKILL.md +420 -0
  258. package/skills/playwright-cli/references/element-attributes.md +23 -0
  259. package/skills/playwright-cli/references/playwright-tests.md +39 -0
  260. package/skills/playwright-cli/references/request-mocking.md +87 -0
  261. package/skills/playwright-cli/references/running-code.md +241 -0
  262. package/skills/playwright-cli/references/session-management.md +225 -0
  263. package/skills/playwright-cli/references/storage-state.md +275 -0
  264. package/skills/playwright-cli/references/test-generation.md +433 -0
  265. package/skills/playwright-cli/references/tracing.md +139 -0
  266. package/skills/playwright-cli/references/video-recording.md +143 -0
  267. package/skills/playwright-cli-case-generator/SKILL.md +74 -0
  268. package/skills/requesting-code-review/SKILL.md +101 -101
  269. package/skills/requesting-code-review/code-reviewer.md +168 -168
  270. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  271. package/skills/systematic-debugging/SKILL.md +296 -296
  272. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  273. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  274. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  275. package/skills/systematic-debugging/find-polluter.sh +63 -63
  276. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  277. package/skills/systematic-debugging/test-academic.md +14 -14
  278. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  279. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  280. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  281. package/skills/test-driven-development/SKILL.md +20 -20
  282. package/skills/using-git-worktrees/SKILL.md +215 -215
  283. package/skills/verification-before-completion/SKILL.md +154 -154
  284. package/skills/webapp-testing/SKILL.md +19 -19
@@ -1,22 +1,22 @@
1
- #!/usr/bin/env node
2
- import { existsSync } from "node:fs";
3
- import { dirname, join } from "node:path";
4
- import { fileURLToPath, pathToFileURL } from "node:url";
5
-
6
- const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
- const cliEntry = join(packageRoot, "dist", "worker", "cli.js");
8
-
9
- if (!existsSync(cliEntry)) {
10
- console.error(
11
- `agent-worker: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
- );
13
- process.exit(1);
14
- }
15
-
16
- try {
17
- const cli = await import(pathToFileURL(cliEntry).href);
18
- await cli.main(process.argv);
19
- } catch (error) {
20
- console.error(error instanceof Error ? error.message : String(error));
21
- process.exit(1);
22
- }
1
+ #!/usr/bin/env node
2
+ import { existsSync } from "node:fs";
3
+ import { dirname, join } from "node:path";
4
+ import { fileURLToPath, pathToFileURL } from "node:url";
5
+
6
+ const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
+ const cliEntry = join(packageRoot, "dist", "worker", "cli.js");
8
+
9
+ if (!existsSync(cliEntry)) {
10
+ console.error(
11
+ `agent-worker: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
+ );
13
+ process.exit(1);
14
+ }
15
+
16
+ try {
17
+ const cli = await import(pathToFileURL(cliEntry).href);
18
+ await cli.main(process.argv);
19
+ } catch (error) {
20
+ console.error(error instanceof Error ? error.message : String(error));
21
+ process.exit(1);
22
+ }
package/bin/loop-agent.js CHANGED
@@ -1,21 +1,21 @@
1
- #!/usr/bin/env node
2
- import { existsSync } from "node:fs";
3
- import { dirname, join } from "node:path";
4
- import { fileURLToPath, pathToFileURL } from "node:url";
5
-
6
- const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
- const cliEntry = join(packageRoot, "dist", "cli.js");
8
-
9
- if (!existsSync(cliEntry)) {
10
- console.error(
11
- `loop-agent: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
- );
13
- process.exit(1);
14
- }
15
-
16
- try {
17
- await import(pathToFileURL(cliEntry).href);
18
- } catch (error) {
19
- console.error(error instanceof Error ? error.message : String(error));
20
- process.exit(1);
21
- }
1
+ #!/usr/bin/env node
2
+ import { existsSync } from "node:fs";
3
+ import { dirname, join } from "node:path";
4
+ import { fileURLToPath, pathToFileURL } from "node:url";
5
+
6
+ const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
+ const cliEntry = join(packageRoot, "dist", "cli.js");
8
+
9
+ if (!existsSync(cliEntry)) {
10
+ console.error(
11
+ `loop-agent: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
+ );
13
+ process.exit(1);
14
+ }
15
+
16
+ try {
17
+ await import(pathToFileURL(cliEntry).href);
18
+ } catch (error) {
19
+ console.error(error instanceof Error ? error.message : String(error));
20
+ process.exit(1);
21
+ }
@@ -1,4 +1,6 @@
1
- import { readFile } from "node:fs/promises";
1
+ import { readFile, mkdir } from "node:fs/promises";
2
+ import { randomUUID } from "node:crypto";
3
+ import path from "node:path";
2
4
  import { resolveAutoRoutingProfile, requiresSupervisedQualityGate, } from "../../workflows/dag/governance-profile.js";
3
5
  import { resolveShellCommands } from "../../executors/shell-executor.js";
4
6
  import { parseDagSpec } from "../../workflows/dag/types.js";
@@ -6,7 +8,6 @@ import { pathMatchesPattern } from "../../shared/git-progress.js";
6
8
  import { loadHarnessManifest } from "../../governance/harness.js";
7
9
  import { assertExecPlanIndexConsistent } from "../../governance/exec-plans.js";
8
10
  import { defaultHybridDagOutputPath, initHybridDagFromTask, } from "../../workflows/dag/init-hybrid.js";
9
- import { loadTaskConfig } from "../../task/runtime.js";
10
11
  import { validateDagUseCase } from "./validate-dag.js";
11
12
  import { runDagUseCase } from "./run-dag.js";
12
13
  const PLACEHOLDER_WRITESET_MARKER = "REPLACE/WITH";
@@ -202,9 +203,6 @@ export async function generateTaskDagUseCase(input) {
202
203
  // before any expensive DAG generation or execution. Empty/consistent
203
204
  // repos stay compatible so the default DAG flow is unblocked.
204
205
  await assertExecPlanIndexConsistent(repoRoot);
205
- const taskConfig = await loadTaskConfig(repoRoot, parsed.taskId);
206
- const isFrontendImplementationTask = taskConfig.taskKind === "frontend-implementation";
207
- const isBackendTestTask = taskConfig.taskKind === "backend-test";
208
206
  const candidateResult = await initHybridDagFromTask(repoRoot, parsed.taskId, {
209
207
  outputPath: parsed.outputPath,
210
208
  template: "standard-dag",
@@ -217,12 +215,16 @@ export async function generateTaskDagUseCase(input) {
217
215
  codeChange: [],
218
216
  reasons: ["dag run-task validate did not report governanceProfile"],
219
217
  });
220
- if (isFrontendImplementationTask) {
221
- profileRouting.selectedTemplate = "frontend-implementation";
222
- profileRouting.source = "taskKind";
223
- profileRouting.routingReasons = [
224
- 'taskKind "frontend-implementation" selects the dedicated frontend DAG template',
225
- ];
218
+ const hasExplicitSpecializedTaskKind = candidateResult.templateSelection.source === "taskKind";
219
+ const hasSafeAutomaticTaskSourceRoute = candidateResult.templateSelection.source === "task-source" &&
220
+ profileRouting.selectedByProfile !== "supervised" &&
221
+ profileRouting.selectedTemplate !== "supervised-implementation";
222
+ if (hasExplicitSpecializedTaskKind || hasSafeAutomaticTaskSourceRoute) {
223
+ profileRouting.selectedTemplate = candidateResult.template;
224
+ profileRouting.source = hasExplicitSpecializedTaskKind
225
+ ? "taskKind"
226
+ : "task-source";
227
+ profileRouting.routingReasons = candidateResult.templateSelection.reasons;
226
228
  if (parsed.profile === "auto") {
227
229
  profileRouting.selectedByProfile =
228
230
  resolveAutoRoutingProfile(profileRouting.candidateProfile);
@@ -231,28 +233,14 @@ export async function generateTaskDagUseCase(input) {
231
233
  profileRouting.selectedByProfile = parsed.profile;
232
234
  }
233
235
  }
234
- if (isBackendTestTask) {
235
- profileRouting.selectedTemplate = "backend-test-dag";
236
- profileRouting.source = "taskKind";
237
- profileRouting.routingReasons = [
238
- 'taskKind "backend-test" selects the dedicated backend test DAG template',
239
- ];
240
- if (parsed.profile === "auto") {
241
- profileRouting.selectedByProfile =
242
- resolveAutoRoutingProfile(profileRouting.candidateProfile);
243
- }
244
- else if (parsed.profileExplicit) {
245
- profileRouting.selectedByProfile = parsed.profile;
246
- }
247
- }
248
- const initResult = profileRouting.selectedTemplate === "standard-dag"
236
+ const initResult = profileRouting.selectedTemplate === candidateResult.template
249
237
  ? candidateResult
250
238
  : await initHybridDagFromTask(repoRoot, parsed.taskId, {
251
239
  outputPath: parsed.outputPath,
252
240
  template: profileRouting.selectedTemplate,
253
241
  });
254
242
  const outputPath = initResult.outputPath;
255
- const validateSummary = profileRouting.selectedTemplate === "standard-dag"
243
+ const validateSummary = profileRouting.selectedTemplate === candidateResult.template
256
244
  ? candidateValidateSummary
257
245
  : await validateDagUseCase(buildValidateInput(repoRoot, outputPath, parsed));
258
246
  const governanceProfile = validateSummary.governanceProfile ?? {
@@ -287,6 +275,17 @@ export async function generateTaskDagUseCase(input) {
287
275
  };
288
276
  }
289
277
  await assertSafeForExecution(outputPath);
278
+ // Mirror worker run-task layout so Observe can find dag-events.jsonl under
279
+ // .harness/task-pool/observability/runs/<runId>/ (CLI path, not Task Pool state).
280
+ const runId = parsed.runId ?? `dag-${Date.now()}-${randomUUID().slice(0, 8)}`;
281
+ const eventsJsonlPath = path.join(parsed.cwd, ".harness", "task-pool", "observability", "runs", runId, "dag-events.jsonl");
282
+ try {
283
+ await mkdir(path.dirname(eventsJsonlPath), { recursive: true });
284
+ }
285
+ catch (error) {
286
+ const message = error instanceof Error ? error.message : String(error);
287
+ throw new Error(`failed to create dag events directory for observe: ${path.dirname(eventsJsonlPath)}: ${message}`);
288
+ }
290
289
  const runSummary = await runDagUseCase({
291
290
  repoRoot,
292
291
  dagPath: outputPath,
@@ -294,7 +293,8 @@ export async function generateTaskDagUseCase(input) {
294
293
  initOnly: parsed.initOnly,
295
294
  dryRun: parsed.dryRun,
296
295
  maxConcurrent: parsed.maxConcurrent,
297
- runId: parsed.runId,
296
+ runId,
297
+ eventsJsonlPath,
298
298
  canvasPath: parsed.canvasPath,
299
299
  canvasName: parsed.canvasName,
300
300
  canvasesDir: parsed.canvasesDir,
@@ -0,0 +1,75 @@
1
+ import { createHash } from "node:crypto";
2
+ /** Deterministic JSON stringify: object keys sorted at every level. */
3
+ export function stableStringify(value) {
4
+ return JSON.stringify(canonicalize(value));
5
+ }
6
+ function canonicalize(value) {
7
+ if (value === null || typeof value !== "object") {
8
+ return value;
9
+ }
10
+ if (Array.isArray(value)) {
11
+ return value.map((item) => canonicalize(item));
12
+ }
13
+ const obj = value;
14
+ const keys = Object.keys(obj).sort();
15
+ const out = {};
16
+ for (const key of keys) {
17
+ out[key] = canonicalize(obj[key]);
18
+ }
19
+ return out;
20
+ }
21
+ export function normalizeContentSha(value) {
22
+ const trimmed = value.trim();
23
+ const bare = trimmed.startsWith("sha256:")
24
+ ? trimmed.slice("sha256:".length)
25
+ : trimmed;
26
+ if (!/^[a-f0-9]{64}$/i.test(bare)) {
27
+ throw new Error(`invalid content sha256: ${value}`);
28
+ }
29
+ return bare.toLowerCase();
30
+ }
31
+ export function formatContentSha(hex) {
32
+ return `sha256:${normalizeContentSha(hex)}`;
33
+ }
34
+ export function normalizeContentRefs(refs) {
35
+ const normalized = refs.map((ref) => ({
36
+ path: ref.path.replace(/\\/g, "/"),
37
+ sha256: normalizeContentSha(ref.sha256),
38
+ }));
39
+ normalized.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
40
+ const seen = new Set();
41
+ for (const ref of normalized) {
42
+ if (seen.has(ref.path)) {
43
+ throw new Error(`duplicate content ref path: ${ref.path}`);
44
+ }
45
+ seen.add(ref.path);
46
+ }
47
+ return normalized;
48
+ }
49
+ /**
50
+ * Canonical payload for bundleHash.
51
+ * Excludes candidateId, createdAt, description, bundleHash, absolute paths,
52
+ * registry location, and any mutable lifecycle state.
53
+ */
54
+ export function buildCanonicalBundlePayload(manifest) {
55
+ return {
56
+ schemaVersion: 1,
57
+ candidateKind: manifest.candidateKind,
58
+ parentCandidateId: manifest.parentCandidateId ?? null,
59
+ contentRefs: normalizeContentRefs(manifest.contentRefs),
60
+ };
61
+ }
62
+ export function computeBundleHash(manifest) {
63
+ const payload = buildCanonicalBundlePayload(manifest);
64
+ const digest = createHash("sha256")
65
+ .update(stableStringify(payload))
66
+ .digest("hex");
67
+ return formatContentSha(digest);
68
+ }
69
+ export function sha256Hex(content) {
70
+ return createHash("sha256").update(content).digest("hex");
71
+ }
72
+ export function eventHashHex(payload) {
73
+ return createHash("sha256").update(stableStringify(payload)).digest("hex");
74
+ }
75
+ export const LIFECYCLE_GENESIS_HASH = "0".repeat(64);
@@ -0,0 +1,52 @@
1
+ import { loadManifestInputFromPath, listCandidateIds, materializeManifest, readCandidateRecord, registerCandidateManifest, transitionCandidateLifecycle, } from "../../infrastructure/evaluation/candidate-store.js";
2
+ export async function registerCandidate(input) {
3
+ const raw = await loadManifestInputFromPath(input.repoRoot, input.manifestPath);
4
+ const manifest = await materializeManifest(input.repoRoot, raw);
5
+ return registerCandidateManifest({
6
+ repoRoot: input.repoRoot,
7
+ manifest,
8
+ now: input.now,
9
+ });
10
+ }
11
+ export async function showCandidate(input) {
12
+ return readCandidateRecord(input.repoRoot, input.candidateId);
13
+ }
14
+ export async function listCandidates(input) {
15
+ const ids = await listCandidateIds(input.repoRoot);
16
+ const rows = [];
17
+ for (const candidateId of ids) {
18
+ const record = await readCandidateRecord(input.repoRoot, candidateId);
19
+ rows.push({
20
+ candidateId: record.manifest.candidateId,
21
+ bundleHash: record.manifest.bundleHash,
22
+ candidateKind: record.manifest.candidateKind,
23
+ status: record.status,
24
+ promotionApplied: false,
25
+ });
26
+ }
27
+ return rows;
28
+ }
29
+ export async function transitionCandidate(input) {
30
+ return transitionCandidateLifecycle(input);
31
+ }
32
+ export function formatCandidateMarkdown(record) {
33
+ const lines = [
34
+ `# Candidate: ${record.manifest.candidateId}`,
35
+ "",
36
+ `- status: \`${record.status}\``,
37
+ `- kind: \`${record.manifest.candidateKind}\``,
38
+ `- bundleHash: \`${record.manifest.bundleHash}\``,
39
+ `- parent: \`${record.manifest.parentCandidateId ?? "null"}\``,
40
+ `- promotionApplied: \`false\` (registry lifecycle only; no alias/incumbent)`,
41
+ "",
42
+ "## Content refs",
43
+ "",
44
+ ...record.manifest.contentRefs.map((ref) => `- \`${ref.path}\` — \`${ref.sha256}\``),
45
+ "",
46
+ "## Lifecycle",
47
+ "",
48
+ ...record.events.map((event) => `- #${event.seq} ${event.from ?? "∅"} → ${event.to}: ${event.reason} (${event.at})`),
49
+ "",
50
+ ];
51
+ return `${lines.join("\n")}\n`;
52
+ }
@@ -0,0 +1,289 @@
1
+ import { createHash } from "node:crypto";
2
+ import { readFile } from "node:fs/promises";
3
+ import path from "node:path";
4
+ import { reportDagUseCase } from "../dag/report-dag.js";
5
+ import { readReplaySpec, writeReplayArtifacts, } from "../../infrastructure/evaluation/store.js";
6
+ function sha256(content) {
7
+ return createHash("sha256").update(content).digest("hex");
8
+ }
9
+ function pairKey(input) {
10
+ return `${input.split}\u0000${input.taskRef}\u0000${input.seed}`;
11
+ }
12
+ async function verifyEvidenceHash(input) {
13
+ const content = await readFile(input.filePath);
14
+ const actual = sha256(content);
15
+ if (actual !== input.expected) {
16
+ throw new Error(`${input.label} hash mismatch: expected ${input.expected}, got ${actual}`);
17
+ }
18
+ }
19
+ function sumOptional(values) {
20
+ if (values.some((value) => value === undefined)) {
21
+ return { value: null, missing: true };
22
+ }
23
+ return {
24
+ value: values.reduce((sum, value) => sum + (value ?? 0), 0),
25
+ missing: false,
26
+ };
27
+ }
28
+ function verifyPassed(input) {
29
+ if (input.status !== "finished")
30
+ return false;
31
+ const verificationNodes = input.nodes.filter((node) => node.executor === "shell" ||
32
+ node.nodeId.includes("verify") ||
33
+ node.nodeId.includes("gate"));
34
+ return (verificationNodes.length > 0 &&
35
+ verificationNodes.every((node) => node.status === "FINISHED" &&
36
+ (!node.failureCategory || node.failureCategory === "success")));
37
+ }
38
+ function metricsForRun(run) {
39
+ const tokens = sumOptional(run.nodes.map((node) => node.tokensUsed));
40
+ const duration = sumOptional(run.nodes.map((node) => node.durationMs));
41
+ const missingFields = [];
42
+ if (tokens.missing)
43
+ missingFields.push("tokens");
44
+ if (duration.missing)
45
+ missingFields.push("durationMs");
46
+ return {
47
+ metrics: {
48
+ verifyPassed: verifyPassed(run),
49
+ tokens: tokens.value,
50
+ durationMs: duration.value,
51
+ executorCalls: run.nodes.length,
52
+ repairPasses: run.convergence?.currentPass ?? 0,
53
+ },
54
+ missingFields,
55
+ };
56
+ }
57
+ function compareRows(incumbent, challenger) {
58
+ const reasons = [];
59
+ let verdict;
60
+ if (incumbent.metrics.verifyPassed !== challenger.metrics.verifyPassed) {
61
+ verdict = challenger.metrics.verifyPassed
62
+ ? "challenger_win"
63
+ : "incumbent_win";
64
+ reasons.push("verification_outcome");
65
+ }
66
+ else if (!incumbent.metrics.verifyPassed) {
67
+ verdict = "tie";
68
+ reasons.push("both_failed_verification");
69
+ }
70
+ else if (incumbent.metrics.tokens === null ||
71
+ challenger.metrics.tokens === null ||
72
+ incumbent.metrics.durationMs === null ||
73
+ challenger.metrics.durationMs === null) {
74
+ verdict = "incomparable";
75
+ reasons.push("missing_cost_metrics");
76
+ }
77
+ else {
78
+ const challengerNoWorse = challenger.metrics.tokens <= incumbent.metrics.tokens &&
79
+ challenger.metrics.durationMs <= incumbent.metrics.durationMs;
80
+ const incumbentNoWorse = incumbent.metrics.tokens <= challenger.metrics.tokens &&
81
+ incumbent.metrics.durationMs <= challenger.metrics.durationMs;
82
+ if (challengerNoWorse && !incumbentNoWorse) {
83
+ verdict = "challenger_win";
84
+ reasons.push("lower_cost");
85
+ }
86
+ else if (incumbentNoWorse && !challengerNoWorse) {
87
+ verdict = "incumbent_win";
88
+ reasons.push("lower_cost");
89
+ }
90
+ else if (challengerNoWorse && incumbentNoWorse) {
91
+ verdict = "tie";
92
+ reasons.push("equal_metrics");
93
+ }
94
+ else {
95
+ verdict = "incomparable";
96
+ reasons.push("cost_tradeoff");
97
+ }
98
+ }
99
+ return {
100
+ taskRef: incumbent.taskRef,
101
+ seed: incumbent.seed,
102
+ split: incumbent.split,
103
+ incumbentRunId: incumbent.runId,
104
+ challengerRunId: challenger.runId,
105
+ verdict,
106
+ reasons,
107
+ };
108
+ }
109
+ function buildScorecard(input) {
110
+ const rows = [...input.rows].sort((left, right) => [
111
+ left.split,
112
+ left.taskRef,
113
+ String(left.seed).padStart(12, "0"),
114
+ left.candidateId,
115
+ left.runId,
116
+ ]
117
+ .join("\u0000")
118
+ .localeCompare([
119
+ right.split,
120
+ right.taskRef,
121
+ String(right.seed).padStart(12, "0"),
122
+ right.candidateId,
123
+ right.runId,
124
+ ].join("\u0000")));
125
+ const incumbentByKey = new Map(rows
126
+ .filter((row) => row.candidateId === input.incumbentCandidateId)
127
+ .map((row) => [pairKey(row), row]));
128
+ const challengerByKey = new Map(rows
129
+ .filter((row) => row.candidateId === input.challengerCandidateId)
130
+ .map((row) => [pairKey(row), row]));
131
+ const keys = [
132
+ ...new Set([...incumbentByKey.keys(), ...challengerByKey.keys()]),
133
+ ].sort();
134
+ const comparisons = [];
135
+ let unpairedEvidenceCount = 0;
136
+ for (const key of keys) {
137
+ const incumbent = incumbentByKey.get(key);
138
+ const challenger = challengerByKey.get(key);
139
+ if (!incumbent || !challenger) {
140
+ unpairedEvidenceCount +=
141
+ Number(Boolean(incumbent)) + Number(Boolean(challenger));
142
+ continue;
143
+ }
144
+ comparisons.push(compareRows(incumbent, challenger));
145
+ }
146
+ const reasons = ["replay_only"];
147
+ if (comparisons.length === 0)
148
+ reasons.push("insufficient_samples");
149
+ if (unpairedEvidenceCount > 0)
150
+ reasons.push("unpaired_evidence");
151
+ return {
152
+ schemaVersion: 1,
153
+ replayId: input.replayId,
154
+ incumbentCandidateId: input.incumbentCandidateId,
155
+ challengerCandidateId: input.challengerCandidateId,
156
+ rows,
157
+ comparisons,
158
+ aggregate: {
159
+ pairedSampleCount: comparisons.length,
160
+ incumbentWins: comparisons.filter((item) => item.verdict === "incumbent_win").length,
161
+ challengerWins: comparisons.filter((item) => item.verdict === "challenger_win").length,
162
+ ties: comparisons.filter((item) => item.verdict === "tie").length,
163
+ incomparable: comparisons.filter((item) => item.verdict === "incomparable").length,
164
+ unpairedEvidenceCount,
165
+ promotionEligible: false,
166
+ reasons,
167
+ },
168
+ };
169
+ }
170
+ export function formatReplayMarkdown(scorecard) {
171
+ const lines = [
172
+ `# Eval Replay Scorecard: ${scorecard.replayId}`,
173
+ "",
174
+ "> Replay-only evidence. This report never authorizes promotion or executes Pi/DAG work.",
175
+ "",
176
+ `- incumbent: ${scorecard.incumbentCandidateId}`,
177
+ `- challenger: ${scorecard.challengerCandidateId}`,
178
+ `- promotionEligible: ${scorecard.aggregate.promotionEligible}`,
179
+ `- reasons: ${scorecard.aggregate.reasons.join(", ")}`,
180
+ "",
181
+ "## Evidence",
182
+ "",
183
+ "| split | task | seed | candidate | run | verified | tokens | durationMs | calls | repairPasses | missing |",
184
+ "|---|---|---:|---|---|---|---:|---:|---:|---:|---|",
185
+ ];
186
+ for (const row of scorecard.rows) {
187
+ lines.push(`| ${row.split} | ${row.taskRef} | ${row.seed} | ${row.candidateId} | ${row.runId} | ${row.metrics.verifyPassed} | ${row.metrics.tokens ?? "n/a"} | ${row.metrics.durationMs ?? "n/a"} | ${row.metrics.executorCalls} | ${row.metrics.repairPasses} | ${row.missingFields.join(", ") || "none"} |`);
188
+ }
189
+ lines.push("", "## Paired Comparisons", "", "| split | task | seed | incumbent run | challenger run | verdict | reasons |", "|---|---|---:|---|---|---|---|");
190
+ for (const comparison of scorecard.comparisons) {
191
+ lines.push(`| ${comparison.split} | ${comparison.taskRef} | ${comparison.seed} | ${comparison.incumbentRunId} | ${comparison.challengerRunId} | ${comparison.verdict} | ${comparison.reasons.join(", ")} |`);
192
+ }
193
+ if (scorecard.comparisons.length === 0) {
194
+ lines.push("| - | - | - | - | - | - | no paired evidence |");
195
+ }
196
+ return `${lines.join("\n")}\n`;
197
+ }
198
+ export async function replayEvaluation(input) {
199
+ const spec = await readReplaySpec(input.repoRoot, input.specPath);
200
+ const rows = [];
201
+ for (const evidence of spec.evidence) {
202
+ const runDir = path.join(input.repoRoot, ".harness", "dag-runs", "completed", evidence.runId);
203
+ const statePath = path.join(runDir, "state.json");
204
+ const runPath = path.join(runDir, "run.json");
205
+ const stateRefPath = path
206
+ .relative(input.repoRoot, statePath)
207
+ .split(path.sep)
208
+ .join("/");
209
+ const runRefPath = path
210
+ .relative(input.repoRoot, runPath)
211
+ .split(path.sep)
212
+ .join("/");
213
+ await verifyEvidenceHash({
214
+ filePath: statePath,
215
+ expected: evidence.stateSha256,
216
+ label: "state.json",
217
+ });
218
+ await verifyEvidenceHash({
219
+ filePath: runPath,
220
+ expected: evidence.runSha256,
221
+ label: "run.json",
222
+ });
223
+ const report = await reportDagUseCase({
224
+ repoRoot: input.repoRoot,
225
+ runId: evidence.runId,
226
+ lifecycle: "completed",
227
+ failedOnly: false,
228
+ latest: false,
229
+ });
230
+ const run = report.runs[0];
231
+ if (!run) {
232
+ throw new Error(`completed DAG run not found: ${evidence.runId}`);
233
+ }
234
+ if (run.evaluationAssociation.status === "present") {
235
+ if (run.evaluationAssociation.candidateId !== evidence.candidateId) {
236
+ throw new Error(`replay evidence candidateId conflict for ${evidence.runId}: evidence=${evidence.candidateId}, run=${run.evaluationAssociation.candidateId}`);
237
+ }
238
+ if (run.evaluationAssociation.seed !== evidence.seed) {
239
+ throw new Error(`replay evidence seed conflict for ${evidence.runId}: evidence=${evidence.seed}, run=${run.evaluationAssociation.seed}`);
240
+ }
241
+ if (run.evaluationAssociation.split &&
242
+ run.evaluationAssociation.split !== evidence.split) {
243
+ throw new Error(`replay evidence split conflict for ${evidence.runId}: evidence=${evidence.split}, run=${run.evaluationAssociation.split}`);
244
+ }
245
+ if (run.evaluationAssociation.taskRef &&
246
+ run.evaluationAssociation.taskRef !== evidence.taskRef) {
247
+ throw new Error(`replay evidence taskRef conflict for ${evidence.runId}: evidence=${evidence.taskRef}, run=${run.evaluationAssociation.taskRef}`);
248
+ }
249
+ }
250
+ if (![
251
+ "finished",
252
+ "failed",
253
+ "partial_failed",
254
+ "superseded",
255
+ "abandoned",
256
+ ].includes(run.status)) {
257
+ throw new Error(`completed DAG run ${evidence.runId} is not terminal (status=${run.status})`);
258
+ }
259
+ const { metrics, missingFields } = metricsForRun(run);
260
+ rows.push({
261
+ ...evidence,
262
+ lifecycle: "completed",
263
+ runStatus: run.status,
264
+ metrics,
265
+ missingFields,
266
+ evidenceRefs: [
267
+ { path: stateRefPath, sha256: evidence.stateSha256 },
268
+ { path: runRefPath, sha256: evidence.runSha256 },
269
+ ],
270
+ });
271
+ }
272
+ const scorecard = buildScorecard({
273
+ replayId: spec.replayId,
274
+ incumbentCandidateId: spec.incumbentCandidateId,
275
+ challengerCandidateId: spec.challengerCandidateId,
276
+ rows,
277
+ });
278
+ const markdown = formatReplayMarkdown(scorecard);
279
+ if (input.writeArtifacts === false) {
280
+ return { scorecard, markdown };
281
+ }
282
+ const paths = await writeReplayArtifacts({
283
+ repoRoot: input.repoRoot,
284
+ replayId: spec.replayId,
285
+ scorecard,
286
+ markdown,
287
+ });
288
+ return { scorecard, markdown, ...paths };
289
+ }