@tea-agent/loop-agent 0.13.0-alpha.0 → 0.13.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (262) hide show
  1. package/AGENTS.md +155 -153
  2. package/CHANGELOG.md +326 -301
  3. package/README.md +345 -326
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/generate-task-dag.js +28 -58
  7. package/dist/application/evaluation/candidate-hash.js +75 -0
  8. package/dist/application/evaluation/candidate.js +52 -0
  9. package/dist/application/evaluation/replay.js +289 -0
  10. package/dist/application/evaluation/types.js +130 -0
  11. package/dist/cli/command-definitions.js +17 -4
  12. package/dist/cli/program.js +8 -4
  13. package/dist/commands/cursor-prompt.js +6 -6
  14. package/dist/commands/eval.js +235 -0
  15. package/dist/commands/init.js +544 -506
  16. package/dist/commands/loop-benchmark.js +11 -11
  17. package/dist/commands/pi-reuse-benchmark.js +16 -16
  18. package/dist/executors/pi-sdk-executor.js +38 -24
  19. package/dist/executors/shell-executor.js +34 -2
  20. package/dist/executors/shell-presets.js +20 -0
  21. package/dist/executors/shell-verification.js +7 -0
  22. package/dist/governance/manifest-types.js +1 -0
  23. package/dist/infrastructure/evaluation/candidate-store.js +435 -0
  24. package/dist/infrastructure/evaluation/store.js +40 -0
  25. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  26. package/dist/task/config-types.js +23 -0
  27. package/dist/task/runtime.js +27 -27
  28. package/dist/worker/observe/routes.js +18 -3
  29. package/dist/worker/observe/spec-evidence.js +1 -1
  30. package/dist/worker/observe/static/api.js +46 -46
  31. package/dist/worker/observe/static/app.js +150 -150
  32. package/dist/worker/observe/static/constants.js +148 -148
  33. package/dist/worker/observe/static/copy.js +67 -67
  34. package/dist/worker/observe/static/dag-helpers.js +172 -172
  35. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  36. package/dist/worker/observe/static/dag-layout.js +83 -83
  37. package/dist/worker/observe/static/dag-model.js +72 -72
  38. package/dist/worker/observe/static/dom.js +61 -61
  39. package/dist/worker/observe/static/format-pool.js +67 -67
  40. package/dist/worker/observe/static/format.js +292 -292
  41. package/dist/worker/observe/static/index.html +308 -308
  42. package/dist/worker/observe/static/kpi.js +94 -94
  43. package/dist/worker/observe/static/relations.js +133 -133
  44. package/dist/worker/observe/static/router.js +93 -93
  45. package/dist/worker/observe/static/run-processing.js +148 -148
  46. package/dist/worker/observe/static/shell-chrome.js +68 -68
  47. package/dist/worker/observe/static/state.js +253 -253
  48. package/dist/worker/observe/static/styles.css +1902 -1902
  49. package/dist/worker/observe/static/views/batch.js +227 -227
  50. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  51. package/dist/worker/observe/static/views/dag-inspector.js +607 -596
  52. package/dist/worker/observe/static/views/dag.js +362 -362
  53. package/dist/worker/observe/static/views/dashboard.js +445 -445
  54. package/dist/worker/observe/static/views/failures.js +143 -143
  55. package/dist/worker/observe/static/views/feature.js +492 -492
  56. package/dist/worker/observe/static/views/pool.js +350 -350
  57. package/dist/worker/observe/static/views/run.js +453 -453
  58. package/dist/worker/observe/static/views/session-timeline.js +205 -205
  59. package/dist/worker/observe/static/views/shell.js +7 -7
  60. package/dist/worker/observe/static/views/task.js +314 -314
  61. package/dist/worker/observe/static/views/timeline.js +163 -163
  62. package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
  63. package/dist/workflows/dag/canvas-observer.js +275 -275
  64. package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
  65. package/dist/workflows/dag/init-hybrid.js +1415 -200
  66. package/dist/workflows/dag/node-execution.js +9 -0
  67. package/dist/workflows/dag/prompt.js +9 -0
  68. package/dist/workflows/dag/report.js +35 -1
  69. package/dist/workflows/dag/runner.js +28 -2
  70. package/dist/workflows/dag/task-demand-routing.js +383 -0
  71. package/dist/workflows/dag/types.js +50 -13
  72. package/dist/workflows/dag/upstream-artifacts.js +1 -0
  73. package/dist/workflows/dag/validate.js +59 -1
  74. package/docs/README.md +106 -104
  75. package/docs/agent-dag-recovery-playbook.md +195 -193
  76. package/docs/agent-dag-runner.md +67 -67
  77. package/docs/architecture/README.md +26 -26
  78. package/docs/architecture/dag-execution.md +140 -140
  79. package/docs/architecture/evolution.md +54 -54
  80. package/docs/architecture/facts-and-state.md +71 -71
  81. package/docs/architecture/runtime-boundaries.md +191 -191
  82. package/docs/architecture/system-overview.md +93 -93
  83. package/docs/architecture/worker-and-feature.md +85 -85
  84. package/docs/cursor-prompt-sidecar.md +36 -36
  85. package/docs/decisions/README.md +18 -18
  86. package/docs/design/README.md +167 -85
  87. package/docs/development-principles.md +73 -73
  88. package/docs/exec-plans/README.md +6 -6
  89. package/docs/exec-plans/active/README.md +15 -11
  90. package/docs/exec-plans/completed/README.md +85 -74
  91. package/docs/feature-workflow.md +389 -339
  92. package/docs/harness-methodology-debugging.md +153 -153
  93. package/docs/harness-methodology-tdd.md +130 -130
  94. package/docs/harness-methodology-verification.md +27 -27
  95. package/docs/init-surface.manifest.json +289 -280
  96. package/docs/loop-agent-harness.md +142 -141
  97. package/docs/production-readiness.md +96 -96
  98. package/docs/progress/README.md +64 -58
  99. package/docs/reports/README.md +117 -100
  100. package/docs/skills/README.md +7 -7
  101. package/docs/skills/vetted-skill-registry.md +29 -27
  102. package/docs/templates/adr.md +60 -60
  103. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  104. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  105. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  106. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  107. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  108. package/docs/templates/agent-dag-report.schema.json +473 -473
  109. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  110. package/docs/templates/agent-dag.base.json +190 -190
  111. package/docs/templates/agent-dag.final-verification.json +185 -185
  112. package/docs/templates/agent-dag.schema.json +411 -383
  113. package/docs/templates/agent-dag.supervised-implementation.json +501 -501
  114. package/docs/templates/backend-test-analysis.schema.json +44 -0
  115. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
  116. package/docs/templates/backend-test-dag.json +311 -288
  117. package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
  118. package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
  119. package/docs/templates/exec-plan.md +64 -64
  120. package/docs/templates/feature-spec.md +53 -53
  121. package/docs/templates/frontend-design-contract.md +42 -33
  122. package/docs/templates/frontend-task-constraints.md +35 -25
  123. package/docs/templates/frontend-task-requirement.md +70 -61
  124. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
  125. package/docs/templates/frontend-test-dag.json +23 -0
  126. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
  127. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
  128. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
  129. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
  130. package/docs/templates/harness.schema.json +221 -221
  131. package/docs/templates/hybrid-dag.json +188 -188
  132. package/docs/templates/init-evolution-review.md +35 -35
  133. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  134. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  135. package/docs/templates/knowledge-sync-dag.json +178 -177
  136. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  137. package/docs/templates/product-line/AGENTS.md +8 -8
  138. package/docs/templates/product-line/README.md +9 -9
  139. package/docs/templates/product-line/acceptance.yaml +14 -14
  140. package/docs/templates/product-line/closeout.yaml +9 -9
  141. package/docs/templates/product-line/design.md +13 -13
  142. package/docs/templates/product-line/links.md +10 -10
  143. package/docs/templates/product-line/requirement.md +17 -17
  144. package/docs/templates/product-line/task-graph.yaml +15 -15
  145. package/docs/templates/product-line/task.yaml +64 -64
  146. package/docs/templates/product-line/test-plan.md +7 -7
  147. package/docs/templates/production-readiness-checklist.md +57 -57
  148. package/docs/templates/progress-log.md +17 -17
  149. package/docs/templates/project-start-checklist.md +9 -9
  150. package/docs/templates/qa-report.md +48 -48
  151. package/docs/templates/sprint-contract.md +29 -29
  152. package/docs/templates/worker-dogfood-evidence.md +80 -80
  153. package/docs/templates/worker-dogfood-setup.md +68 -68
  154. package/docs/verification-matrix.md +70 -67
  155. package/examples/decision-gate-agent-dag.json +177 -177
  156. package/examples/example-dag.json +46 -46
  157. package/examples/hybrid-loop-agent-dag.json +189 -189
  158. package/harness.json +66 -66
  159. package/package.json +88 -52
  160. package/scripts/check-product-line-docs.sh +29 -29
  161. package/scripts/check-task-pool-root.sh +32 -32
  162. package/scripts/kb-bootstrap-init-skeleton.sh +240 -239
  163. package/scripts/kb-graph-incremental-prepare.mjs +386 -372
  164. package/scripts/kb-graph-incremental-prepare.sh +5 -5
  165. package/scripts/kb-graph-materialize.mjs +105 -105
  166. package/scripts/kb-graph-materialize.sh +4 -4
  167. package/scripts/kb-graph-promote.mjs +164 -153
  168. package/scripts/kb-graph-promote.sh +4 -4
  169. package/scripts/kb-query.mjs +554 -554
  170. package/scripts/kb-query.sh +5 -5
  171. package/skills/agent-worker/SKILL.md +39 -39
  172. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  173. package/skills/ai-engineering-context/SKILL.md +48 -48
  174. package/skills/analyze-product-dependencies/SKILL.md +67 -0
  175. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
  176. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
  177. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
  178. package/skills/analyze-product-dependencies/references/example.md +76 -0
  179. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
  180. package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
  181. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
  182. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
  183. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
  184. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
  185. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
  186. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
  187. package/skills/analyze-product-requirements/SKILL.md +90 -0
  188. package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
  189. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
  190. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
  191. package/skills/analyze-product-requirements/references/example.md +86 -0
  192. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
  193. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
  194. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
  195. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
  196. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
  197. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
  198. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
  199. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
  200. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
  201. package/skills/code-review-core/SKILL.md +20 -20
  202. package/skills/codebase-scout/SKILL.md +19 -19
  203. package/skills/frontend-design-review/SKILL.md +66 -61
  204. package/skills/frontend-design-review/references/review-checklist.md +58 -37
  205. package/skills/frontend-implementation/SKILL.md +45 -52
  206. package/skills/frontend-implementation/references/code-standards.md +32 -34
  207. package/skills/frontend-implementation/references/design-spec.md +46 -46
  208. package/skills/frontend-implementation/references/node-contracts.md +76 -63
  209. package/skills/frontend-review/SKILL.md +59 -53
  210. package/skills/frontend-review/references/review-findings.md +47 -42
  211. package/skills/frontend-verification/SKILL.md +53 -40
  212. package/skills/frontend-verification/references/verification-checklist.md +68 -56
  213. package/skills/grill-me/SKILL.md +10 -10
  214. package/skills/grill-with-docs/SKILL.md +88 -88
  215. package/skills/grill-with-docs/adr-format.md +47 -47
  216. package/skills/grill-with-docs/context-format.md +60 -60
  217. package/skills/init-capability-evolution/SKILL.md +70 -70
  218. package/skills/loop-agent/SKILL.md +151 -151
  219. package/skills/loop-agent/references/README.md +67 -67
  220. package/skills/loop-agent/references/command-reference.md +505 -453
  221. package/skills/loop-agent/references/docs-converge.md +126 -126
  222. package/skills/loop-agent/references/harness-policy.md +263 -263
  223. package/skills/loop-agent/references/hybrid-dag.md +238 -233
  224. package/skills/loop-agent/references/learned/README.md +21 -21
  225. package/skills/loop-agent/references/long-running-loop.md +57 -57
  226. package/skills/loop-agent/references/model-routing.md +36 -36
  227. package/skills/loop-agent/references/multi-worktree.md +54 -54
  228. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  229. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  230. package/skills/loop-agent/references/pi-prompt.md +23 -23
  231. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  232. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  233. package/skills/loop-agent/references/task-workflow.md +89 -89
  234. package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
  235. package/skills/playwright-cli/SKILL.md +420 -0
  236. package/skills/playwright-cli/references/element-attributes.md +23 -0
  237. package/skills/playwright-cli/references/playwright-tests.md +39 -0
  238. package/skills/playwright-cli/references/request-mocking.md +87 -0
  239. package/skills/playwright-cli/references/running-code.md +241 -0
  240. package/skills/playwright-cli/references/session-management.md +225 -0
  241. package/skills/playwright-cli/references/storage-state.md +275 -0
  242. package/skills/playwright-cli/references/test-generation.md +433 -0
  243. package/skills/playwright-cli/references/tracing.md +139 -0
  244. package/skills/playwright-cli/references/video-recording.md +143 -0
  245. package/skills/playwright-cli-case-generator/SKILL.md +74 -0
  246. package/skills/requesting-code-review/SKILL.md +101 -101
  247. package/skills/requesting-code-review/code-reviewer.md +168 -168
  248. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  249. package/skills/systematic-debugging/SKILL.md +296 -296
  250. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  251. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  252. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  253. package/skills/systematic-debugging/find-polluter.sh +63 -63
  254. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  255. package/skills/systematic-debugging/test-academic.md +14 -14
  256. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  257. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  258. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  259. package/skills/test-driven-development/SKILL.md +20 -20
  260. package/skills/using-git-worktrees/SKILL.md +215 -215
  261. package/skills/verification-before-completion/SKILL.md +154 -154
  262. package/skills/webapp-testing/SKILL.md +19 -19
@@ -1,22 +1,22 @@
1
- #!/usr/bin/env node
2
- import { existsSync } from "node:fs";
3
- import { dirname, join } from "node:path";
4
- import { fileURLToPath, pathToFileURL } from "node:url";
5
-
6
- const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
- const cliEntry = join(packageRoot, "dist", "worker", "cli.js");
8
-
9
- if (!existsSync(cliEntry)) {
10
- console.error(
11
- `agent-worker: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
- );
13
- process.exit(1);
14
- }
15
-
16
- try {
17
- const cli = await import(pathToFileURL(cliEntry).href);
18
- await cli.main(process.argv);
19
- } catch (error) {
20
- console.error(error instanceof Error ? error.message : String(error));
21
- process.exit(1);
22
- }
1
+ #!/usr/bin/env node
2
+ import { existsSync } from "node:fs";
3
+ import { dirname, join } from "node:path";
4
+ import { fileURLToPath, pathToFileURL } from "node:url";
5
+
6
+ const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
+ const cliEntry = join(packageRoot, "dist", "worker", "cli.js");
8
+
9
+ if (!existsSync(cliEntry)) {
10
+ console.error(
11
+ `agent-worker: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
+ );
13
+ process.exit(1);
14
+ }
15
+
16
+ try {
17
+ const cli = await import(pathToFileURL(cliEntry).href);
18
+ await cli.main(process.argv);
19
+ } catch (error) {
20
+ console.error(error instanceof Error ? error.message : String(error));
21
+ process.exit(1);
22
+ }
package/bin/loop-agent.js CHANGED
@@ -1,21 +1,21 @@
1
- #!/usr/bin/env node
2
- import { existsSync } from "node:fs";
3
- import { dirname, join } from "node:path";
4
- import { fileURLToPath, pathToFileURL } from "node:url";
5
-
6
- const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
- const cliEntry = join(packageRoot, "dist", "cli.js");
8
-
9
- if (!existsSync(cliEntry)) {
10
- console.error(
11
- `loop-agent: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
- );
13
- process.exit(1);
14
- }
15
-
16
- try {
17
- await import(pathToFileURL(cliEntry).href);
18
- } catch (error) {
19
- console.error(error instanceof Error ? error.message : String(error));
20
- process.exit(1);
21
- }
1
+ #!/usr/bin/env node
2
+ import { existsSync } from "node:fs";
3
+ import { dirname, join } from "node:path";
4
+ import { fileURLToPath, pathToFileURL } from "node:url";
5
+
6
+ const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
7
+ const cliEntry = join(packageRoot, "dist", "cli.js");
8
+
9
+ if (!existsSync(cliEntry)) {
10
+ console.error(
11
+ `loop-agent: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
12
+ );
13
+ process.exit(1);
14
+ }
15
+
16
+ try {
17
+ await import(pathToFileURL(cliEntry).href);
18
+ } catch (error) {
19
+ console.error(error instanceof Error ? error.message : String(error));
20
+ process.exit(1);
21
+ }
@@ -1,4 +1,6 @@
1
- import { readFile } from "node:fs/promises";
1
+ import { readFile, mkdir } from "node:fs/promises";
2
+ import { randomUUID } from "node:crypto";
3
+ import path from "node:path";
2
4
  import { resolveAutoRoutingProfile, requiresSupervisedQualityGate, } from "../../workflows/dag/governance-profile.js";
3
5
  import { resolveShellCommands } from "../../executors/shell-executor.js";
4
6
  import { parseDagSpec } from "../../workflows/dag/types.js";
@@ -6,7 +8,6 @@ import { pathMatchesPattern } from "../../shared/git-progress.js";
6
8
  import { loadHarnessManifest } from "../../governance/harness.js";
7
9
  import { assertExecPlanIndexConsistent } from "../../governance/exec-plans.js";
8
10
  import { defaultHybridDagOutputPath, initHybridDagFromTask, } from "../../workflows/dag/init-hybrid.js";
9
- import { loadTaskConfig } from "../../task/runtime.js";
10
11
  import { validateDagUseCase } from "./validate-dag.js";
11
12
  import { runDagUseCase } from "./run-dag.js";
12
13
  const PLACEHOLDER_WRITESET_MARKER = "REPLACE/WITH";
@@ -202,11 +203,6 @@ export async function generateTaskDagUseCase(input) {
202
203
  // before any expensive DAG generation or execution. Empty/consistent
203
204
  // repos stay compatible so the default DAG flow is unblocked.
204
205
  await assertExecPlanIndexConsistent(repoRoot);
205
- const taskConfig = await loadTaskConfig(repoRoot, parsed.taskId);
206
- const isFrontendImplementationTask = taskConfig.taskKind === "frontend-implementation";
207
- const isBackendTestTask = taskConfig.taskKind === "backend-test";
208
- const isKnowledgeSyncTask = taskConfig.taskKind === "knowledge-sync";
209
- const isKnowledgeGraphBootstrapTask = taskConfig.taskKind === "knowledge-graph-bootstrap";
210
206
  const candidateResult = await initHybridDagFromTask(repoRoot, parsed.taskId, {
211
207
  outputPath: parsed.outputPath,
212
208
  template: "standard-dag",
@@ -219,12 +215,16 @@ export async function generateTaskDagUseCase(input) {
219
215
  codeChange: [],
220
216
  reasons: ["dag run-task validate did not report governanceProfile"],
221
217
  });
222
- if (isFrontendImplementationTask) {
223
- profileRouting.selectedTemplate = "frontend-implementation";
224
- profileRouting.source = "taskKind";
225
- profileRouting.routingReasons = [
226
- 'taskKind "frontend-implementation" selects the dedicated frontend DAG template',
227
- ];
218
+ const hasExplicitSpecializedTaskKind = candidateResult.templateSelection.source === "taskKind";
219
+ const hasSafeAutomaticTaskSourceRoute = candidateResult.templateSelection.source === "task-source" &&
220
+ profileRouting.selectedByProfile !== "supervised" &&
221
+ profileRouting.selectedTemplate !== "supervised-implementation";
222
+ if (hasExplicitSpecializedTaskKind || hasSafeAutomaticTaskSourceRoute) {
223
+ profileRouting.selectedTemplate = candidateResult.template;
224
+ profileRouting.source = hasExplicitSpecializedTaskKind
225
+ ? "taskKind"
226
+ : "task-source";
227
+ profileRouting.routingReasons = candidateResult.templateSelection.reasons;
228
228
  if (parsed.profile === "auto") {
229
229
  profileRouting.selectedByProfile =
230
230
  resolveAutoRoutingProfile(profileRouting.candidateProfile);
@@ -233,56 +233,14 @@ export async function generateTaskDagUseCase(input) {
233
233
  profileRouting.selectedByProfile = parsed.profile;
234
234
  }
235
235
  }
236
- if (isBackendTestTask) {
237
- profileRouting.selectedTemplate = "backend-test-dag";
238
- profileRouting.source = "taskKind";
239
- profileRouting.routingReasons = [
240
- 'taskKind "backend-test" selects the dedicated backend test DAG template',
241
- ];
242
- if (parsed.profile === "auto") {
243
- profileRouting.selectedByProfile =
244
- resolveAutoRoutingProfile(profileRouting.candidateProfile);
245
- }
246
- else if (parsed.profileExplicit) {
247
- profileRouting.selectedByProfile = parsed.profile;
248
- }
249
- }
250
- if (isKnowledgeSyncTask) {
251
- profileRouting.selectedTemplate = "knowledge-sync-dag";
252
- profileRouting.source = "taskKind";
253
- profileRouting.routingReasons = [
254
- 'taskKind "knowledge-sync" selects the dedicated knowledge-sync DAG template',
255
- ];
256
- if (parsed.profile === "auto") {
257
- profileRouting.selectedByProfile =
258
- resolveAutoRoutingProfile(profileRouting.candidateProfile);
259
- }
260
- else if (parsed.profileExplicit) {
261
- profileRouting.selectedByProfile = parsed.profile;
262
- }
263
- }
264
- if (isKnowledgeGraphBootstrapTask) {
265
- profileRouting.selectedTemplate = "knowledge-graph-bootstrap-dag";
266
- profileRouting.source = "taskKind";
267
- profileRouting.routingReasons = [
268
- 'taskKind "knowledge-graph-bootstrap" selects the dedicated knowledge-graph bootstrap DAG template',
269
- ];
270
- if (parsed.profile === "auto") {
271
- profileRouting.selectedByProfile =
272
- resolveAutoRoutingProfile(profileRouting.candidateProfile);
273
- }
274
- else if (parsed.profileExplicit) {
275
- profileRouting.selectedByProfile = parsed.profile;
276
- }
277
- }
278
- const initResult = profileRouting.selectedTemplate === "standard-dag"
236
+ const initResult = profileRouting.selectedTemplate === candidateResult.template
279
237
  ? candidateResult
280
238
  : await initHybridDagFromTask(repoRoot, parsed.taskId, {
281
239
  outputPath: parsed.outputPath,
282
240
  template: profileRouting.selectedTemplate,
283
241
  });
284
242
  const outputPath = initResult.outputPath;
285
- const validateSummary = profileRouting.selectedTemplate === "standard-dag"
243
+ const validateSummary = profileRouting.selectedTemplate === candidateResult.template
286
244
  ? candidateValidateSummary
287
245
  : await validateDagUseCase(buildValidateInput(repoRoot, outputPath, parsed));
288
246
  const governanceProfile = validateSummary.governanceProfile ?? {
@@ -317,6 +275,17 @@ export async function generateTaskDagUseCase(input) {
317
275
  };
318
276
  }
319
277
  await assertSafeForExecution(outputPath);
278
+ // Mirror worker run-task layout so Observe can find dag-events.jsonl under
279
+ // .harness/task-pool/observability/runs/<runId>/ (CLI path, not Task Pool state).
280
+ const runId = parsed.runId ?? `dag-${Date.now()}-${randomUUID().slice(0, 8)}`;
281
+ const eventsJsonlPath = path.join(parsed.cwd, ".harness", "task-pool", "observability", "runs", runId, "dag-events.jsonl");
282
+ try {
283
+ await mkdir(path.dirname(eventsJsonlPath), { recursive: true });
284
+ }
285
+ catch (error) {
286
+ const message = error instanceof Error ? error.message : String(error);
287
+ throw new Error(`failed to create dag events directory for observe: ${path.dirname(eventsJsonlPath)}: ${message}`);
288
+ }
320
289
  const runSummary = await runDagUseCase({
321
290
  repoRoot,
322
291
  dagPath: outputPath,
@@ -324,7 +293,8 @@ export async function generateTaskDagUseCase(input) {
324
293
  initOnly: parsed.initOnly,
325
294
  dryRun: parsed.dryRun,
326
295
  maxConcurrent: parsed.maxConcurrent,
327
- runId: parsed.runId,
296
+ runId,
297
+ eventsJsonlPath,
328
298
  canvasPath: parsed.canvasPath,
329
299
  canvasName: parsed.canvasName,
330
300
  canvasesDir: parsed.canvasesDir,
@@ -0,0 +1,75 @@
1
+ import { createHash } from "node:crypto";
2
+ /** Deterministic JSON stringify: object keys sorted at every level. */
3
+ export function stableStringify(value) {
4
+ return JSON.stringify(canonicalize(value));
5
+ }
6
+ function canonicalize(value) {
7
+ if (value === null || typeof value !== "object") {
8
+ return value;
9
+ }
10
+ if (Array.isArray(value)) {
11
+ return value.map((item) => canonicalize(item));
12
+ }
13
+ const obj = value;
14
+ const keys = Object.keys(obj).sort();
15
+ const out = {};
16
+ for (const key of keys) {
17
+ out[key] = canonicalize(obj[key]);
18
+ }
19
+ return out;
20
+ }
21
+ export function normalizeContentSha(value) {
22
+ const trimmed = value.trim();
23
+ const bare = trimmed.startsWith("sha256:")
24
+ ? trimmed.slice("sha256:".length)
25
+ : trimmed;
26
+ if (!/^[a-f0-9]{64}$/i.test(bare)) {
27
+ throw new Error(`invalid content sha256: ${value}`);
28
+ }
29
+ return bare.toLowerCase();
30
+ }
31
+ export function formatContentSha(hex) {
32
+ return `sha256:${normalizeContentSha(hex)}`;
33
+ }
34
+ export function normalizeContentRefs(refs) {
35
+ const normalized = refs.map((ref) => ({
36
+ path: ref.path.replace(/\\/g, "/"),
37
+ sha256: normalizeContentSha(ref.sha256),
38
+ }));
39
+ normalized.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
40
+ const seen = new Set();
41
+ for (const ref of normalized) {
42
+ if (seen.has(ref.path)) {
43
+ throw new Error(`duplicate content ref path: ${ref.path}`);
44
+ }
45
+ seen.add(ref.path);
46
+ }
47
+ return normalized;
48
+ }
49
+ /**
50
+ * Canonical payload for bundleHash.
51
+ * Excludes candidateId, createdAt, description, bundleHash, absolute paths,
52
+ * registry location, and any mutable lifecycle state.
53
+ */
54
+ export function buildCanonicalBundlePayload(manifest) {
55
+ return {
56
+ schemaVersion: 1,
57
+ candidateKind: manifest.candidateKind,
58
+ parentCandidateId: manifest.parentCandidateId ?? null,
59
+ contentRefs: normalizeContentRefs(manifest.contentRefs),
60
+ };
61
+ }
62
+ export function computeBundleHash(manifest) {
63
+ const payload = buildCanonicalBundlePayload(manifest);
64
+ const digest = createHash("sha256")
65
+ .update(stableStringify(payload))
66
+ .digest("hex");
67
+ return formatContentSha(digest);
68
+ }
69
+ export function sha256Hex(content) {
70
+ return createHash("sha256").update(content).digest("hex");
71
+ }
72
+ export function eventHashHex(payload) {
73
+ return createHash("sha256").update(stableStringify(payload)).digest("hex");
74
+ }
75
+ export const LIFECYCLE_GENESIS_HASH = "0".repeat(64);
@@ -0,0 +1,52 @@
1
+ import { loadManifestInputFromPath, listCandidateIds, materializeManifest, readCandidateRecord, registerCandidateManifest, transitionCandidateLifecycle, } from "../../infrastructure/evaluation/candidate-store.js";
2
+ export async function registerCandidate(input) {
3
+ const raw = await loadManifestInputFromPath(input.repoRoot, input.manifestPath);
4
+ const manifest = await materializeManifest(input.repoRoot, raw);
5
+ return registerCandidateManifest({
6
+ repoRoot: input.repoRoot,
7
+ manifest,
8
+ now: input.now,
9
+ });
10
+ }
11
+ export async function showCandidate(input) {
12
+ return readCandidateRecord(input.repoRoot, input.candidateId);
13
+ }
14
+ export async function listCandidates(input) {
15
+ const ids = await listCandidateIds(input.repoRoot);
16
+ const rows = [];
17
+ for (const candidateId of ids) {
18
+ const record = await readCandidateRecord(input.repoRoot, candidateId);
19
+ rows.push({
20
+ candidateId: record.manifest.candidateId,
21
+ bundleHash: record.manifest.bundleHash,
22
+ candidateKind: record.manifest.candidateKind,
23
+ status: record.status,
24
+ promotionApplied: false,
25
+ });
26
+ }
27
+ return rows;
28
+ }
29
+ export async function transitionCandidate(input) {
30
+ return transitionCandidateLifecycle(input);
31
+ }
32
+ export function formatCandidateMarkdown(record) {
33
+ const lines = [
34
+ `# Candidate: ${record.manifest.candidateId}`,
35
+ "",
36
+ `- status: \`${record.status}\``,
37
+ `- kind: \`${record.manifest.candidateKind}\``,
38
+ `- bundleHash: \`${record.manifest.bundleHash}\``,
39
+ `- parent: \`${record.manifest.parentCandidateId ?? "null"}\``,
40
+ `- promotionApplied: \`false\` (registry lifecycle only; no alias/incumbent)`,
41
+ "",
42
+ "## Content refs",
43
+ "",
44
+ ...record.manifest.contentRefs.map((ref) => `- \`${ref.path}\` — \`${ref.sha256}\``),
45
+ "",
46
+ "## Lifecycle",
47
+ "",
48
+ ...record.events.map((event) => `- #${event.seq} ${event.from ?? "∅"} → ${event.to}: ${event.reason} (${event.at})`),
49
+ "",
50
+ ];
51
+ return `${lines.join("\n")}\n`;
52
+ }
@@ -0,0 +1,289 @@
1
+ import { createHash } from "node:crypto";
2
+ import { readFile } from "node:fs/promises";
3
+ import path from "node:path";
4
+ import { reportDagUseCase } from "../dag/report-dag.js";
5
+ import { readReplaySpec, writeReplayArtifacts, } from "../../infrastructure/evaluation/store.js";
6
+ function sha256(content) {
7
+ return createHash("sha256").update(content).digest("hex");
8
+ }
9
+ function pairKey(input) {
10
+ return `${input.split}\u0000${input.taskRef}\u0000${input.seed}`;
11
+ }
12
+ async function verifyEvidenceHash(input) {
13
+ const content = await readFile(input.filePath);
14
+ const actual = sha256(content);
15
+ if (actual !== input.expected) {
16
+ throw new Error(`${input.label} hash mismatch: expected ${input.expected}, got ${actual}`);
17
+ }
18
+ }
19
+ function sumOptional(values) {
20
+ if (values.some((value) => value === undefined)) {
21
+ return { value: null, missing: true };
22
+ }
23
+ return {
24
+ value: values.reduce((sum, value) => sum + (value ?? 0), 0),
25
+ missing: false,
26
+ };
27
+ }
28
+ function verifyPassed(input) {
29
+ if (input.status !== "finished")
30
+ return false;
31
+ const verificationNodes = input.nodes.filter((node) => node.executor === "shell" ||
32
+ node.nodeId.includes("verify") ||
33
+ node.nodeId.includes("gate"));
34
+ return (verificationNodes.length > 0 &&
35
+ verificationNodes.every((node) => node.status === "FINISHED" &&
36
+ (!node.failureCategory || node.failureCategory === "success")));
37
+ }
38
+ function metricsForRun(run) {
39
+ const tokens = sumOptional(run.nodes.map((node) => node.tokensUsed));
40
+ const duration = sumOptional(run.nodes.map((node) => node.durationMs));
41
+ const missingFields = [];
42
+ if (tokens.missing)
43
+ missingFields.push("tokens");
44
+ if (duration.missing)
45
+ missingFields.push("durationMs");
46
+ return {
47
+ metrics: {
48
+ verifyPassed: verifyPassed(run),
49
+ tokens: tokens.value,
50
+ durationMs: duration.value,
51
+ executorCalls: run.nodes.length,
52
+ repairPasses: run.convergence?.currentPass ?? 0,
53
+ },
54
+ missingFields,
55
+ };
56
+ }
57
+ function compareRows(incumbent, challenger) {
58
+ const reasons = [];
59
+ let verdict;
60
+ if (incumbent.metrics.verifyPassed !== challenger.metrics.verifyPassed) {
61
+ verdict = challenger.metrics.verifyPassed
62
+ ? "challenger_win"
63
+ : "incumbent_win";
64
+ reasons.push("verification_outcome");
65
+ }
66
+ else if (!incumbent.metrics.verifyPassed) {
67
+ verdict = "tie";
68
+ reasons.push("both_failed_verification");
69
+ }
70
+ else if (incumbent.metrics.tokens === null ||
71
+ challenger.metrics.tokens === null ||
72
+ incumbent.metrics.durationMs === null ||
73
+ challenger.metrics.durationMs === null) {
74
+ verdict = "incomparable";
75
+ reasons.push("missing_cost_metrics");
76
+ }
77
+ else {
78
+ const challengerNoWorse = challenger.metrics.tokens <= incumbent.metrics.tokens &&
79
+ challenger.metrics.durationMs <= incumbent.metrics.durationMs;
80
+ const incumbentNoWorse = incumbent.metrics.tokens <= challenger.metrics.tokens &&
81
+ incumbent.metrics.durationMs <= challenger.metrics.durationMs;
82
+ if (challengerNoWorse && !incumbentNoWorse) {
83
+ verdict = "challenger_win";
84
+ reasons.push("lower_cost");
85
+ }
86
+ else if (incumbentNoWorse && !challengerNoWorse) {
87
+ verdict = "incumbent_win";
88
+ reasons.push("lower_cost");
89
+ }
90
+ else if (challengerNoWorse && incumbentNoWorse) {
91
+ verdict = "tie";
92
+ reasons.push("equal_metrics");
93
+ }
94
+ else {
95
+ verdict = "incomparable";
96
+ reasons.push("cost_tradeoff");
97
+ }
98
+ }
99
+ return {
100
+ taskRef: incumbent.taskRef,
101
+ seed: incumbent.seed,
102
+ split: incumbent.split,
103
+ incumbentRunId: incumbent.runId,
104
+ challengerRunId: challenger.runId,
105
+ verdict,
106
+ reasons,
107
+ };
108
+ }
109
+ function buildScorecard(input) {
110
+ const rows = [...input.rows].sort((left, right) => [
111
+ left.split,
112
+ left.taskRef,
113
+ String(left.seed).padStart(12, "0"),
114
+ left.candidateId,
115
+ left.runId,
116
+ ]
117
+ .join("\u0000")
118
+ .localeCompare([
119
+ right.split,
120
+ right.taskRef,
121
+ String(right.seed).padStart(12, "0"),
122
+ right.candidateId,
123
+ right.runId,
124
+ ].join("\u0000")));
125
+ const incumbentByKey = new Map(rows
126
+ .filter((row) => row.candidateId === input.incumbentCandidateId)
127
+ .map((row) => [pairKey(row), row]));
128
+ const challengerByKey = new Map(rows
129
+ .filter((row) => row.candidateId === input.challengerCandidateId)
130
+ .map((row) => [pairKey(row), row]));
131
+ const keys = [
132
+ ...new Set([...incumbentByKey.keys(), ...challengerByKey.keys()]),
133
+ ].sort();
134
+ const comparisons = [];
135
+ let unpairedEvidenceCount = 0;
136
+ for (const key of keys) {
137
+ const incumbent = incumbentByKey.get(key);
138
+ const challenger = challengerByKey.get(key);
139
+ if (!incumbent || !challenger) {
140
+ unpairedEvidenceCount +=
141
+ Number(Boolean(incumbent)) + Number(Boolean(challenger));
142
+ continue;
143
+ }
144
+ comparisons.push(compareRows(incumbent, challenger));
145
+ }
146
+ const reasons = ["replay_only"];
147
+ if (comparisons.length === 0)
148
+ reasons.push("insufficient_samples");
149
+ if (unpairedEvidenceCount > 0)
150
+ reasons.push("unpaired_evidence");
151
+ return {
152
+ schemaVersion: 1,
153
+ replayId: input.replayId,
154
+ incumbentCandidateId: input.incumbentCandidateId,
155
+ challengerCandidateId: input.challengerCandidateId,
156
+ rows,
157
+ comparisons,
158
+ aggregate: {
159
+ pairedSampleCount: comparisons.length,
160
+ incumbentWins: comparisons.filter((item) => item.verdict === "incumbent_win").length,
161
+ challengerWins: comparisons.filter((item) => item.verdict === "challenger_win").length,
162
+ ties: comparisons.filter((item) => item.verdict === "tie").length,
163
+ incomparable: comparisons.filter((item) => item.verdict === "incomparable").length,
164
+ unpairedEvidenceCount,
165
+ promotionEligible: false,
166
+ reasons,
167
+ },
168
+ };
169
+ }
170
+ export function formatReplayMarkdown(scorecard) {
171
+ const lines = [
172
+ `# Eval Replay Scorecard: ${scorecard.replayId}`,
173
+ "",
174
+ "> Replay-only evidence. This report never authorizes promotion or executes Pi/DAG work.",
175
+ "",
176
+ `- incumbent: ${scorecard.incumbentCandidateId}`,
177
+ `- challenger: ${scorecard.challengerCandidateId}`,
178
+ `- promotionEligible: ${scorecard.aggregate.promotionEligible}`,
179
+ `- reasons: ${scorecard.aggregate.reasons.join(", ")}`,
180
+ "",
181
+ "## Evidence",
182
+ "",
183
+ "| split | task | seed | candidate | run | verified | tokens | durationMs | calls | repairPasses | missing |",
184
+ "|---|---|---:|---|---|---|---:|---:|---:|---:|---|",
185
+ ];
186
+ for (const row of scorecard.rows) {
187
+ lines.push(`| ${row.split} | ${row.taskRef} | ${row.seed} | ${row.candidateId} | ${row.runId} | ${row.metrics.verifyPassed} | ${row.metrics.tokens ?? "n/a"} | ${row.metrics.durationMs ?? "n/a"} | ${row.metrics.executorCalls} | ${row.metrics.repairPasses} | ${row.missingFields.join(", ") || "none"} |`);
188
+ }
189
+ lines.push("", "## Paired Comparisons", "", "| split | task | seed | incumbent run | challenger run | verdict | reasons |", "|---|---|---:|---|---|---|---|");
190
+ for (const comparison of scorecard.comparisons) {
191
+ lines.push(`| ${comparison.split} | ${comparison.taskRef} | ${comparison.seed} | ${comparison.incumbentRunId} | ${comparison.challengerRunId} | ${comparison.verdict} | ${comparison.reasons.join(", ")} |`);
192
+ }
193
+ if (scorecard.comparisons.length === 0) {
194
+ lines.push("| - | - | - | - | - | - | no paired evidence |");
195
+ }
196
+ return `${lines.join("\n")}\n`;
197
+ }
198
+ export async function replayEvaluation(input) {
199
+ const spec = await readReplaySpec(input.repoRoot, input.specPath);
200
+ const rows = [];
201
+ for (const evidence of spec.evidence) {
202
+ const runDir = path.join(input.repoRoot, ".harness", "dag-runs", "completed", evidence.runId);
203
+ const statePath = path.join(runDir, "state.json");
204
+ const runPath = path.join(runDir, "run.json");
205
+ const stateRefPath = path
206
+ .relative(input.repoRoot, statePath)
207
+ .split(path.sep)
208
+ .join("/");
209
+ const runRefPath = path
210
+ .relative(input.repoRoot, runPath)
211
+ .split(path.sep)
212
+ .join("/");
213
+ await verifyEvidenceHash({
214
+ filePath: statePath,
215
+ expected: evidence.stateSha256,
216
+ label: "state.json",
217
+ });
218
+ await verifyEvidenceHash({
219
+ filePath: runPath,
220
+ expected: evidence.runSha256,
221
+ label: "run.json",
222
+ });
223
+ const report = await reportDagUseCase({
224
+ repoRoot: input.repoRoot,
225
+ runId: evidence.runId,
226
+ lifecycle: "completed",
227
+ failedOnly: false,
228
+ latest: false,
229
+ });
230
+ const run = report.runs[0];
231
+ if (!run) {
232
+ throw new Error(`completed DAG run not found: ${evidence.runId}`);
233
+ }
234
+ if (run.evaluationAssociation.status === "present") {
235
+ if (run.evaluationAssociation.candidateId !== evidence.candidateId) {
236
+ throw new Error(`replay evidence candidateId conflict for ${evidence.runId}: evidence=${evidence.candidateId}, run=${run.evaluationAssociation.candidateId}`);
237
+ }
238
+ if (run.evaluationAssociation.seed !== evidence.seed) {
239
+ throw new Error(`replay evidence seed conflict for ${evidence.runId}: evidence=${evidence.seed}, run=${run.evaluationAssociation.seed}`);
240
+ }
241
+ if (run.evaluationAssociation.split &&
242
+ run.evaluationAssociation.split !== evidence.split) {
243
+ throw new Error(`replay evidence split conflict for ${evidence.runId}: evidence=${evidence.split}, run=${run.evaluationAssociation.split}`);
244
+ }
245
+ if (run.evaluationAssociation.taskRef &&
246
+ run.evaluationAssociation.taskRef !== evidence.taskRef) {
247
+ throw new Error(`replay evidence taskRef conflict for ${evidence.runId}: evidence=${evidence.taskRef}, run=${run.evaluationAssociation.taskRef}`);
248
+ }
249
+ }
250
+ if (![
251
+ "finished",
252
+ "failed",
253
+ "partial_failed",
254
+ "superseded",
255
+ "abandoned",
256
+ ].includes(run.status)) {
257
+ throw new Error(`completed DAG run ${evidence.runId} is not terminal (status=${run.status})`);
258
+ }
259
+ const { metrics, missingFields } = metricsForRun(run);
260
+ rows.push({
261
+ ...evidence,
262
+ lifecycle: "completed",
263
+ runStatus: run.status,
264
+ metrics,
265
+ missingFields,
266
+ evidenceRefs: [
267
+ { path: stateRefPath, sha256: evidence.stateSha256 },
268
+ { path: runRefPath, sha256: evidence.runSha256 },
269
+ ],
270
+ });
271
+ }
272
+ const scorecard = buildScorecard({
273
+ replayId: spec.replayId,
274
+ incumbentCandidateId: spec.incumbentCandidateId,
275
+ challengerCandidateId: spec.challengerCandidateId,
276
+ rows,
277
+ });
278
+ const markdown = formatReplayMarkdown(scorecard);
279
+ if (input.writeArtifacts === false) {
280
+ return { scorecard, markdown };
281
+ }
282
+ const paths = await writeReplayArtifacts({
283
+ repoRoot: input.repoRoot,
284
+ replayId: spec.replayId,
285
+ scorecard,
286
+ markdown,
287
+ });
288
+ return { scorecard, markdown, ...paths };
289
+ }