@tea-agent/loop-agent 0.12.0 → 0.13.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (284) hide show
  1. package/AGENTS.md +155 -153
  2. package/CHANGELOG.md +338 -265
  3. package/README.md +345 -298
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/dag/generate-task-dag.js +28 -28
  7. package/dist/application/evaluation/candidate-hash.js +75 -0
  8. package/dist/application/evaluation/candidate.js +52 -0
  9. package/dist/application/evaluation/replay.js +289 -0
  10. package/dist/application/evaluation/types.js +130 -0
  11. package/dist/cli/command-definitions.js +27 -7
  12. package/dist/cli/program.js +8 -4
  13. package/dist/commands/cursor-prompt.js +6 -6
  14. package/dist/commands/eval.js +235 -0
  15. package/dist/commands/init.js +544 -506
  16. package/dist/commands/knowledge.js +129 -31
  17. package/dist/commands/loop-benchmark.js +11 -11
  18. package/dist/commands/pi-reuse-benchmark.js +16 -16
  19. package/dist/executors/pi-sdk-executor.js +38 -24
  20. package/dist/executors/shell-executor.js +34 -2
  21. package/dist/executors/shell-presets.js +20 -0
  22. package/dist/executors/shell-verification.js +7 -0
  23. package/dist/governance/manifest-types.js +4 -0
  24. package/dist/infrastructure/evaluation/candidate-store.js +435 -0
  25. package/dist/infrastructure/evaluation/store.js +40 -0
  26. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  27. package/dist/task/config-types.js +28 -1
  28. package/dist/task/runtime.js +27 -27
  29. package/dist/worker/cli.js +96 -1
  30. package/dist/worker/delivery/package.js +3 -3
  31. package/dist/worker/feature/decision-loader.js +37 -6
  32. package/dist/worker/feature/next-action.js +10 -2
  33. package/dist/worker/feature/ready-plan-projection.js +81 -0
  34. package/dist/worker/feature/reducer.js +2 -1
  35. package/dist/worker/feature/review.js +19 -2
  36. package/dist/worker/feature/run.js +27 -2
  37. package/dist/worker/follow-up/approve.js +5 -2
  38. package/dist/worker/follow-up/factory.js +1 -1
  39. package/dist/worker/observability/read-model.js +246 -41
  40. package/dist/worker/observe/routes.js +173 -15
  41. package/dist/worker/observe/spec-evidence.js +281 -0
  42. package/dist/worker/observe/static/api.js +46 -27
  43. package/dist/worker/observe/static/app.js +150 -150
  44. package/dist/worker/observe/static/constants.js +148 -148
  45. package/dist/worker/observe/static/copy.js +67 -67
  46. package/dist/worker/observe/static/dag-helpers.js +172 -172
  47. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  48. package/dist/worker/observe/static/dag-layout.js +83 -83
  49. package/dist/worker/observe/static/dag-model.js +72 -72
  50. package/dist/worker/observe/static/dom.js +61 -61
  51. package/dist/worker/observe/static/format-pool.js +67 -67
  52. package/dist/worker/observe/static/format.js +292 -292
  53. package/dist/worker/observe/static/index.html +308 -308
  54. package/dist/worker/observe/static/kpi.js +94 -94
  55. package/dist/worker/observe/static/relations.js +133 -128
  56. package/dist/worker/observe/static/router.js +93 -85
  57. package/dist/worker/observe/static/run-processing.js +148 -148
  58. package/dist/worker/observe/static/shell-chrome.js +68 -68
  59. package/dist/worker/observe/static/state.js +253 -253
  60. package/dist/worker/observe/static/styles.css +1902 -1890
  61. package/dist/worker/observe/static/views/batch.js +227 -226
  62. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  63. package/dist/worker/observe/static/views/dag-inspector.js +607 -477
  64. package/dist/worker/observe/static/views/dag.js +362 -362
  65. package/dist/worker/observe/static/views/dashboard.js +445 -442
  66. package/dist/worker/observe/static/views/failures.js +143 -143
  67. package/dist/worker/observe/static/views/feature.js +492 -453
  68. package/dist/worker/observe/static/views/pool.js +350 -347
  69. package/dist/worker/observe/static/views/run.js +453 -453
  70. package/dist/worker/observe/static/views/session-timeline.js +205 -205
  71. package/dist/worker/observe/static/views/shell.js +7 -7
  72. package/dist/worker/observe/static/views/task.js +314 -260
  73. package/dist/worker/observe/static/views/timeline.js +163 -163
  74. package/dist/worker/pool/doctor.js +165 -0
  75. package/dist/worker/pool/migrate-state.js +303 -0
  76. package/dist/worker/pool/run-store.js +205 -17
  77. package/dist/worker/pool/types.js +17 -1
  78. package/dist/worker/pool/validation.js +100 -15
  79. package/dist/worker/report/morning-report.js +12 -2
  80. package/dist/worker/runner/run-ready.js +41 -26
  81. package/dist/worker/task-graph/ready-planner.js +136 -0
  82. package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
  83. package/dist/workflows/dag/canvas-observer.js +275 -275
  84. package/dist/workflows/dag/convergence/controller.js +16 -8
  85. package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
  86. package/dist/workflows/dag/failure-routing.js +12 -1
  87. package/dist/workflows/dag/init-hybrid.js +2404 -360
  88. package/dist/workflows/dag/node-execution.js +9 -0
  89. package/dist/workflows/dag/prompt.js +9 -0
  90. package/dist/workflows/dag/report.js +35 -1
  91. package/dist/workflows/dag/runner.js +28 -2
  92. package/dist/workflows/dag/task-demand-routing.js +383 -0
  93. package/dist/workflows/dag/types.js +51 -13
  94. package/dist/workflows/dag/upstream-artifacts.js +1 -0
  95. package/dist/workflows/dag/validate.js +59 -1
  96. package/docs/README.md +106 -104
  97. package/docs/agent-dag-recovery-playbook.md +195 -184
  98. package/docs/agent-dag-runner.md +67 -67
  99. package/docs/architecture/README.md +26 -26
  100. package/docs/architecture/dag-execution.md +140 -140
  101. package/docs/architecture/evolution.md +54 -53
  102. package/docs/architecture/facts-and-state.md +71 -58
  103. package/docs/architecture/runtime-boundaries.md +191 -191
  104. package/docs/architecture/system-overview.md +93 -93
  105. package/docs/architecture/worker-and-feature.md +85 -81
  106. package/docs/cursor-prompt-sidecar.md +36 -36
  107. package/docs/decisions/README.md +18 -15
  108. package/docs/design/README.md +167 -77
  109. package/docs/development-principles.md +73 -73
  110. package/docs/exec-plans/README.md +6 -6
  111. package/docs/exec-plans/active/README.md +15 -9
  112. package/docs/exec-plans/completed/README.md +85 -73
  113. package/docs/feature-workflow.md +389 -261
  114. package/docs/harness-methodology-debugging.md +153 -153
  115. package/docs/harness-methodology-tdd.md +130 -130
  116. package/docs/harness-methodology-verification.md +27 -27
  117. package/docs/init-surface.manifest.json +289 -280
  118. package/docs/loop-agent-harness.md +142 -130
  119. package/docs/production-readiness.md +96 -96
  120. package/docs/progress/README.md +64 -54
  121. package/docs/reports/README.md +117 -94
  122. package/docs/skills/README.md +7 -7
  123. package/docs/skills/vetted-skill-registry.md +29 -27
  124. package/docs/templates/adr.md +60 -60
  125. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  126. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  127. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  128. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  129. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  130. package/docs/templates/agent-dag-report.schema.json +473 -473
  131. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  132. package/docs/templates/agent-dag.base.json +190 -190
  133. package/docs/templates/agent-dag.final-verification.json +185 -185
  134. package/docs/templates/agent-dag.schema.json +411 -383
  135. package/docs/templates/agent-dag.supervised-implementation.json +501 -501
  136. package/docs/templates/backend-test-analysis.schema.json +44 -0
  137. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
  138. package/docs/templates/backend-test-dag.json +311 -276
  139. package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
  140. package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
  141. package/docs/templates/exec-plan.md +64 -64
  142. package/docs/templates/feature-spec.md +53 -53
  143. package/docs/templates/frontend-design-contract.md +42 -33
  144. package/docs/templates/frontend-task-constraints.md +35 -25
  145. package/docs/templates/frontend-task-requirement.md +70 -61
  146. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
  147. package/docs/templates/frontend-test-dag.json +23 -0
  148. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
  149. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
  150. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
  151. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
  152. package/docs/templates/harness.schema.json +221 -221
  153. package/docs/templates/hybrid-dag.json +188 -188
  154. package/docs/templates/init-evolution-review.md +35 -35
  155. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  156. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -0
  157. package/docs/templates/knowledge-sync-dag.json +178 -0
  158. package/docs/templates/knowledge-sync-draft.schema.json +71 -0
  159. package/docs/templates/product-line/AGENTS.md +8 -8
  160. package/docs/templates/product-line/README.md +9 -9
  161. package/docs/templates/product-line/acceptance.yaml +14 -14
  162. package/docs/templates/product-line/closeout.yaml +9 -9
  163. package/docs/templates/product-line/design.md +13 -13
  164. package/docs/templates/product-line/links.md +10 -10
  165. package/docs/templates/product-line/requirement.md +17 -17
  166. package/docs/templates/product-line/task-graph.yaml +15 -15
  167. package/docs/templates/product-line/task.yaml +64 -64
  168. package/docs/templates/product-line/test-plan.md +7 -7
  169. package/docs/templates/production-readiness-checklist.md +57 -57
  170. package/docs/templates/progress-log.md +17 -17
  171. package/docs/templates/project-start-checklist.md +9 -9
  172. package/docs/templates/qa-report.md +48 -48
  173. package/docs/templates/sprint-contract.md +29 -29
  174. package/docs/templates/worker-dogfood-evidence.md +80 -80
  175. package/docs/templates/worker-dogfood-setup.md +68 -68
  176. package/docs/verification-matrix.md +70 -66
  177. package/examples/decision-gate-agent-dag.json +177 -177
  178. package/examples/example-dag.json +46 -46
  179. package/examples/hybrid-loop-agent-dag.json +189 -189
  180. package/harness.json +66 -66
  181. package/package.json +88 -46
  182. package/scripts/check-product-line-docs.sh +29 -29
  183. package/scripts/check-task-pool-root.sh +32 -32
  184. package/scripts/kb-bootstrap-init-skeleton.sh +240 -0
  185. package/scripts/kb-graph-incremental-prepare.mjs +386 -0
  186. package/scripts/kb-graph-incremental-prepare.sh +5 -0
  187. package/scripts/kb-graph-materialize.mjs +105 -0
  188. package/scripts/kb-graph-materialize.sh +4 -0
  189. package/scripts/kb-graph-promote.mjs +164 -0
  190. package/scripts/kb-graph-promote.sh +4 -0
  191. package/scripts/kb-query.mjs +554 -0
  192. package/scripts/kb-query.sh +5 -0
  193. package/skills/agent-worker/SKILL.md +39 -37
  194. package/skills/agent-worker/references/agent-worker-operator.md +60 -43
  195. package/skills/ai-engineering-context/SKILL.md +48 -48
  196. package/skills/analyze-product-dependencies/SKILL.md +67 -0
  197. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
  198. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
  199. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
  200. package/skills/analyze-product-dependencies/references/example.md +76 -0
  201. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
  202. package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
  203. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
  204. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
  205. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
  206. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
  207. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
  208. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
  209. package/skills/analyze-product-requirements/SKILL.md +90 -0
  210. package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
  211. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
  212. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
  213. package/skills/analyze-product-requirements/references/example.md +86 -0
  214. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
  215. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
  216. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
  217. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
  218. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
  219. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
  220. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
  221. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
  222. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
  223. package/skills/code-review-core/SKILL.md +20 -20
  224. package/skills/codebase-scout/SKILL.md +19 -19
  225. package/skills/frontend-design-review/SKILL.md +66 -59
  226. package/skills/frontend-design-review/references/review-checklist.md +58 -37
  227. package/skills/frontend-implementation/SKILL.md +47 -51
  228. package/skills/frontend-implementation/references/code-standards.md +32 -34
  229. package/skills/frontend-implementation/references/design-spec.md +46 -46
  230. package/skills/frontend-implementation/references/node-contracts.md +76 -32
  231. package/skills/frontend-review/SKILL.md +59 -53
  232. package/skills/frontend-review/references/review-findings.md +47 -42
  233. package/skills/frontend-verification/SKILL.md +53 -40
  234. package/skills/frontend-verification/references/verification-checklist.md +68 -56
  235. package/skills/grill-me/SKILL.md +10 -10
  236. package/skills/grill-with-docs/SKILL.md +88 -88
  237. package/skills/grill-with-docs/adr-format.md +47 -47
  238. package/skills/grill-with-docs/context-format.md +60 -60
  239. package/skills/init-capability-evolution/SKILL.md +70 -70
  240. package/skills/loop-agent/SKILL.md +151 -151
  241. package/skills/loop-agent/references/README.md +67 -67
  242. package/skills/loop-agent/references/command-reference.md +505 -452
  243. package/skills/loop-agent/references/docs-converge.md +126 -126
  244. package/skills/loop-agent/references/harness-policy.md +263 -263
  245. package/skills/loop-agent/references/hybrid-dag.md +238 -233
  246. package/skills/loop-agent/references/learned/README.md +21 -21
  247. package/skills/loop-agent/references/long-running-loop.md +57 -57
  248. package/skills/loop-agent/references/model-routing.md +36 -36
  249. package/skills/loop-agent/references/multi-worktree.md +54 -54
  250. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  251. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  252. package/skills/loop-agent/references/pi-prompt.md +23 -23
  253. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  254. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  255. package/skills/loop-agent/references/task-workflow.md +89 -89
  256. package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
  257. package/skills/playwright-cli/SKILL.md +420 -0
  258. package/skills/playwright-cli/references/element-attributes.md +23 -0
  259. package/skills/playwright-cli/references/playwright-tests.md +39 -0
  260. package/skills/playwright-cli/references/request-mocking.md +87 -0
  261. package/skills/playwright-cli/references/running-code.md +241 -0
  262. package/skills/playwright-cli/references/session-management.md +225 -0
  263. package/skills/playwright-cli/references/storage-state.md +275 -0
  264. package/skills/playwright-cli/references/test-generation.md +433 -0
  265. package/skills/playwright-cli/references/tracing.md +139 -0
  266. package/skills/playwright-cli/references/video-recording.md +143 -0
  267. package/skills/playwright-cli-case-generator/SKILL.md +74 -0
  268. package/skills/requesting-code-review/SKILL.md +101 -101
  269. package/skills/requesting-code-review/code-reviewer.md +168 -168
  270. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  271. package/skills/systematic-debugging/SKILL.md +296 -296
  272. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  273. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  274. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  275. package/skills/systematic-debugging/find-polluter.sh +63 -63
  276. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  277. package/skills/systematic-debugging/test-academic.md +14 -14
  278. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  279. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  280. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  281. package/skills/test-driven-development/SKILL.md +20 -20
  282. package/skills/using-git-worktrees/SKILL.md +215 -215
  283. package/skills/verification-before-completion/SKILL.md +154 -154
  284. package/skills/webapp-testing/SKILL.md +19 -19
@@ -5,9 +5,11 @@ import { redactSecrets, truncateUtf8Preview } from "../../shared/preview.js";
5
5
  import { parseWorkerEventLine } from "../observability/events.js";
6
6
  import { isSafeObservabilityIdentifier } from "../observability/event-store.js";
7
7
  import { clampEventHistoryLimit, listBatchEventHistory, listPoolEventHistory, } from "../observability/event-history.js";
8
- import { buildGlobalSnapshot, clampTaskRunHistoryLimit, listTaskRunHistory, } from "../observability/read-model.js";
8
+ import { buildGlobalSnapshot, clampTaskRunHistoryLimit, listTaskRunHistory, resolveLegacyTask, } from "../observability/read-model.js";
9
+ import { dagSourceBindingSchema } from "../../workflows/dag/types.js";
9
10
  import { getTaskPoolRoot } from "../pool/run-store.js";
10
11
  import { isAllowedArtifactTextPath, resolveArtifactPath, toRepoRelativeArtifactPath, } from "./paths.js";
12
+ import { extractSpecEvidence, } from "./spec-evidence.js";
11
13
  const ARTIFACT_PREVIEW_MAX_BYTES = 64 * 1024;
12
14
  export function createObserveSnapshotCache() {
13
15
  return { expiresAt: 0 };
@@ -32,6 +34,17 @@ const ROUTES = [
32
34
  */
33
35
  { method: "GET", pattern: /^\/api\/events$/, handler: handlePoolEvents },
34
36
  { method: "GET", pattern: /^\/api\/tasks$/, handler: handleTasks },
37
+ // Canonical feature-scoped task routes (must precede legacy /api/tasks/:id).
38
+ {
39
+ method: "GET",
40
+ pattern: /^\/api\/features\/([^/]+)\/tasks\/([^/]+)\/runs$/,
41
+ handler: handleFeatureTaskRuns,
42
+ },
43
+ {
44
+ method: "GET",
45
+ pattern: /^\/api\/features\/([^/]+)\/tasks\/([^/]+)$/,
46
+ handler: handleFeatureTaskById,
47
+ },
35
48
  {
36
49
  method: "GET",
37
50
  pattern: /^\/api\/tasks\/([^/]+)\/runs$/,
@@ -58,6 +71,11 @@ const ROUTES = [
58
71
  pattern: /^\/api\/dag-runs\/([^/]+)\/nodes\/([^/]+)\/session-events$/,
59
72
  handler: handleDagNodeSessionEvents,
60
73
  },
74
+ {
75
+ method: "GET",
76
+ pattern: /^\/api\/dag-runs\/([^/]+)\/nodes\/([^/]+)\/spec-evidence$/,
77
+ handler: handleDagNodeSpecEvidence,
78
+ },
61
79
  {
62
80
  method: "GET",
63
81
  pattern: /^\/api\/dag-runs\/([^/]+)$/,
@@ -208,15 +226,71 @@ async function handleTasks(_req, res, _match, ctx) {
208
226
  const snapshot = await getSnapshot(ctx);
209
227
  sendJson(res, 200, snapshot.tasks);
210
228
  }
211
- async function handleTaskById(_req, res, match, ctx) {
229
+ async function handleFeatureTaskById(_req, res, match, ctx) {
230
+ const resolved = resolveFeatureTaskParams(match);
231
+ if (!resolved) {
232
+ sendJson(res, 400, { error: "Invalid task identifier" });
233
+ return;
234
+ }
235
+ if (!isSafeObservabilityIdentifier(resolved.featureId) ||
236
+ !isSafeObservabilityIdentifier(resolved.taskId)) {
237
+ sendJson(res, 400, { error: "Invalid task identifier" });
238
+ return;
239
+ }
212
240
  const snapshot = await getSnapshot(ctx);
213
- const task = snapshot.tasks.find((t) => t.taskId === match.params.id);
241
+ const task = snapshot.tasks.find((t) => t.featureId === resolved.featureId && t.taskId === resolved.taskId);
214
242
  if (!task) {
215
243
  sendJson(res, 404, { error: "Task not found" });
216
244
  return;
217
245
  }
218
246
  sendJson(res, 200, task);
219
247
  }
248
+ async function handleFeatureTaskRuns(_req, res, match, ctx) {
249
+ const resolved = resolveFeatureTaskParams(match);
250
+ if (!resolved) {
251
+ sendJson(res, 400, { error: "Invalid task identifier" });
252
+ return;
253
+ }
254
+ if (!isSafeObservabilityIdentifier(resolved.featureId) ||
255
+ !isSafeObservabilityIdentifier(resolved.taskId)) {
256
+ sendJson(res, 400, { error: "Invalid task identifier" });
257
+ return;
258
+ }
259
+ const limit = clampTaskRunHistoryLimit(parsePositiveInt(match.query.get("limit"), 0) || undefined);
260
+ const before = match.query.get("before");
261
+ const page = await listTaskRunHistory(ctx.repoRoot, resolved.featureId, resolved.taskId, {
262
+ limit,
263
+ before: before && before.length > 0 ? before : null,
264
+ });
265
+ // 404 only when neither snapshot row nor ledger history exists for this composite id.
266
+ if (page.runs.length === 0) {
267
+ const snapshot = await getSnapshot(ctx);
268
+ const known = snapshot.tasks.some((t) => t.featureId === resolved.featureId && t.taskId === resolved.taskId);
269
+ if (!known) {
270
+ sendJson(res, 404, { error: "Task not found" });
271
+ return;
272
+ }
273
+ }
274
+ sendJson(res, 200, page);
275
+ }
276
+ async function handleTaskById(_req, res, match, ctx) {
277
+ const taskId = match.params.id;
278
+ if (!isSafeObservabilityIdentifier(taskId)) {
279
+ sendJson(res, 400, { error: "Invalid task identifier" });
280
+ return;
281
+ }
282
+ const snapshot = await getSnapshot(ctx);
283
+ const resolved = resolveLegacyTask(snapshot.tasks, taskId);
284
+ if (resolved.kind === "missing") {
285
+ sendJson(res, 404, { error: "Task not found" });
286
+ return;
287
+ }
288
+ if (resolved.kind === "ambiguous") {
289
+ sendJson(res, 409, ambiguousTaskBody(taskId, resolved.candidates));
290
+ return;
291
+ }
292
+ sendJson(res, 200, resolved.task);
293
+ }
220
294
  async function handleTaskRuns(_req, res, match, ctx) {
221
295
  const taskId = match.params.id;
222
296
  if (!isSafeObservabilityIdentifier(taskId)) {
@@ -224,26 +298,45 @@ async function handleTaskRuns(_req, res, match, ctx) {
224
298
  return;
225
299
  }
226
300
  const snapshot = await getSnapshot(ctx);
227
- const known = snapshot.tasks.some((t) => t.taskId === taskId) ||
228
- snapshot.batches.some((b) => b.tasks.some((t) => t.taskId === taskId));
229
- if (!known) {
230
- // Still allow history when ledger has runs but current snapshot has no task row
231
- const pageProbe = await listTaskRunHistory(ctx.repoRoot, taskId, {
232
- limit: 1,
233
- });
234
- if (pageProbe.runs.length === 0) {
235
- sendJson(res, 404, { error: "Task not found" });
236
- return;
237
- }
301
+ const resolved = resolveLegacyTask(snapshot.tasks, taskId);
302
+ if (resolved.kind === "ambiguous") {
303
+ sendJson(res, 409, ambiguousTaskBody(taskId, resolved.candidates));
304
+ return;
305
+ }
306
+ if (resolved.kind === "missing") {
307
+ // No snapshot row: do not guess featureId for history.
308
+ sendJson(res, 404, { error: "Task not found" });
309
+ return;
310
+ }
311
+ const featureId = resolved.task.featureId;
312
+ if (!featureId) {
313
+ sendJson(res, 404, { error: "Task not found" });
314
+ return;
238
315
  }
239
316
  const limit = clampTaskRunHistoryLimit(parsePositiveInt(match.query.get("limit"), 0) || undefined);
240
317
  const before = match.query.get("before");
241
- const page = await listTaskRunHistory(ctx.repoRoot, taskId, {
318
+ const page = await listTaskRunHistory(ctx.repoRoot, featureId, taskId, {
242
319
  limit,
243
320
  before: before && before.length > 0 ? before : null,
244
321
  });
245
322
  sendJson(res, 200, page);
246
323
  }
324
+ function ambiguousTaskBody(taskId, candidates) {
325
+ return {
326
+ error: "Task identity ambiguous",
327
+ taskId,
328
+ candidates,
329
+ };
330
+ }
331
+ /** Extract featureId/taskId from /api/features/:featureId/tasks/:taskId[...]. */
332
+ function resolveFeatureTaskParams(match) {
333
+ // handleRequest maps capture groups to params.id (group 1) and params.sub (group 2).
334
+ const featureId = match.params.featureId ?? match.params.id;
335
+ const taskId = match.params.taskId ?? match.params.sub;
336
+ if (!featureId || !taskId)
337
+ return null;
338
+ return { featureId, taskId };
339
+ }
247
340
  async function handleRunById(_req, res, match, ctx) {
248
341
  if (!isSafeObservabilityIdentifier(match.params.id)) {
249
342
  sendJson(res, 400, { error: "Invalid run identifier" });
@@ -613,6 +706,71 @@ function sendJson(res, status, body) {
613
706
  });
614
707
  res.end(payload);
615
708
  }
709
+ async function handleDagNodeSpecEvidence(_req, res, match, ctx) {
710
+ const dagRunId = match.params.id;
711
+ const nodeId = match.params.sub;
712
+ if (!isSafeObservabilityIdentifier(dagRunId) || !isSafeObservabilityIdentifier(nodeId)) {
713
+ sendJson(res, 400, { error: "Invalid dag run or node identifier" });
714
+ return;
715
+ }
716
+ // Extract skill injection info from the DAG run spec (run.json)
717
+ const skillInjection = { skills: [], references: [] };
718
+ let sourceBinding;
719
+ const dagRunsRoot = path.resolve(ctx.repoRoot, ".harness", "dag-runs");
720
+ for (const lifecycle of ["active", "completed", "paused"]) {
721
+ const runJsonPath = path.join(dagRunsRoot, lifecycle, dagRunId, "run.json");
722
+ try {
723
+ const runRaw = await readFile(runJsonPath, "utf-8");
724
+ const runSpec = JSON.parse(runRaw);
725
+ const parsedSourceBinding = dagSourceBindingSchema.safeParse(runSpec.sourceBinding);
726
+ if (parsedSourceBinding.success)
727
+ sourceBinding = parsedSourceBinding.data;
728
+ const task = runSpec.tasks?.find((t) => t.id === nodeId);
729
+ if (task?.skills && Array.isArray(task.skills)) {
730
+ skillInjection.skills = task.skills;
731
+ }
732
+ break;
733
+ }
734
+ catch {
735
+ // run.json may not exist; continue to next lifecycle
736
+ }
737
+ }
738
+ const evidence = await extractSpecEvidence(ctx.repoRoot, dagRunId, nodeId);
739
+ if (!evidence) {
740
+ const hasSourceBinding = Boolean(sourceBinding);
741
+ sendJson(res, 200, {
742
+ dagRunId,
743
+ nodeId,
744
+ status: hasSourceBinding ? "source-bound" : "no-evidence",
745
+ skillInjection,
746
+ sourceBinding,
747
+ specReads: [],
748
+ specSearches: [],
749
+ knowledgeBaseQueries: [],
750
+ summary: hasSourceBinding
751
+ ? `已绑定 ${sourceBinding.sources.length} 个任务源文件和 ${sourceBinding.requirementIds.length} 个显式需求编号;未观察到 read 工具调用。`
752
+ : "未找到该节点的 session events 记录。",
753
+ });
754
+ return;
755
+ }
756
+ evidence.skillInjection = skillInjection;
757
+ evidence.sourceBinding = sourceBinding;
758
+ // Recompute status considering skill injection
759
+ if (evidence.specReads.length === 0 && evidence.knowledgeBaseQueries.length === 0) {
760
+ if (evidence.specSearches.length > 0) {
761
+ evidence.status = "search-only";
762
+ }
763
+ else if (sourceBinding) {
764
+ evidence.status = "source-bound";
765
+ evidence.summary = `已绑定 ${sourceBinding.sources.length} 个任务源文件和 ${sourceBinding.requirementIds.length} 个显式需求编号;未观察到 read 工具调用。`;
766
+ }
767
+ else if (skillInjection.skills.length > 0) {
768
+ evidence.status = "spec-injected";
769
+ evidence.summary = "规范 skill 已注入但未观察到规范文件读取或知识库查询。";
770
+ }
771
+ }
772
+ sendJson(res, 200, evidence);
773
+ }
616
774
  function contentTypeFor(pathname) {
617
775
  if (pathname.endsWith(".html"))
618
776
  return "text/html; charset=utf-8";
@@ -0,0 +1,281 @@
1
+ import { existsSync } from "node:fs";
2
+ import { readFile } from "node:fs/promises";
3
+ import path from "node:path";
4
+ import { isSafeObservabilityIdentifier } from "../observability/event-store.js";
5
+ const DAG_RUN_LIFECYCLE_DIRS = ["active", "completed", "paused"];
6
+ /**
7
+ * Known knowledge-base connector tool names.
8
+ */
9
+ const KB_CONNECTOR_TOOLS = new Set([
10
+ "knowledge-base-query",
11
+ "kb_query",
12
+ "kb-search",
13
+ "knowledge_base_search",
14
+ "rag_search",
15
+ "vector_search",
16
+ "mcp__knowledge",
17
+ "mcp__kb",
18
+ ]);
19
+ /**
20
+ * Pattern for detecting spec-related files:
21
+ * - openSpec/** files
22
+ * - *.spec.md / *.spec.ts / *.spec.tsx
23
+ * - project-specs/**
24
+ * - design-spec.md, code-standards.md, review-checklist.md, etc.
25
+ * - SKILL.md files
26
+ * - docs/** specification files
27
+ */
28
+ const SPEC_FILE_PATTERNS = [
29
+ /openspec\//i,
30
+ /\/openSpec\//i,
31
+ /\/project-specs\//i,
32
+ /\/spec\//i,
33
+ /\.spec\.(md|tsx?|jsx?)$/i,
34
+ /\bdesign-spec\b/i,
35
+ /\bcode-standards\b/i,
36
+ /\breview-checklist\b/i,
37
+ /\bverification-checklist\b/i,
38
+ /\bnode-contracts\b/i,
39
+ /\breview-findings\b/i,
40
+ /\bSKILL\.md$/i,
41
+ /\bskills\//i,
42
+ /\bdocs\/design\//i,
43
+ /\bdocs\/templates\//i,
44
+ ];
45
+ function isSpecFilePath(filePath) {
46
+ return SPEC_FILE_PATTERNS.some((pattern) => pattern.test(filePath));
47
+ }
48
+ function isKnowledgeBaseTool(toolName) {
49
+ return KB_CONNECTOR_TOOLS.has(toolName);
50
+ }
51
+ function resolveSessionEventsPath(repoRoot, dagRunId, nodeId) {
52
+ if (!isSafeObservabilityIdentifier(dagRunId) || !isSafeObservabilityIdentifier(nodeId)) {
53
+ return null;
54
+ }
55
+ const dagRunsRoot = path.resolve(repoRoot, ".harness", "dag-runs");
56
+ const candidates = [];
57
+ for (const lifecycle of DAG_RUN_LIFECYCLE_DIRS) {
58
+ candidates.push(path.join(dagRunsRoot, lifecycle, dagRunId, nodeId, "session-events.jsonl"));
59
+ }
60
+ for (const candidate of candidates) {
61
+ const resolved = path.resolve(candidate);
62
+ if (!isPathInside(dagRunsRoot, resolved))
63
+ continue;
64
+ if (!isExpectedSessionEventsRelativePath(dagRunsRoot, resolved, dagRunId, nodeId)) {
65
+ continue;
66
+ }
67
+ if (existsSync(resolved))
68
+ return resolved;
69
+ }
70
+ return null;
71
+ }
72
+ function isExpectedSessionEventsRelativePath(dagRunsRoot, resolvedPath, dagRunId, nodeId) {
73
+ const relative = path.relative(dagRunsRoot, resolvedPath);
74
+ const parts = relative.split(path.sep).filter(Boolean);
75
+ return (parts.length === 4 &&
76
+ DAG_RUN_LIFECYCLE_DIRS.includes(parts[0]) &&
77
+ parts[1] === dagRunId &&
78
+ parts[2] === nodeId &&
79
+ parts[3] === "session-events.jsonl");
80
+ }
81
+ function isPathInside(root, target) {
82
+ const normalizedRoot = root.endsWith(path.sep) ? root : `${root}${path.sep}`;
83
+ const normalizedTarget = path.normalize(target);
84
+ return normalizedTarget === root || normalizedTarget.startsWith(normalizedRoot);
85
+ }
86
+ function resolveToolName(event) {
87
+ return (event.toolName ?? "").toLowerCase();
88
+ }
89
+ function eventTimestamp(event) {
90
+ return event.timestamp ?? event.at;
91
+ }
92
+ /**
93
+ * Parse a tool input path to a repo-relative path.
94
+ * Handles both absolute paths and paths that are already repo-relative.
95
+ */
96
+ function toRepoRelative(repoRoot, filePath) {
97
+ if (!filePath)
98
+ return null;
99
+ try {
100
+ const resolved = path.resolve(repoRoot, filePath);
101
+ const rel = path.relative(repoRoot, resolved);
102
+ if (rel === "" || rel.startsWith("..") || path.isAbsolute(rel)) {
103
+ return null;
104
+ }
105
+ return rel.replaceAll(path.sep, "/");
106
+ }
107
+ catch {
108
+ return null;
109
+ }
110
+ }
111
+ async function readSessionEvents(filePath) {
112
+ try {
113
+ const raw = await readFile(filePath, "utf-8");
114
+ return raw
115
+ .split("\n")
116
+ .map((line) => line.trim())
117
+ .filter(Boolean)
118
+ .map((line) => {
119
+ try {
120
+ return JSON.parse(line);
121
+ }
122
+ catch {
123
+ return null;
124
+ }
125
+ })
126
+ .filter((event) => event !== null);
127
+ }
128
+ catch {
129
+ return [];
130
+ }
131
+ }
132
+ export async function extractSpecEvidence(repoRoot, dagRunId, nodeId) {
133
+ if (!isSafeObservabilityIdentifier(dagRunId) || !isSafeObservabilityIdentifier(nodeId)) {
134
+ return null;
135
+ }
136
+ const eventsPath = resolveSessionEventsPath(repoRoot, dagRunId, nodeId);
137
+ if (!eventsPath) {
138
+ return null;
139
+ }
140
+ const events = await readSessionEvents(eventsPath);
141
+ // Track successful read calls with their paths (dedup by path)
142
+ const readPaths = new Map();
143
+ const searches = [];
144
+ const kbQueries = [];
145
+ // Track tool_execution_start events by toolCallId for pairing.
146
+ // Only tool_execution_end events with a matching start are counted as reads.
147
+ const startMap = new Map();
148
+ // Scan events for tool calls
149
+ for (const event of events) {
150
+ const type = event.type;
151
+ const toolName = resolveToolName(event);
152
+ const toolCallId = event.toolCallId;
153
+ const toolInput = event.toolInput ?? event.input ?? {};
154
+ const ts = eventTimestamp(event);
155
+ // Collect tool_execution_start events by toolCallId
156
+ if (type === "tool_execution_start" && toolCallId) {
157
+ startMap.set(toolCallId, { toolName, input: toolInput });
158
+ continue;
159
+ }
160
+ // Process tool_execution_end events
161
+ if (type === "tool_execution_end") {
162
+ // Successful read calls: must have matching start event and no error
163
+ if (toolName === "read") {
164
+ const isErrored = event.isError === true ||
165
+ event.toolResult?.error != null ||
166
+ event.toolResult?.ok === false;
167
+ if (isErrored)
168
+ continue;
169
+ let filePath;
170
+ // Only paired read calls qualify: must have toolCallId, matching start
171
+ // with same toolName, and start must be a "read" tool.
172
+ if (toolCallId) {
173
+ const start = startMap.get(toolCallId);
174
+ if (start && start.toolName === "read") {
175
+ const startInput = start.input ?? {};
176
+ filePath = typeof startInput.path === "string" ? startInput.path : undefined;
177
+ startMap.delete(toolCallId);
178
+ }
179
+ // No fallback: unmatched end, mismatched toolName, or missing start → skip
180
+ }
181
+ // No legacy path: toolCallId-less read ends never qualify as spec reads
182
+ if (filePath) {
183
+ const relPath = toRepoRelative(repoRoot, filePath);
184
+ if (relPath && isSpecFilePath(relPath) && !readPaths.has(relPath)) {
185
+ readPaths.set(relPath, ts ?? "");
186
+ }
187
+ }
188
+ }
189
+ // Search/scan tool calls (grep, find, ls, glob)
190
+ if (["grep", "find", "ls", "glob"].includes(toolName)) {
191
+ const query = typeof toolInput.pattern === "string"
192
+ ? toolInput.pattern
193
+ : typeof toolInput.query === "string"
194
+ ? toolInput.query
195
+ : typeof toolInput.path === "string"
196
+ ? toolInput.path
197
+ : toolName;
198
+ searches.push({
199
+ tool: toolName,
200
+ query,
201
+ timestamp: ts,
202
+ });
203
+ }
204
+ // Knowledge base connector calls
205
+ if (isKnowledgeBaseTool(toolName)) {
206
+ kbQueries.push({
207
+ connector: toolName,
208
+ timestamp: ts,
209
+ });
210
+ }
211
+ }
212
+ }
213
+ // Determine status
214
+ let status;
215
+ const specReads = Array.from(readPaths.entries()).map(([p, t]) => ({
216
+ path: p,
217
+ timestamp: t || undefined,
218
+ }));
219
+ const hasInjection = false; // Injection is authoritative only after run.json is inspected by the route.
220
+ if (specReads.length > 0) {
221
+ status = "spec-read";
222
+ }
223
+ else if (kbQueries.length > 0) {
224
+ status = "kb-queried";
225
+ }
226
+ else if (searches.length > 0) {
227
+ status = "search-only";
228
+ }
229
+ else if (hasInjection) {
230
+ status = "spec-injected";
231
+ }
232
+ else {
233
+ status = "no-evidence";
234
+ }
235
+ // Build summary
236
+ const summaryLines = [];
237
+ if (status === "spec-read") {
238
+ summaryLines.push(`已读取 ${specReads.length} 个规范文件:${specReads.map((r) => r.path).join("、")}`);
239
+ }
240
+ if (searches.length > 0) {
241
+ summaryLines.push(`已执行 ${searches.length} 次检索操作。`);
242
+ }
243
+ if (kbQueries.length > 0) {
244
+ summaryLines.push(`已观察到 ${kbQueries.length} 次知识库查询。`);
245
+ }
246
+ if (status === "spec-injected") {
247
+ summaryLines.push("规范 skill 已注入但未观察到规范文件读取或知识库查询。");
248
+ }
249
+ if (status === "no-evidence") {
250
+ summaryLines.push("未观察到任何规范证据:无 skill 注入、无文件读取、无检索操作。");
251
+ }
252
+ return {
253
+ dagRunId,
254
+ nodeId,
255
+ status,
256
+ skillInjection: {
257
+ skills: [],
258
+ references: [],
259
+ },
260
+ specReads,
261
+ specSearches: searches,
262
+ knowledgeBaseQueries: kbQueries,
263
+ summary: summaryLines.join(" "),
264
+ };
265
+ }
266
+ export async function extractSpecEvidenceWithSkills(repoRoot, dagRunId, nodeId, skillInjection) {
267
+ const base = await extractSpecEvidence(repoRoot, dagRunId, nodeId);
268
+ if (!base)
269
+ return null;
270
+ base.skillInjection = skillInjection;
271
+ // Recompute status considering skill injection knowledge
272
+ if (base.status === "no-evidence" && skillInjection.skills.length > 0) {
273
+ base.status = "spec-injected";
274
+ base.summary = "规范 skill 已注入但未观察到规范文件读取或知识库查询。";
275
+ }
276
+ else if (base.status === "spec-injected" ||
277
+ (base.status === "no-evidence" && skillInjection.skills.length === 0)) {
278
+ base.summary = base.summary || "规范 skill 已注入但未观察到规范文件读取或知识库查询。";
279
+ }
280
+ return base;
281
+ }
@@ -1,27 +1,46 @@
1
- /** Fetch and artifact helpers (Observe UI R5). */
2
-
3
- export async function fetchJson(url) {
4
- try {
5
- const res = await fetch(url);
6
- if (!res.ok) return null;
7
- return await res.json();
8
- } catch {
9
- return null;
10
- }
11
- }
12
-
13
- export function artifactUrl(artifactPath) {
14
- return `/api/artifacts?path=${encodeURIComponent(artifactPath)}&tail=200`;
15
- }
16
-
17
- export function parseArtifactPreviewResponse(text) {
18
- const parsed = JSON.parse(text);
19
- if (!parsed || typeof parsed.content !== "string") {
20
- throw new Error("Artifact preview response is invalid");
21
- }
22
- return {
23
- content: parsed.content,
24
- truncated: parsed.truncated === true,
25
- };
26
- }
27
-
1
+ /** Fetch and artifact helpers (Observe UI R5). */
2
+
3
+ export async function fetchJson(url) {
4
+ try {
5
+ const res = await fetch(url);
6
+ if (!res.ok) return null;
7
+ return await res.json();
8
+ } catch {
9
+ return null;
10
+ }
11
+ }
12
+
13
+ /**
14
+ * Read-only JSON fetch that preserves HTTP status so callers can distinguish
15
+ * 404 (missing) from 409 (ambiguous task identity). Body is parsed best-effort.
16
+ */
17
+ export async function fetchJsonResult(url) {
18
+ try {
19
+ const res = await fetch(url);
20
+ let body = null;
21
+ try {
22
+ body = await res.json();
23
+ } catch {
24
+ body = null;
25
+ }
26
+ return { ok: res.ok, status: res.status, body };
27
+ } catch {
28
+ return { ok: false, status: 0, body: null };
29
+ }
30
+ }
31
+
32
+ export function artifactUrl(artifactPath) {
33
+ return `/api/artifacts?path=${encodeURIComponent(artifactPath)}&tail=200`;
34
+ }
35
+
36
+ export function parseArtifactPreviewResponse(text) {
37
+ const parsed = JSON.parse(text);
38
+ if (!parsed || typeof parsed.content !== "string") {
39
+ throw new Error("Artifact preview response is invalid");
40
+ }
41
+ return {
42
+ content: parsed.content,
43
+ truncated: parsed.truncated === true,
44
+ };
45
+ }
46
+