@tea-agent/loop-agent 0.13.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (272) hide show
  1. package/AGENTS.md +157 -157
  2. package/CHANGELOG.md +116 -305
  3. package/README.md +357 -334
  4. package/bin/agent-worker.js +22 -22
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/commands/cursor-prompt.js +6 -6
  7. package/dist/commands/init.js +505 -505
  8. package/dist/commands/loop-benchmark.js +11 -11
  9. package/dist/commands/pi-reuse-benchmark.js +16 -16
  10. package/dist/executors/pi-event-serializer.js +33 -11
  11. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  12. package/dist/task/runtime.js +27 -27
  13. package/dist/worker/observe/spec-evidence.js +19 -10
  14. package/dist/worker/observe/static/api.js +46 -46
  15. package/dist/worker/observe/static/app.js +151 -150
  16. package/dist/worker/observe/static/constants.js +156 -148
  17. package/dist/worker/observe/static/copy.js +67 -67
  18. package/dist/worker/observe/static/dag-helpers.js +201 -172
  19. package/dist/worker/observe/static/dag-layout.d.ts +31 -31
  20. package/dist/worker/observe/static/dag-layout.js +83 -83
  21. package/dist/worker/observe/static/dag-model.js +72 -72
  22. package/dist/worker/observe/static/dom.js +122 -122
  23. package/dist/worker/observe/static/format-pool.d.ts +71 -0
  24. package/dist/worker/observe/static/format-pool.js +134 -67
  25. package/dist/worker/observe/static/format.js +317 -292
  26. package/dist/worker/observe/static/index.html +350 -308
  27. package/dist/worker/observe/static/kpi.js +100 -94
  28. package/dist/worker/observe/static/markdown-render.js +124 -0
  29. package/dist/worker/observe/static/relations.js +133 -133
  30. package/dist/worker/observe/static/router.js +93 -93
  31. package/dist/worker/observe/static/run-processing.js +148 -148
  32. package/dist/worker/observe/static/shell-chrome.js +74 -68
  33. package/dist/worker/observe/static/state.js +273 -267
  34. package/dist/worker/observe/static/styles.css +2504 -1902
  35. package/dist/worker/observe/static/views/batch.js +227 -227
  36. package/dist/worker/observe/static/views/dag-graph.js +172 -172
  37. package/dist/worker/observe/static/views/dag-inspector.js +530 -627
  38. package/dist/worker/observe/static/views/dag.js +371 -371
  39. package/dist/worker/observe/static/views/dashboard.js +86 -100
  40. package/dist/worker/observe/static/views/failures.js +143 -143
  41. package/dist/worker/observe/static/views/feature.js +492 -492
  42. package/dist/worker/observe/static/views/pool.js +708 -350
  43. package/dist/worker/observe/static/views/run.js +453 -453
  44. package/dist/worker/observe/static/views/session-timeline.js +771 -219
  45. package/dist/worker/observe/static/views/shell.js +7 -7
  46. package/dist/worker/observe/static/views/task.js +314 -314
  47. package/dist/worker/observe/static/views/timeline.js +163 -163
  48. package/dist/workflows/dag/canvas-observer.js +275 -275
  49. package/dist/workflows/dag/init-hybrid.js +27 -11
  50. package/docs/README.md +106 -104
  51. package/docs/architecture/README.md +26 -26
  52. package/docs/architecture/dag-execution.md +140 -140
  53. package/docs/architecture/evolution.md +54 -54
  54. package/docs/architecture/facts-and-state.md +71 -71
  55. package/docs/architecture/runtime-boundaries.md +191 -191
  56. package/docs/architecture/system-overview.md +93 -93
  57. package/docs/architecture/worker-and-feature.md +85 -85
  58. package/docs/harness-methodology-debugging.md +153 -153
  59. package/docs/harness-methodology-tdd.md +130 -130
  60. package/docs/harness-methodology-verification.md +27 -27
  61. package/docs/init-surface.manifest.json +304 -307
  62. package/docs/skills/README.md +7 -7
  63. package/docs/skills/vetted-skill-registry.md +29 -29
  64. package/docs/templates/adr.md +60 -60
  65. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  66. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  67. package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
  68. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  69. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  70. package/docs/templates/agent-dag-report.schema.json +473 -473
  71. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  72. package/docs/templates/agent-dag.base.json +190 -190
  73. package/docs/templates/agent-dag.final-verification.json +185 -185
  74. package/docs/templates/agent-dag.schema.json +411 -411
  75. package/docs/templates/agent-dag.supervised-implementation.json +620 -620
  76. package/docs/templates/backend-test-analysis.schema.json +44 -44
  77. package/docs/templates/backend-test-case-manifest.schema.json +190 -190
  78. package/docs/templates/backend-test-dag.classify.prompt.md +75 -75
  79. package/docs/templates/backend-test-dag.generate-pytest.prompt.md +204 -204
  80. package/docs/templates/backend-test-dag.json +559 -559
  81. package/docs/templates/backend-test-dag.retrospect.prompt.md +139 -139
  82. package/docs/templates/backend-test-dag.review-cases.prompt.md +83 -83
  83. package/docs/templates/backend-test-execution.schema.json +133 -133
  84. package/docs/templates/backend-test-result.schema.json +99 -99
  85. package/docs/templates/branch-merge-report.md +0 -1
  86. package/docs/templates/exec-plan.md +64 -64
  87. package/docs/templates/feature-spec.md +53 -53
  88. package/docs/templates/frontend-design-contract.md +42 -42
  89. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  90. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  91. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  92. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  93. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  94. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  95. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  96. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  97. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  98. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  99. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  100. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  101. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  102. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  103. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  104. package/docs/templates/frontend-eval/metrics.md +138 -138
  105. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  106. package/docs/templates/frontend-implementation-contract.schema.json +27 -27
  107. package/docs/templates/frontend-task-constraints.md +35 -35
  108. package/docs/templates/frontend-task-requirement.md +70 -70
  109. package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -5
  110. package/docs/templates/frontend-test-dag.json +23 -23
  111. package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -3
  112. package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -3
  113. package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -3
  114. package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -3
  115. package/docs/templates/harness.schema.json +221 -221
  116. package/docs/templates/hybrid-dag.json +188 -188
  117. package/docs/templates/init-evolution-review.md +35 -35
  118. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  119. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  120. package/docs/templates/knowledge-sync-dag.json +178 -178
  121. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  122. package/docs/templates/product-line/AGENTS.md +8 -8
  123. package/docs/templates/product-line/README.md +9 -9
  124. package/docs/templates/product-line/acceptance.yaml +14 -14
  125. package/docs/templates/product-line/closeout.yaml +9 -9
  126. package/docs/templates/product-line/design.md +13 -13
  127. package/docs/templates/product-line/links.md +10 -10
  128. package/docs/templates/product-line/requirement.md +17 -17
  129. package/docs/templates/product-line/task-graph.yaml +15 -15
  130. package/docs/templates/product-line/task.yaml +64 -64
  131. package/docs/templates/product-line/test-plan.md +7 -7
  132. package/docs/templates/production-readiness-checklist.md +57 -57
  133. package/docs/templates/progress-log.md +17 -17
  134. package/docs/templates/project-start-checklist.md +9 -9
  135. package/docs/templates/qa-report.md +48 -48
  136. package/docs/templates/sprint-contract.md +29 -29
  137. package/docs/templates/worker-dogfood-evidence.md +80 -80
  138. package/docs/templates/worker-dogfood-setup.md +68 -68
  139. package/examples/decision-gate-agent-dag.json +173 -173
  140. package/examples/example-dag.json +46 -46
  141. package/examples/hybrid-loop-agent-dag.json +188 -188
  142. package/harness.json +66 -66
  143. package/package.json +78 -52
  144. package/scripts/kb-bootstrap-init-skeleton.sh +240 -240
  145. package/scripts/kb-graph-incremental-prepare.mjs +386 -386
  146. package/scripts/kb-graph-materialize.mjs +105 -105
  147. package/scripts/kb-graph-promote.mjs +164 -164
  148. package/scripts/kb-query.mjs +554 -554
  149. package/skills/agent-worker/SKILL.md +39 -39
  150. package/skills/agent-worker/references/agent-worker-operator.md +60 -60
  151. package/skills/ai-engineering-context/SKILL.md +48 -48
  152. package/skills/analyze-product-dependencies/SKILL.md +67 -67
  153. package/skills/analyze-product-dependencies/agents/openai.yaml +4 -4
  154. package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -30
  155. package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -28
  156. package/skills/analyze-product-dependencies/references/example.md +76 -76
  157. package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -35
  158. package/skills/analyze-product-dependencies/references/input-contract.md +11 -11
  159. package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -61
  160. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -267
  161. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -101
  162. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -142
  163. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -76
  164. package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -146
  165. package/skills/analyze-product-requirements/SKILL.md +90 -90
  166. package/skills/analyze-product-requirements/agents/openai.yaml +4 -4
  167. package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -91
  168. package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -56
  169. package/skills/analyze-product-requirements/references/example.md +86 -86
  170. package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -66
  171. package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -32
  172. package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -33
  173. package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -35
  174. package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -193
  175. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -69
  176. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -97
  177. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -98
  178. package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -156
  179. package/skills/browser-tools/SKILL.md +196 -196
  180. package/skills/browser-tools/browser-content.js +103 -103
  181. package/skills/browser-tools/browser-cookies.js +35 -35
  182. package/skills/browser-tools/browser-eval.js +53 -53
  183. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  184. package/skills/browser-tools/browser-nav.js +44 -44
  185. package/skills/browser-tools/browser-pick.js +162 -162
  186. package/skills/browser-tools/browser-screenshot.js +34 -34
  187. package/skills/browser-tools/browser-start.js +86 -86
  188. package/skills/browser-tools/package-lock.json +2556 -2556
  189. package/skills/browser-tools/package.json +19 -19
  190. package/skills/code-review-core/SKILL.md +20 -20
  191. package/skills/codebase-scout/SKILL.md +19 -19
  192. package/skills/frontend-design-review/SKILL.md +66 -66
  193. package/skills/frontend-design-review/references/review-checklist.md +40 -58
  194. package/skills/frontend-implementation/SKILL.md +49 -49
  195. package/skills/frontend-implementation/references/code-standards.md +32 -32
  196. package/skills/frontend-implementation/references/design-spec.md +46 -46
  197. package/skills/frontend-implementation/references/node-contracts.md +27 -27
  198. package/skills/frontend-review/SKILL.md +61 -59
  199. package/skills/frontend-review/references/review-findings.md +48 -47
  200. package/skills/frontend-verification/SKILL.md +55 -53
  201. package/skills/frontend-verification/references/verification-checklist.md +59 -68
  202. package/skills/grill-me/SKILL.md +10 -10
  203. package/skills/grill-with-docs/SKILL.md +88 -88
  204. package/skills/grill-with-docs/adr-format.md +47 -47
  205. package/skills/grill-with-docs/context-format.md +60 -60
  206. package/skills/init-capability-evolution/SKILL.md +70 -70
  207. package/skills/loop-agent/SKILL.md +151 -151
  208. package/skills/loop-agent/references/README.md +67 -67
  209. package/skills/loop-agent/references/command-reference.md +527 -527
  210. package/skills/loop-agent/references/docs-converge.md +126 -126
  211. package/skills/loop-agent/references/harness-policy.md +263 -263
  212. package/skills/loop-agent/references/hybrid-dag.md +243 -243
  213. package/skills/loop-agent/references/learned/README.md +21 -21
  214. package/skills/loop-agent/references/long-running-loop.md +57 -57
  215. package/skills/loop-agent/references/model-routing.md +36 -36
  216. package/skills/loop-agent/references/multi-worktree.md +54 -54
  217. package/skills/loop-agent/references/one-shot-runs.md +85 -85
  218. package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
  219. package/skills/loop-agent/references/pi-prompt.md +23 -23
  220. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  221. package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
  222. package/skills/loop-agent/references/task-workflow.md +89 -89
  223. package/skills/loop-agent/references/verification-and-failure-handling.md +141 -141
  224. package/skills/playwright-cli/SKILL.md +420 -420
  225. package/skills/playwright-cli/references/element-attributes.md +23 -23
  226. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  227. package/skills/playwright-cli/references/request-mocking.md +87 -87
  228. package/skills/playwright-cli/references/running-code.md +241 -241
  229. package/skills/playwright-cli/references/session-management.md +225 -225
  230. package/skills/playwright-cli/references/storage-state.md +275 -275
  231. package/skills/playwright-cli/references/test-generation.md +433 -433
  232. package/skills/playwright-cli/references/tracing.md +139 -139
  233. package/skills/playwright-cli/references/video-recording.md +143 -143
  234. package/skills/playwright-cli-case-generator/SKILL.md +74 -74
  235. package/skills/requesting-code-review/SKILL.md +101 -101
  236. package/skills/requesting-code-review/code-reviewer.md +168 -168
  237. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  238. package/skills/systematic-debugging/SKILL.md +296 -296
  239. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  240. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  241. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  242. package/skills/systematic-debugging/find-polluter.sh +63 -63
  243. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  244. package/skills/systematic-debugging/test-academic.md +14 -14
  245. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  246. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  247. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  248. package/skills/test-driven-development/SKILL.md +20 -20
  249. package/skills/using-git-worktrees/SKILL.md +215 -215
  250. package/skills/verification-before-completion/SKILL.md +154 -154
  251. package/skills/webapp-testing/SKILL.md +19 -19
  252. package/docs/agent-dag-recovery-playbook.md +0 -195
  253. package/docs/agent-dag-runner.md +0 -67
  254. package/docs/cursor-prompt-sidecar.md +0 -36
  255. package/docs/decisions/README.md +0 -18
  256. package/docs/design/README.md +0 -167
  257. package/docs/development-principles.md +0 -73
  258. package/docs/exec-plans/README.md +0 -6
  259. package/docs/exec-plans/active/README.md +0 -12
  260. package/docs/exec-plans/completed/README.md +0 -107
  261. package/docs/feature-workflow.md +0 -414
  262. package/docs/loop-agent-harness.md +0 -142
  263. package/docs/production-readiness.md +0 -96
  264. package/docs/progress/README.md +0 -80
  265. package/docs/reports/README.md +0 -159
  266. package/docs/verification-matrix.md +0 -70
  267. package/scripts/check-product-line-docs.sh +0 -29
  268. package/scripts/check-task-pool-root.sh +0 -32
  269. package/scripts/kb-graph-incremental-prepare.sh +0 -5
  270. package/scripts/kb-graph-materialize.sh +0 -4
  271. package/scripts/kb-graph-promote.sh +0 -4
  272. package/scripts/kb-query.sh +0 -5
@@ -37,17 +37,17 @@ export function parseLoopBenchmarkArgs(args) {
37
37
  return { json, markdown, outputPath };
38
38
  }
39
39
  export function printLoopBenchmarkUsage() {
40
- console.log(`usage: loop-benchmark [options]
41
-
42
- Deterministic loop-agent loop benchmark baseline (no live Pi/Cursor calls).
43
-
44
- Options:
45
- --json Emit JSON (default when no format flag is set)
46
- --markdown Emit Markdown report
47
- --output <path> Write Markdown report to a repo-relative or absolute path
48
- -h, --help Show this help
49
-
50
- Control groups: single-repair, 3-pass-convergence, 3-pass-convergence+quota.
40
+ console.log(`usage: loop-benchmark [options]
41
+
42
+ Deterministic loop-agent loop benchmark baseline (no live Pi/Cursor calls).
43
+
44
+ Options:
45
+ --json Emit JSON (default when no format flag is set)
46
+ --markdown Emit Markdown report
47
+ --output <path> Write Markdown report to a repo-relative or absolute path
48
+ -h, --help Show this help
49
+
50
+ Control groups: single-repair, 3-pass-convergence, 3-pass-convergence+quota.
51
51
  Recommendation never changes convergence.enabled default.`);
52
52
  }
53
53
  export async function runLoopBenchmark(repoRoot, rawArgs) {
@@ -106,22 +106,22 @@ export function parsePiReuseBenchmarkArgs(args) {
106
106
  };
107
107
  }
108
108
  export function printPiReuseBenchmarkUsage() {
109
- console.log(`usage: pi-reuse-benchmark [options]
110
-
111
- Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
112
-
113
- Options:
114
- --report <path> Benchmark report markdown (approval status)
115
- --approval <path> Explicit approval JSON artifact
116
- --off-executor <path> Baseline executor.jsonl (reuse off)
117
- --on-executor <path> Treatment executor.jsonl (reuse on)
118
- --off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
119
- --on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
120
- --json Emit JSON (default when no format flag is set)
121
- --markdown Emit Markdown summary
122
- -h, --help Show this help
123
-
124
- Recommendations: defer | maintain-opt-in | eligible-for-human-review
109
+ console.log(`usage: pi-reuse-benchmark [options]
110
+
111
+ Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
112
+
113
+ Options:
114
+ --report <path> Benchmark report markdown (approval status)
115
+ --approval <path> Explicit approval JSON artifact
116
+ --off-executor <path> Baseline executor.jsonl (reuse off)
117
+ --on-executor <path> Treatment executor.jsonl (reuse on)
118
+ --off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
119
+ --on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
120
+ --json Emit JSON (default when no format flag is set)
121
+ --markdown Emit Markdown summary
122
+ -h, --help Show this help
123
+
124
+ Recommendations: defer | maintain-opt-in | eligible-for-human-review
125
125
  Never changes CODE_AGENT_PI_REUSE_RUNTIME default (off).`);
126
126
  }
127
127
  function resolveRepoRelative(repoRoot, filePath) {
@@ -5,39 +5,61 @@
5
5
  function isRecord(value) {
6
6
  return typeof value === 'object' && value !== null;
7
7
  }
8
+ function unifyToolPayloadFields(event) {
9
+ const out = { ...event };
10
+ const args = out.args ?? out.input ?? out.toolInput;
11
+ if (args !== undefined) {
12
+ out.args = args;
13
+ if (out.input === undefined)
14
+ out.input = args;
15
+ }
16
+ const result = out.result ?? out.toolResult;
17
+ if (result !== undefined) {
18
+ out.result = result;
19
+ }
20
+ return out;
21
+ }
8
22
  /** Map SDK event shapes to the JSONL format emitted by `pi -p --mode json`. */
9
23
  export function normalizePiSessionEvent(event) {
10
24
  const type = typeof event.type === 'string' ? event.type : undefined;
11
25
  if (type === 'assistant_message' && isRecord(event.message)) {
12
- return {
26
+ return unifyToolPayloadFields({
13
27
  type: 'turn_end',
14
28
  message: event.message,
15
29
  usage: event.usage,
16
- };
30
+ });
17
31
  }
18
32
  if (type === 'tool_start' && typeof event.toolName === 'string') {
19
- return {
33
+ const args = event.args ?? event.input ?? event.toolInput;
34
+ return unifyToolPayloadFields({
20
35
  type: 'tool_execution_start',
21
36
  toolName: event.toolName,
22
37
  toolCallId: event.toolCallId,
23
- input: event.input,
24
- };
38
+ input: args,
39
+ args,
40
+ });
25
41
  }
26
42
  if (type === 'tool_end' && typeof event.toolName === 'string') {
27
- return {
43
+ const result = event.result ?? event.toolResult;
44
+ return unifyToolPayloadFields({
28
45
  type: 'tool_execution_end',
29
46
  toolName: event.toolName,
30
47
  toolCallId: event.toolCallId,
31
48
  isError: event.isError,
32
- };
49
+ ...(result !== undefined ? { result } : {}),
50
+ });
33
51
  }
34
- return event;
52
+ return unifyToolPayloadFields(event);
35
53
  }
36
54
  /** Serialize one SDK session event as a single JSONL line. */
37
- export function serializeSessionEvent(event) {
38
- return JSON.stringify(normalizePiSessionEvent(event));
55
+ export function serializeSessionEvent(event, options) {
56
+ const normalized = normalizePiSessionEvent(event);
57
+ if (typeof normalized.recordedAt !== 'string') {
58
+ normalized.recordedAt = options?.recordedAt ?? new Date().toISOString();
59
+ }
60
+ return JSON.stringify(normalized);
39
61
  }
40
62
  /** Serialize multiple events into newline-delimited JSON (no trailing newline). */
41
63
  export function serializeSessionEvents(events) {
42
- return events.map(serializeSessionEvent).join('\n');
64
+ return events.map((event) => serializeSessionEvent(event)).join('\n');
43
65
  }
@@ -29,7 +29,7 @@ export function resolveArtifactWriteDir(options) {
29
29
  }
30
30
  export function buildArtifactPathPrompt(writeDir) {
31
31
  if (!writeDir)
32
- return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
32
+ return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
33
33
  ${ARTIFACT_INSTRUCTIONS}`;
34
34
  return [
35
35
  `After changes, write the following files:`,
@@ -236,35 +236,35 @@ export function shouldInjectSubagentGuidance(step, mode) {
236
236
  return false;
237
237
  }
238
238
  /** Advisory guidance for `analyze-plan` mode. */
239
- export const SUBAGENT_GUIDANCE_STANDARD = `<subagent_guidance>
240
- You have access to the \`subagent\` tool for lightweight delegation within this step.
241
- Use it only for read-only tasks:
242
- - Parallel scout: dispatch multiple subagents to search/read different areas simultaneously
243
- - Chain: scout -> planner (one subagent scouts, another plans based on findings)
244
- - Reviewer: have a subagent review your analysis/plan before finalizing
245
- Do NOT use subagent for writing, editing, or executing commands.
246
- Subagent output is advisory only; always verify and incorporate findings into your own output.
247
- Do NOT treat subagent results as authoritative state or artifact sources.
239
+ export const SUBAGENT_GUIDANCE_STANDARD = `<subagent_guidance>
240
+ You have access to the \`subagent\` tool for lightweight delegation within this step.
241
+ Use it only for read-only tasks:
242
+ - Parallel scout: dispatch multiple subagents to search/read different areas simultaneously
243
+ - Chain: scout -> planner (one subagent scouts, another plans based on findings)
244
+ - Reviewer: have a subagent review your analysis/plan before finalizing
245
+ Do NOT use subagent for writing, editing, or executing commands.
246
+ Subagent output is advisory only; always verify and incorporate findings into your own output.
247
+ Do NOT treat subagent results as authoritative state or artifact sources.
248
248
  </subagent_guidance>`;
249
249
  /** Strong guidance for `full` mode — prescriptive when to delegate. */
250
- export const SUBAGENT_GUIDANCE_STRONG = `<subagent_guidance>
251
- You have access to the \`subagent\` tool for lightweight delegation within this step.
252
-
253
- You SHOULD delegate to subagent scouts when:
254
- - The task requires scanning 3+ directories or comparing implementations across modules
255
- - You would otherwise need 5+ sequential read/grep calls to gather context
256
- - A reviewer subagent can independently catch scope drift before you finalize your output
257
-
258
- Delegation saves context tokens and produces better results.
259
-
260
- Allowed patterns:
261
- - Parallel scout: dispatch 2-3 subagents simultaneously to cover different file trees
262
- - Chain: scout -> planner (one subagent scouts, another plans based on findings)
263
- - Reviewer: have a subagent review your analysis/plan before finalizing
264
-
265
- Do NOT use subagent for writing, editing, or executing commands.
266
- Subagent output is advisory only; always verify and incorporate findings into your own output.
267
- Do NOT treat subagent results as authoritative state or artifact sources.
250
+ export const SUBAGENT_GUIDANCE_STRONG = `<subagent_guidance>
251
+ You have access to the \`subagent\` tool for lightweight delegation within this step.
252
+
253
+ You SHOULD delegate to subagent scouts when:
254
+ - The task requires scanning 3+ directories or comparing implementations across modules
255
+ - You would otherwise need 5+ sequential read/grep calls to gather context
256
+ - A reviewer subagent can independently catch scope drift before you finalize your output
257
+
258
+ Delegation saves context tokens and produces better results.
259
+
260
+ Allowed patterns:
261
+ - Parallel scout: dispatch 2-3 subagents simultaneously to cover different file trees
262
+ - Chain: scout -> planner (one subagent scouts, another plans based on findings)
263
+ - Reviewer: have a subagent review your analysis/plan before finalizing
264
+
265
+ Do NOT use subagent for writing, editing, or executing commands.
266
+ Subagent output is advisory only; always verify and incorporate findings into your own output.
267
+ Do NOT treat subagent results as authoritative state or artifact sources.
268
268
  </subagent_guidance>`;
269
269
  /** Preserved for backward compatibility (alias of STANDARD). */
270
270
  export const SUBAGENT_GUIDANCE = SUBAGENT_GUIDANCE_STANDARD;
@@ -87,7 +87,11 @@ function resolveToolName(event) {
87
87
  return (event.toolName ?? "").toLowerCase();
88
88
  }
89
89
  function eventTimestamp(event) {
90
- return event.timestamp ?? event.at;
90
+ return event.timestamp ?? event.at ?? event.recordedAt;
91
+ }
92
+ function eventToolArgs(event) {
93
+ const args = event.args ?? event.toolInput ?? event.input;
94
+ return args && typeof args === "object" && !Array.isArray(args) ? args : {};
91
95
  }
92
96
  /**
93
97
  * Parse a tool input path to a repo-relative path.
@@ -150,7 +154,7 @@ export async function extractSpecEvidence(repoRoot, dagRunId, nodeId) {
150
154
  const type = event.type;
151
155
  const toolName = resolveToolName(event);
152
156
  const toolCallId = event.toolCallId;
153
- const toolInput = event.toolInput ?? event.input ?? {};
157
+ const toolInput = eventToolArgs(event);
154
158
  const ts = eventTimestamp(event);
155
159
  // Collect tool_execution_start events by toolCallId
156
160
  if (type === "tool_execution_start" && toolCallId) {
@@ -159,18 +163,21 @@ export async function extractSpecEvidence(repoRoot, dagRunId, nodeId) {
159
163
  }
160
164
  // Process tool_execution_end events
161
165
  if (type === "tool_execution_end") {
166
+ const start = toolCallId ? startMap.get(toolCallId) : undefined;
167
+ const pairedInput = start?.input ?? toolInput;
162
168
  // Successful read calls: must have matching start event and no error
163
169
  if (toolName === "read") {
164
170
  const isErrored = event.isError === true ||
165
171
  event.toolResult?.error != null ||
166
- event.toolResult?.ok === false;
172
+ event.toolResult?.ok === false ||
173
+ event.result?.error != null ||
174
+ event.result?.ok === false;
167
175
  if (isErrored)
168
176
  continue;
169
177
  let filePath;
170
178
  // Only paired read calls qualify: must have toolCallId, matching start
171
179
  // with same toolName, and start must be a "read" tool.
172
180
  if (toolCallId) {
173
- const start = startMap.get(toolCallId);
174
181
  if (start && start.toolName === "read") {
175
182
  const startInput = start.input ?? {};
176
183
  filePath = typeof startInput.path === "string" ? startInput.path : undefined;
@@ -188,18 +195,20 @@ export async function extractSpecEvidence(repoRoot, dagRunId, nodeId) {
188
195
  }
189
196
  // Search/scan tool calls (grep, find, ls, glob)
190
197
  if (["grep", "find", "ls", "glob"].includes(toolName)) {
191
- const query = typeof toolInput.pattern === "string"
192
- ? toolInput.pattern
193
- : typeof toolInput.query === "string"
194
- ? toolInput.query
195
- : typeof toolInput.path === "string"
196
- ? toolInput.path
198
+ const query = typeof pairedInput.pattern === "string"
199
+ ? pairedInput.pattern
200
+ : typeof pairedInput.query === "string"
201
+ ? pairedInput.query
202
+ : typeof pairedInput.path === "string"
203
+ ? pairedInput.path
197
204
  : toolName;
198
205
  searches.push({
199
206
  tool: toolName,
200
207
  query,
201
208
  timestamp: ts,
202
209
  });
210
+ if (toolCallId)
211
+ startMap.delete(toolCallId);
203
212
  }
204
213
  // Knowledge base connector calls
205
214
  if (isKnowledgeBaseTool(toolName)) {
@@ -1,46 +1,46 @@
1
- /** Fetch and artifact helpers (Observe UI R5). */
2
-
3
- export async function fetchJson(url) {
4
- try {
5
- const res = await fetch(url);
6
- if (!res.ok) return null;
7
- return await res.json();
8
- } catch {
9
- return null;
10
- }
11
- }
12
-
13
- /**
14
- * Read-only JSON fetch that preserves HTTP status so callers can distinguish
15
- * 404 (missing) from 409 (ambiguous task identity). Body is parsed best-effort.
16
- */
17
- export async function fetchJsonResult(url) {
18
- try {
19
- const res = await fetch(url);
20
- let body = null;
21
- try {
22
- body = await res.json();
23
- } catch {
24
- body = null;
25
- }
26
- return { ok: res.ok, status: res.status, body };
27
- } catch {
28
- return { ok: false, status: 0, body: null };
29
- }
30
- }
31
-
32
- export function artifactUrl(artifactPath) {
33
- return `/api/artifacts?path=${encodeURIComponent(artifactPath)}&tail=200`;
34
- }
35
-
36
- export function parseArtifactPreviewResponse(text) {
37
- const parsed = JSON.parse(text);
38
- if (!parsed || typeof parsed.content !== "string") {
39
- throw new Error("Artifact preview response is invalid");
40
- }
41
- return {
42
- content: parsed.content,
43
- truncated: parsed.truncated === true,
44
- };
45
- }
46
-
1
+ /** Fetch and artifact helpers (Observe UI R5). */
2
+
3
+ export async function fetchJson(url) {
4
+ try {
5
+ const res = await fetch(url);
6
+ if (!res.ok) return null;
7
+ return await res.json();
8
+ } catch {
9
+ return null;
10
+ }
11
+ }
12
+
13
+ /**
14
+ * Read-only JSON fetch that preserves HTTP status so callers can distinguish
15
+ * 404 (missing) from 409 (ambiguous task identity). Body is parsed best-effort.
16
+ */
17
+ export async function fetchJsonResult(url) {
18
+ try {
19
+ const res = await fetch(url);
20
+ let body = null;
21
+ try {
22
+ body = await res.json();
23
+ } catch {
24
+ body = null;
25
+ }
26
+ return { ok: res.ok, status: res.status, body };
27
+ } catch {
28
+ return { ok: false, status: 0, body: null };
29
+ }
30
+ }
31
+
32
+ export function artifactUrl(artifactPath) {
33
+ return `/api/artifacts?path=${encodeURIComponent(artifactPath)}&tail=200`;
34
+ }
35
+
36
+ export function parseArtifactPreviewResponse(text) {
37
+ const parsed = JSON.parse(text);
38
+ if (!parsed || typeof parsed.content !== "string") {
39
+ throw new Error("Artifact preview response is invalid");
40
+ }
41
+ return {
42
+ content: parsed.content,
43
+ truncated: parsed.truncated === true,
44
+ };
45
+ }
46
+