@tea-agent/loop-agent 0.42.0-next.9 → 0.42.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +111 -45
  2. package/dist/application/dag/run-dag.js +8 -2
  3. package/dist/application/evaluation/budget.js +19 -1
  4. package/dist/application/task-lifecycle/advance.js +26 -8
  5. package/dist/application/task-lifecycle/observe.js +43 -29
  6. package/dist/application/task-lifecycle/plan-transitions.js +5 -4
  7. package/dist/build-stamp.json +3 -3
  8. package/dist/cli/program.js +1 -1
  9. package/dist/commands/dag-rerun-task.js +2 -0
  10. package/dist/commands/task-source-prepare.js +3 -1
  11. package/dist/executors/dag-pi-executor.js +2584 -600
  12. package/dist/executors/pi-executor.js +22 -1
  13. package/dist/executors/pi-extension-resolver.js +14 -2
  14. package/dist/executors/pi-sdk-executor.js +140 -39
  15. package/dist/executors/shell-executor.js +135 -55
  16. package/dist/shared/dag-failure-category.js +6 -0
  17. package/dist/shared/frontend-execution-policy.js +23 -0
  18. package/dist/task/config-types.js +4 -0
  19. package/dist/task/contract/apply.js +36 -2
  20. package/dist/task/source-prepare/ledger-reconciliation.js +2 -2
  21. package/dist/task/source-prepare/ledger-review.js +6 -9
  22. package/dist/task/source-prepare/parse-intent.js +7 -0
  23. package/dist/task/source-prepare/semantic-intake.js +16 -26
  24. package/dist/task/source-prepare/source-fidelity-pi.js +28 -9
  25. package/dist/task/source-references.js +48 -23
  26. package/dist/worker/console/chat/assistant-content.js +23 -2
  27. package/dist/worker/console/chat/browser-policy.js +143 -0
  28. package/dist/worker/console/chat/browser-routes.js +148 -0
  29. package/dist/worker/console/chat/chat-event-store.js +4 -2
  30. package/dist/worker/console/chat/explore-tools.js +13 -0
  31. package/dist/worker/console/chat/pi-runtime.js +173 -94
  32. package/dist/worker/console/chat/resource-loader.js +4 -1
  33. package/dist/worker/console/chat/routes.js +208 -66
  34. package/dist/worker/console/chat/scm-routes.js +217 -0
  35. package/dist/worker/console/chat/scm-service.js +283 -0
  36. package/dist/worker/console/chat/scm-tools.js +111 -0
  37. package/dist/worker/console/chat/sdd-data-alignment.js +78 -11
  38. package/dist/worker/console/chat/session-catalog.js +32 -0
  39. package/dist/worker/console/chat/session-mode-view.js +48 -0
  40. package/dist/worker/console/chat/session-mode.js +218 -0
  41. package/dist/worker/console/chat/session-store.js +27 -7
  42. package/dist/worker/console/chat/shortcuts.js +6 -0
  43. package/dist/worker/console/chat/subagents/agent-tool.js +62 -0
  44. package/dist/worker/console/chat/subagents/explore-agent.js +169 -0
  45. package/dist/worker/console/chat/subagents/index.js +5 -0
  46. package/dist/worker/console/chat/subagents/orchestrator.js +273 -0
  47. package/dist/worker/console/chat/subagents/tool-policy.js +53 -0
  48. package/dist/worker/console/chat/subagents/types.js +13 -0
  49. package/dist/worker/console/chat/terminal-routes.js +216 -0
  50. package/dist/worker/console/chat/terminal-sessions.js +284 -0
  51. package/dist/worker/console/chat/terminal-tools.js +199 -0
  52. package/dist/worker/console/chat/tool-preview.js +21 -0
  53. package/dist/worker/console/chat/tools.js +13 -1
  54. package/dist/worker/console/chat/turn-process.js +1 -0
  55. package/dist/worker/console/interview/tools.js +1 -0
  56. package/dist/worker/console/server.js +2 -28
  57. package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-C6n9_m0P.js → abnfDiagram-N423BO3Z-C9eI7nEo.js} +1 -1
  58. package/dist/worker/console/static/assets/{arc-DQh-IfZ1.js → arc-VJWsxWhB.js} +1 -1
  59. package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-54NnrwUC.js → architectureDiagram-T3A2C74G-BGcyiMSo.js} +1 -1
  60. package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-pivRALGK.js → blockDiagram-VBNYF7ZC-BScnFwyi.js} +1 -1
  61. package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-BR7OV2NJ.js → c4Diagram-5PPSVZJV-Bl9_BsLi.js} +1 -1
  62. package/dist/worker/console/static/assets/channel-B6sQYuOE.js +1 -0
  63. package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-B1Aq6BcK.js → chunk-2GRJ4B5K-F1uKxiUt.js} +1 -1
  64. package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-CbdU0rjo.js → chunk-2Q5K7J3B-rSRMynVu.js} +1 -1
  65. package/dist/worker/console/static/assets/{chunk-5RXB4S5H-Dqi2GJeD.js → chunk-5RXB4S5H-DAsJm1WD.js} +1 -1
  66. package/dist/worker/console/static/assets/{chunk-5VM5RSS4-BYn1Hu4R.js → chunk-5VM5RSS4-DQIdVI0a.js} +1 -1
  67. package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-DRLjDL0k.js → chunk-6Q2QTUOP-4A_8rJ-Q.js} +1 -1
  68. package/dist/worker/console/static/assets/{chunk-GF5L2VYU-CY9Xx2jV.js → chunk-GF5L2VYU-BHnT-vjh.js} +1 -1
  69. package/dist/worker/console/static/assets/{chunk-JWPE2WC7-BNCZs7_z.js → chunk-JWPE2WC7-Rq7QKBkn.js} +1 -1
  70. package/dist/worker/console/static/assets/{chunk-KBJHAD2P-DZ8AStLL.js → chunk-KBJHAD2P-BXh-AI2u.js} +1 -1
  71. package/dist/worker/console/static/assets/{chunk-RYQCIY6F-DsxjYrzz.js → chunk-RYQCIY6F-BblcLy8m.js} +1 -1
  72. package/dist/worker/console/static/assets/{chunk-XXDRQBXY-Ddx1KC1l.js → chunk-XXDRQBXY-zHZmVnB6.js} +1 -1
  73. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-Z6s5QoeI.js +1 -0
  74. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-Z6s5QoeI.js +1 -0
  75. package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-W9TveCnK.js → cose-bilkent-JH36ORCC-Cmwz0Wi4.js} +1 -1
  76. package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-l9j_ztZH.js → cynefin-VYW2F7L2-Dq76MHQU.js} +1 -1
  77. package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-BMJMKi4G.js → cynefinDiagram-MW4NZA55-zPahV1fM.js} +1 -1
  78. package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-Bb4mG9pH.js → dagre-VZM6K2ZE-DQJ3xPc_.js} +1 -1
  79. package/dist/worker/console/static/assets/{diagram-7IWD3JNH-BMgL_Qv0.js → diagram-7IWD3JNH-BwX7u6fv.js} +1 -1
  80. package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-Bvo2T4OQ.js → diagram-B4RE2ZJO-CodnDCco.js} +1 -1
  81. package/dist/worker/console/static/assets/{diagram-LBJQPF4R-_5kWRN9c.js → diagram-LBJQPF4R-gAjtWkxG.js} +1 -1
  82. package/dist/worker/console/static/assets/{diagram-Q27KOJAE-DqdMrltM.js → diagram-Q27KOJAE-CGCqz0cS.js} +1 -1
  83. package/dist/worker/console/static/assets/{diagram-UB23O5K3-CDDYsNkp.js → diagram-UB23O5K3-RHwYEywd.js} +1 -1
  84. package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-c4nfnZUV.js → ebnfDiagram-BXEA7PRR-B6seb_p_.js} +1 -1
  85. package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-DnkUcNMs.js → erDiagram-JOGREHBK-CmwCKYe8.js} +1 -1
  86. package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-Dz0KGLZE.js → flowDiagram-UKHOOZJN-BfNsaZ0V.js} +1 -1
  87. package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-Cj1t1uka.js → ganttDiagram-PKOTCBZU-zmYgi_Z3.js} +1 -1
  88. package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-tp2FrBHd.js → gitGraphDiagram-DS77QQ5N-BDEzzSKF.js} +1 -1
  89. package/dist/worker/console/static/assets/index-DgenUAfc.js +468 -0
  90. package/dist/worker/console/static/assets/index-aF4u-Y__.css +1 -0
  91. package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-DQwJS8DH.js → infoDiagram-6WML65LV-DeSz8l42.js} +1 -1
  92. package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-eimCQWD4.js → ishikawaDiagram-WSZJBQD7-B1LV29aH.js} +1 -1
  93. package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-Dd9Gbwgv.js → journeyDiagram-NVQOT4AX-CCw4W2oU.js} +1 -1
  94. package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-Qve3cyft.js → kanban-definition-27J2QSJJ-BozaMoRJ.js} +1 -1
  95. package/dist/worker/console/static/assets/{linear-UXKSe36Z.js → linear-G1JR40SX.js} +1 -1
  96. package/dist/worker/console/static/assets/{mermaid.core-CWhj4JXN.js → mermaid.core-CCgPF0oA.js} +5 -5
  97. package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-yTcwf4SJ.js → mindmap-definition-FAOFIHXS-CEqnCHnD.js} +1 -1
  98. package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-_7qToaZP.js → pegDiagram-VL7TDLO6-DpmQPKuI.js} +1 -1
  99. package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-CXmrKmDm.js → pieDiagram-7S7Q4E2Y-BP501AxR.js} +1 -1
  100. package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-hMmrJyaS.js → quadrantDiagram-CIZ2JOQS-D1dBhl_A.js} +1 -1
  101. package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-BXw8tCBe.js → railroadDiagram-AXF67PYL-DgZDS5pY.js} +1 -1
  102. package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-BmjgnLZl.js → requirementDiagram-LRYGKXZP-BzB3pyIq.js} +1 -1
  103. package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-Bj77SPpN.js → sankeyDiagram-W5VNT64P-B9ZprFFP.js} +1 -1
  104. package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-B2FDVysL.js → sequenceDiagram-SI44F4Z6-CC2-_XrA.js} +1 -1
  105. package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-UChNY9ZZ.js → sizeCapture-X5ZJPWSS-UVtAUqgB.js} +1 -1
  106. package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-wbAEkbd7.js → stateDiagram-OKZ733FA-CWcS5JYG.js} +1 -1
  107. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-BooX8u1Q.js +1 -0
  108. package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-DCEkU8HR.js → swimlanes-SLNWSIFB-C4p-dn06.js} +2 -2
  109. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-D64_eqqE.js +8 -0
  110. package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-BGp7H06U.js → timeline-definition-Z64GVDOM-Dnz52OHX.js} +1 -1
  111. package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-DsCIQuWR.js → vennDiagram-T6HMQDX7-CzTE0Nfu.js} +1 -1
  112. package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-cRZLP4Xe.js → wardleyDiagram-T6FBY63Y-NAR9cjQQ.js} +1 -1
  113. package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-CP0xLR9Y.js → xychartDiagram-ELKLHX3M-DVMk8413.js} +1 -1
  114. package/dist/worker/console/static/index.html +2 -2
  115. package/dist/worker/console/static-src/chat-view-types.js +1 -1
  116. package/dist/worker/console/static-src/operator-chat/chat-link.js +94 -0
  117. package/dist/worker/console/static-src/operator-chat/chat-sse-events.js +142 -7
  118. package/dist/worker/console/static-src/operator-chat/pending-user-message.js +44 -0
  119. package/dist/worker/console/static-src/operator-chat/process-label.js +41 -0
  120. package/dist/worker/console/static-src/operator-chat/tools-catalog.js +21 -2
  121. package/dist/worker/console/static-src/operator-chat/turn-group-equality.js +13 -0
  122. package/dist/worker/console/static-src/operator-chat/turn-stream-controller.js +2 -0
  123. package/dist/worker/console/static-src/operator-chat/use-searchable-hidden.js +51 -0
  124. package/dist/worker/console/static-src/operator-chat/useChatSessions.js +33 -18
  125. package/dist/worker/console/static-src/operator-chat/useChatStream.js +48 -16
  126. package/dist/worker/console/static-src/operator-chat/useChatThread.js +170 -7
  127. package/dist/worker/console/static-src/operator-chat/useComposer.js +13 -7
  128. package/dist/worker/console/static-src/operator-chat/user-turn-anchor-equality.js +21 -0
  129. package/dist/worker/console/static-src/shell/workspace-route.js +4 -0
  130. package/dist/worker/console/workspace-context.js +15 -1
  131. package/dist/worker/observe/node-transparency.js +81 -72
  132. package/dist/worker/observe/routes.js +20 -1
  133. package/dist/worker/observe/static/constants.js +22 -22
  134. package/dist/worker/observe/static/dag-context-reason-labels.js +19 -0
  135. package/dist/worker/observe/static/dag-history-labels.js +1 -0
  136. package/dist/worker/observe/static/dag-inspector-humanize.d.ts +16 -0
  137. package/dist/worker/observe/static/dag-inspector-humanize.js +342 -0
  138. package/dist/worker/observe/static/dag-node-purpose.js +10 -10
  139. package/dist/worker/observe/static/dom.js +20 -1
  140. package/dist/worker/observe/static/format-pool.d.ts +2 -0
  141. package/dist/worker/observe/static/format-pool.js +6 -0
  142. package/dist/worker/observe/static/format.js +7 -0
  143. package/dist/worker/observe/static/index.html +4 -4
  144. package/dist/worker/observe/static/inspect-workspace.js +34 -7
  145. package/dist/worker/observe/static/inspector-submission.js +32 -0
  146. package/dist/worker/observe/static/kpi.js +1 -0
  147. package/dist/worker/observe/static/prompt-restart-candidates.js +4 -2
  148. package/dist/worker/observe/static/relations.js +7 -5
  149. package/dist/worker/observe/static/router.js +13 -0
  150. package/dist/worker/observe/static/run-processing.js +2 -0
  151. package/dist/worker/observe/static/shell-chrome.js +36 -3
  152. package/dist/worker/observe/static/state.js +35 -2
  153. package/dist/worker/observe/static/styles.css +431 -39
  154. package/dist/worker/observe/static/task-failure-labels.d.ts +4 -0
  155. package/dist/worker/observe/static/task-failure-labels.js +67 -0
  156. package/dist/worker/observe/static/task-history.js +12 -0
  157. package/dist/worker/observe/static/views/batch.js +6 -13
  158. package/dist/worker/observe/static/views/dag-graph.js +51 -4
  159. package/dist/worker/observe/static/views/dag-inspector.js +1069 -426
  160. package/dist/worker/observe/static/views/dag-trajectory.js +3 -0
  161. package/dist/worker/observe/static/views/dag.d.ts +6 -0
  162. package/dist/worker/observe/static/views/dag.js +64 -22
  163. package/dist/worker/observe/static/views/dags.js +2 -0
  164. package/dist/worker/observe/static/views/dashboard.js +21 -12
  165. package/dist/worker/observe/static/views/failures.js +21 -11
  166. package/dist/worker/observe/static/views/feature.js +11 -29
  167. package/dist/worker/observe/static/views/pool.js +37 -28
  168. package/dist/worker/observe/static/views/run.js +48 -5
  169. package/dist/worker/observe/static/views/session-timeline.js +194 -243
  170. package/dist/worker/observe/static/views/task.js +81 -62
  171. package/dist/workflows/dag/backend-test-plan-protocol.js +104 -0
  172. package/dist/workflows/dag/budget-enforcement.js +53 -3
  173. package/dist/workflows/dag/dag-retry-schema.js +3 -0
  174. package/dist/workflows/dag/frontend-capacity.js +9 -0
  175. package/dist/workflows/dag/frontend-contract-facts.js +130 -0
  176. package/dist/workflows/dag/frontend-design-policy.js +103 -17
  177. package/dist/workflows/dag/frontend-durable-tools.js +193 -0
  178. package/dist/workflows/dag/frontend-execution-groups.js +24 -0
  179. package/dist/workflows/dag/frontend-implementation-contract.js +423 -41
  180. package/dist/workflows/dag/frontend-input-projection.js +76 -0
  181. package/dist/workflows/dag/frontend-plan-completeness.js +186 -0
  182. package/dist/workflows/dag/frontend-plan-recovery-policy.js +18 -0
  183. package/dist/workflows/dag/frontend-plan-render.js +21 -3
  184. package/dist/workflows/dag/frontend-prewrite-gate.js +1 -1
  185. package/dist/workflows/dag/frontend-recovery-controller.js +32 -8
  186. package/dist/workflows/dag/frontend-recovery-lineage.js +13 -0
  187. package/dist/workflows/dag/frontend-recovery-plan.js +4 -1
  188. package/dist/workflows/dag/frontend-recovery-run.js +146 -16
  189. package/dist/workflows/dag/frontend-review-scopes.js +139 -0
  190. package/dist/workflows/dag/frontend-risk.js +2 -0
  191. package/dist/workflows/dag/frontend-session-budget.js +249 -0
  192. package/dist/workflows/dag/frontend-shadow-dual-write.js +37 -3
  193. package/dist/workflows/dag/frontend-shape-facts.js +12 -2
  194. package/dist/workflows/dag/frontend-shape.js +46 -7
  195. package/dist/workflows/dag/frontend-test-execution-evidence.js +49 -0
  196. package/dist/workflows/dag/frontend-typed-event-store.js +14 -0
  197. package/dist/workflows/dag/frontend-verification-trace.js +38 -4
  198. package/dist/workflows/dag/frontend-writer-admission.js +2 -2
  199. package/dist/workflows/dag/init-hybrid.js +70 -35
  200. package/dist/workflows/dag/node-execution.js +225 -226
  201. package/dist/workflows/dag/prompt.js +4 -0
  202. package/dist/workflows/dag/recovery-lease.js +106 -16
  203. package/dist/workflows/dag/rerun-feedback.js +1 -0
  204. package/dist/workflows/dag/rerun-plan.js +96 -4
  205. package/dist/workflows/dag/rerun-run.js +28 -6
  206. package/dist/workflows/dag/rerun-task.js +221 -18
  207. package/dist/workflows/dag/retry-policy.js +27 -10
  208. package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
  209. package/dist/workflows/dag/runner.js +246 -19
  210. package/dist/workflows/dag/structured-output-repair.js +4 -1
  211. package/dist/workflows/dag/types.js +10 -3
  212. package/dist/workflows/dag/validate.js +10 -8
  213. package/dist/workflows/dag/workspace-checkpoint.js +66 -0
  214. package/docs/operations/README.md +2 -0
  215. package/docs/templates/agent-dag.schema.json +2 -2
  216. package/docs/templates/backend-test-dag.json +6 -4
  217. package/docs/templates/frontend-design-contract.md +4 -4
  218. package/docs/templates/frontend-implementation-contract.schema.json +68 -4
  219. package/docs/templates/frontend-implementation-dag.json +5 -5
  220. package/harness.json +1 -1
  221. package/package.json +8 -6
  222. package/skills/frontend-contract/SKILL.md +2 -1
  223. package/skills/frontend-contract/references/contract-protocol.md +19 -3
  224. package/skills/frontend-design-review/SKILL.md +12 -11
  225. package/skills/frontend-plan/SKILL.md +22 -2
  226. package/skills/frontend-plan/references/decision-contract.md +114 -5
  227. package/skills/frontend-plan/references/design-decisions.md +32 -0
  228. package/skills/frontend-review/SKILL.md +10 -11
  229. package/skills/frontend-scout/references/scout-evidence.md +4 -0
  230. package/dist/worker/console/static/assets/channel-DsZxrgqe.js +0 -1
  231. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-CuToPeGV.js +0 -1
  232. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-CuToPeGV.js +0 -1
  233. package/dist/worker/console/static/assets/index-D9Sc0f0y.js +0 -449
  234. package/dist/worker/console/static/assets/index-DDQc5a50.css +0 -1
  235. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-yPoT19ft.js +0 -1
  236. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-VWdarchG.js +0 -8
@@ -1,3 +1,13 @@
1
+ import { collectFrontendPlanMissingFacts, collectFrontendPlanPhaseMissingFacts, committedFactFromPlanRecord, planFactStringList, planFactScopeIntersects } from "../workflows/dag/frontend-plan-completeness.js";
2
+ export { collectFrontendPlanMissingFacts, collectFrontendPlanPhaseMissingFacts } from "../workflows/dag/frontend-plan-completeness.js";
3
+ import { classifyFrontendPlanRecovery } from "../workflows/dag/frontend-plan-recovery-policy.js";
4
+ import { collectFrontendExecutionGroups, frontendExecutionSchema } from "../workflows/dag/frontend-execution-groups.js";
5
+ import { FRONTEND_SCOPE_TARGET_BYTES, packFrontendInputUnits, parseFrontendInputBlock, projectFrontendContractPrompt, projectFrontendInputScope } from "../workflows/dag/frontend-input-projection.js";
6
+ import { createDurableFrontendTools } from "../workflows/dag/frontend-durable-tools.js";
7
+ import { sha256OfCanonicalJson } from "../task/contract/hash.js";
8
+ import { z } from "zod";
9
+ import { createFrontendReviewScopeProtocol, loadFrontendReviewScopes } from "../workflows/dag/frontend-review-scopes.js";
10
+ import { observeFrontendSession } from "../workflows/dag/frontend-session-budget.js";
1
11
  import path from "node:path";
2
12
  import { createHash, randomUUID } from "node:crypto";
3
13
  import { readFile, stat } from "node:fs/promises";
@@ -12,34 +22,236 @@ import { createPiReadBudgetCustomTools, } from "./pi-read-budget-policy.js";
12
22
  import { cleanupPlaywrightCliDefaultSession, createPlaywrightCliTool, PI_COMMAND_CAPABILITY_REGISTRY, resolveCaseIdFromWriteSet, resolveEvidenceDirFromWriteSet, } from "./pi-playwright-cli-tool.js";
13
23
  import { dagCommandPolicyAllows, resolveDagCommandPolicy, } from "../workflows/dag/types.js";
14
24
  import { parseLedgerJson } from "../task/source-prepare/ledger.js";
25
+ import { splitDocumentIntoFragments, sha256Text, } from "../task/source-prepare/fragment-inventory.js";
15
26
  import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation.js";
16
27
  import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
17
28
  import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
18
29
  import { pathMatchesPattern } from "../shared/git-progress.js";
19
- import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
30
+ import { readTypedEventStoreFromJsonl, typedEventPayloadSha256, } from "../workflows/dag/frontend-typed-event-store.js";
31
+ import { collectCanonicalStateFlowNames, frontendEvidenceExpectationSchema, resolveFrontendContractRequirements, validateFrontendRequiredDeliverables, } from "../workflows/dag/frontend-contract-facts.js";
32
+ import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
20
33
  import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestCollectionRepairOutcomeRecoveryCandidate, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
21
34
  import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
35
+ import { assessBackendTestPlanProtocol } from "../workflows/dag/backend-test-plan-protocol.js";
22
36
  import { frontendTestLayoutFromSpec } from "../workflows/dag/frontend-test-layout.js";
23
37
  import { redactSecrets, truncateUtf8Preview } from "../shared/preview.js";
24
38
  import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.js";
25
39
  /**
26
- * Writer classification for a length-stopped thinking-only attempt: the model
27
- * exhausted its output budget thinking but never issued a write/edit tool call
28
- * and produced zero attributed diff. This is a terminal, non-retryable
29
- * diagnosis (the same prompt + model will hit the same budget wall); the
30
- * recommendation is a model switch plus a fresh run. It must NOT mask a
31
- * recoverable partial-write-set (incomplete-write-set) upgrade.
40
+ * Legacy diagnostic label retained for artifact compatibility. New executions
41
+ * classify this signal as output-limit so the node can retry incrementally.
32
42
  */
33
43
  export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
34
44
  /**
35
- * Planner classification mirroring writer-thinking-exhausted: a read-only
36
- * planning session stopped on length, observed thinking, and committed zero
37
- * typed facts with no assistant text. By the time this survives the segmented
38
- * ladder the scope has already been degraded, so the durable fix is a
39
- * thinking-capped or non-thinking model for the tier — not another replay of
40
- * the same full-scope prompt.
45
+ * Legacy diagnostic label retained for artifact compatibility. New executions
46
+ * classify this signal as output-limit and preserve committed typed facts.
41
47
  */
42
48
  export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
49
+ /**
50
+ * Parallel coverage shards namespace their verification target ids with
51
+ * `VT-SHARD-<shard>-` so the reducer can detect cross-shard conflicts. Once
52
+ * every shard's facts are known the namespace must be restored: the canonical
53
+ * contract has to carry the exact ids the task source froze (a surviving
54
+ * `VT-SHARD-N-…` id is a contract-requirement gap that design review
55
+ * rejects). Stripping happens per shard before adoption — after the strip,
56
+ * the existing identity-conflict check catches genuine cross-shard
57
+ * duplicate base ids and fails the merge closed.
58
+ */
59
+ /**
60
+ * Deterministic pre-reduce for parallel coverage shard facts. Shards partition
61
+ * requirements, but a frozen verification target can legitimately be derived
62
+ * by several shards (one VT covers multiple ACs across shard boundaries), so
63
+ * same-id verification targets are MERGED: requirementIds and uiStates union,
64
+ * while divergent file/commandId is a real conflict. Everything else passes
65
+ * through unchanged.
66
+ */
67
+ export function reduceParallelCoverageShardRecords(records) {
68
+ // Resolve replacements inside each source shard before comparing shards.
69
+ // A correction is an event-log operation, not a second independent target;
70
+ // retaining both records makes a valid same-shard correction look like a
71
+ // cross-shard conflict during the later reduction.
72
+ const byShard = new Map();
73
+ for (const record of records) {
74
+ const shard = byShard.get(record.attemptId) ?? [];
75
+ shard.push(record);
76
+ byShard.set(record.attemptId, shard);
77
+ }
78
+ const effectiveRecords = [];
79
+ for (const shardRecords of byShard.values()) {
80
+ const effective = [];
81
+ for (const record of shardRecords) {
82
+ const fact = record.fact;
83
+ const entry = fact?.entry;
84
+ const kind = typeof fact?.kind === "string" ? fact.kind : "";
85
+ const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
86
+ typeof entry?.id === "string"
87
+ ? `${kind}:${entry.id}`
88
+ : undefined;
89
+ if (!identity) {
90
+ effective.push(record);
91
+ continue;
92
+ }
93
+ const replaces = typeof fact?.replaces === "string" ? `${kind}:${fact.replaces}` : undefined;
94
+ if (replaces) {
95
+ for (let index = effective.length - 1; index >= 0; index -= 1) {
96
+ const prior = effective[index];
97
+ const priorFact = prior.fact;
98
+ const priorEntry = priorFact?.entry;
99
+ const priorIdentity = typeof priorFact?.kind === "string" &&
100
+ typeof priorEntry?.id === "string"
101
+ ? `${priorFact.kind}:${priorEntry.id}`
102
+ : undefined;
103
+ if (priorIdentity === replaces)
104
+ effective.splice(index, 1);
105
+ }
106
+ }
107
+ // Keep the source record immutable. The replacement is represented by
108
+ // the latest event and its own payload hash.
109
+ effective.push(record);
110
+ }
111
+ effectiveRecords.push(...effective);
112
+ }
113
+ const verificationTargets = new Map();
114
+ const merged = [];
115
+ for (const record of effectiveRecords) {
116
+ const fact = record.fact;
117
+ if (!fact || typeof fact !== "object")
118
+ continue;
119
+ if (fact.kind !== "plan-verification-target") {
120
+ merged.push(record);
121
+ continue;
122
+ }
123
+ const entry = fact.entry;
124
+ if (!entry || typeof entry.id !== "string")
125
+ continue;
126
+ const existing = verificationTargets.get(entry.id);
127
+ if (!existing) {
128
+ verificationTargets.set(entry.id, {
129
+ record, entry: { ...entry },
130
+ sources: [`${record.attemptId}:${record.eventId}:${record.payloadSha256}`],
131
+ });
132
+ continue;
133
+ }
134
+ // Never mutate the entry held by the source record. The reducer creates
135
+ // a new merged entry and recomputes the payload hash for that derived
136
+ // fact, leaving source-event replay integrity intact.
137
+ const baseEntry = { ...existing.entry };
138
+ for (const field of ["commandId", "commandLabel", "file", "scope"]) {
139
+ if (entry[field] !== baseEntry[field]) {
140
+ throw new Error(`frontend plan coverage shard reduce conflict: verification target ${entry.id} has divergent ${field} "${String(entry[field])}" vs "${String(baseEntry[field])}"`);
141
+ }
142
+ }
143
+ const unionSorted = (a, b) => {
144
+ const left = Array.isArray(a) ? a : [];
145
+ const right = Array.isArray(b) ? b : [];
146
+ return [...new Set([...left, ...right])].sort();
147
+ };
148
+ baseEntry.requirementIds = unionSorted(baseEntry.requirementIds, entry.requirementIds);
149
+ baseEntry.uiStates = unionSorted(baseEntry.uiStates, entry.uiStates);
150
+ const mergedFact = { ...fact, entry: { ...baseEntry } };
151
+ existing.entry = baseEntry;
152
+ existing.sources.push(`${record.attemptId}:${record.eventId}:${record.payloadSha256}`);
153
+ existing.record = {
154
+ ...existing.record,
155
+ fact: mergedFact,
156
+ payloadSha256: typedEventPayloadSha256(mergedFact),
157
+ };
158
+ }
159
+ merged.push(...[...verificationTargets.values()].map((item) => {
160
+ // The reducer owns every VT revision, including a one-shard partial
161
+ // result. Content-addressed revisions replay idempotently, while an
162
+ // expanded/replaced source set appends an explicit replacement rather
163
+ // than masquerading as a changed source event or deleting audit facts.
164
+ const fact = {
165
+ ...item.record.fact,
166
+ replaces: item.entry.id,
167
+ entry: {
168
+ ...item.entry,
169
+ requirementIds: [...new Set(planFactStringList(item.entry.requirementIds))].sort(),
170
+ uiStates: [...new Set(planFactStringList(item.entry.uiStates))].sort(),
171
+ },
172
+ };
173
+ const payloadSha256 = typedEventPayloadSha256(fact);
174
+ const revisionId = createHash("sha256")
175
+ .update(JSON.stringify({ sources: [...new Set(item.sources)].sort(), payloadSha256 }))
176
+ .digest("hex");
177
+ return {
178
+ ...item.record,
179
+ attemptId: "frontend-plan-coverage-reducer",
180
+ eventId: `coverage-reduced:${revisionId}`,
181
+ requestId: `coverage-reduced:${revisionId}`,
182
+ fact, payloadSha256,
183
+ };
184
+ }));
185
+ return merged;
186
+ }
187
+ export function normalizeParallelCoverageShardRecords(records, shardNumber) {
188
+ const prefix = `VT-SHARD-${shardNumber}-`;
189
+ // Models sometimes drop the `VT-` stem when applying the shard namespace
190
+ // (observed: frozen `VT-X` became `VT-SHARD-2-X`), so restoring the
191
+ // canonical id requires re-adding the stem after the strip.
192
+ const stripId = (id) => {
193
+ if (typeof id !== "string" || !id.startsWith(prefix))
194
+ return id;
195
+ const base = id.slice(prefix.length);
196
+ return base.startsWith("VT-") ? base : `VT-${base}`;
197
+ };
198
+ const stripIdList = (ids) => Array.isArray(ids) ? ids.map((id) => stripId(id)) : ids;
199
+ return records.map((record) => {
200
+ const fact = record.fact;
201
+ if (!fact || typeof fact !== "object")
202
+ return record;
203
+ const kind = fact.kind;
204
+ const entry = fact.entry;
205
+ if (!entry || typeof entry !== "object")
206
+ return record;
207
+ let rewritten;
208
+ const replaces = kind === "plan-verification-target" ? stripId(fact.replaces) : fact.replaces;
209
+ if (kind === "plan-verification-target") {
210
+ const id = stripId(entry.id);
211
+ if (id !== entry.id || replaces !== fact.replaces)
212
+ rewritten = { ...entry, id };
213
+ }
214
+ else if (kind === "plan-requirement") {
215
+ const verificationTargetIds = stripIdList(entry.verificationTargetIds);
216
+ if (verificationTargetIds !== entry.verificationTargetIds) {
217
+ rewritten = { ...entry, verificationTargetIds };
218
+ }
219
+ }
220
+ else if (kind === "state-flow") {
221
+ const rewriteBoundTargets = (item) => {
222
+ if (!item || typeof item !== "object")
223
+ return item;
224
+ return {
225
+ ...item,
226
+ verificationTargetIds: stripIdList(item.verificationTargetIds),
227
+ };
228
+ };
229
+ const uiStates = Array.isArray(entry.uiStates)
230
+ ? entry.uiStates.map(rewriteBoundTargets)
231
+ : entry.uiStates;
232
+ const interactions = Array.isArray(entry.interactions)
233
+ ? entry.interactions.map(rewriteBoundTargets)
234
+ : entry.interactions;
235
+ if (uiStates !== entry.uiStates || interactions !== entry.interactions) {
236
+ rewritten = { ...entry, uiStates, interactions };
237
+ }
238
+ }
239
+ return rewritten
240
+ ? {
241
+ ...record,
242
+ fact: { ...fact, ...(replaces !== undefined ? { replaces } : {}), entry: rewritten },
243
+ // Keep the integrity hash consistent with the rewritten
244
+ // payload, or the adoption replay check reports the
245
+ // normalized record as a tampered source event.
246
+ payloadSha256: typedEventPayloadSha256({
247
+ ...fact,
248
+ ...(replaces !== undefined ? { replaces } : {}),
249
+ entry: rewritten,
250
+ }),
251
+ }
252
+ : record;
253
+ });
254
+ }
43
255
  export function isPlannerThinkingExhausted(result, committedAnyFacts) {
44
256
  if (result.ok)
45
257
  return false;
@@ -250,6 +462,9 @@ export const FRONTEND_CONTRACT_RECORD_TOOL_NAMES = [
250
462
  "record_handoff_intent",
251
463
  "record_open_question",
252
464
  "record_split_proposal",
465
+ "record_ui_state",
466
+ "record_required_deliverables",
467
+ "complete_contract_scope",
253
468
  ];
254
469
  export const FRONTEND_CONTRACT_TERMINAL_TOOL_NAMES = [
255
470
  "finalize_contract",
@@ -263,12 +478,15 @@ export const FRONTEND_SCOUT_EVIDENCE_TOOL_NAMES = [
263
478
  export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
264
479
  "record_route_selection",
265
480
  "record_component_choice",
481
+ "record_state_registry",
266
482
  "record_state_flow",
267
483
  "record_data_flow",
268
484
  "record_mock_api",
485
+ "record_mock_endpoint",
269
486
  "record_design_deviation",
270
487
  "record_dependency",
271
488
  "record_plan_requirement",
489
+ "record_plan_group_coverage",
272
490
  "record_plan_verification_target",
273
491
  "record_plan_evidence_gap",
274
492
  ];
@@ -343,6 +561,8 @@ export function resolveDagPiToolNames(task) {
343
561
  if (isFrontendReviewTypedTerminalNode(task)) {
344
562
  return [
345
563
  ...DAG_PI_READONLY_TOOLS,
564
+ "complete_review_scope",
565
+ "record_review_finding",
346
566
  "approve_review",
347
567
  "request_review_changes",
348
568
  ];
@@ -350,6 +570,8 @@ export function resolveDagPiToolNames(task) {
350
570
  if (isFrontendDesignTypedTerminalNode(task)) {
351
571
  return [
352
572
  ...DAG_PI_READONLY_TOOLS,
573
+ "complete_review_scope",
574
+ "record_design_finding",
353
575
  "approve_design",
354
576
  "request_design_changes",
355
577
  ];
@@ -375,6 +597,7 @@ export function resolveDagPiToolNames(task) {
375
597
  ...FRONTEND_PLAN_RECORD_TOOL_NAMES,
376
598
  ...FRONTEND_PLAN_TERMINAL_TOOL_NAMES,
377
599
  ...FRONTEND_PLAN_ADOPT_TOOL_NAMES,
600
+ "read_plan_facts",
378
601
  ];
379
602
  }
380
603
  if ((task.readSet?.length ?? 0) > 0) {
@@ -554,21 +777,35 @@ export async function createFrontendReviewTerminalTools(input) {
554
777
  ]);
555
778
  const { approveReviewFactSchema, readCommittedEvents, requestReviewChangesFactSchema, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
556
779
  const { adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
557
- const store = input.store;
780
+ let store = input.store;
558
781
  const attemptId = input.attemptId;
559
782
  const findingSchema = Type.Object({
560
- severity: Type.String({
561
- description: "Critical | Important | Minor | Info",
562
- }),
563
- file: Type.Optional(Type.String({})),
564
- line: Type.Optional(Type.Number({})),
565
- issue: Type.String({}),
566
- requiredChange: Type.Optional(Type.String({})),
783
+ severity: Type.Enum({ Critical: "Critical", Important: "Important", Minor: "Minor", Info: "Info" }),
784
+ file: Type.Optional(Type.String({ minLength: 1 })),
785
+ line: Type.Optional(Type.Integer({ minimum: 1 })),
786
+ issue: Type.String({ minLength: 1 }),
787
+ requiredChange: Type.Optional(Type.String({ minLength: 1 })),
567
788
  }, { additionalProperties: false });
789
+ const scopeProtocol = createFrontendReviewScopeProtocol({ phase: "review", inventory: input.inventory, getStore: () => store, attemptId });
790
+ const savedFindings = () => [...new Map(readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "review-finding").map(r => [r.fact.id, r.fact.finding])).values()];
791
+ const allFindings = (direct) => [...new Map([...savedFindings(), ...(Array.isArray(direct) ? direct : [])].map(finding => [JSON.stringify(finding), finding])).values()];
792
+ const recordFindingTool = defineTool({
793
+ name: "record_review_finding", label: "record_review_finding",
794
+ description: "Save one finding with a stable id. Submit findings incrementally, then finalize without repeating the findings array. Saved blocking findings cannot be omitted from approval; correct a finding explicitly with replace:true.",
795
+ parameters: Type.Object({ id: Type.String({ minLength: 1 }), finding: findingSchema }, { additionalProperties: false }),
796
+ async execute(_callId, params) {
797
+ const fact = { kind: "review-finding", id: params.id, finding: params.finding };
798
+ const requestId = `${attemptId}:finding:${randomUUID()}`;
799
+ const staged = stageTypedEventFact({ store, requestId, attemptId, fact });
800
+ const adopted = await adoptTypedEventFact({ store, requestId, attemptId, fact, eventId: staged.eventId, expectedRevision: store.revision });
801
+ const details = { ok: true, eventId: adopted.eventId, revision: adopted.revision };
802
+ return { content: [{ type: "text", text: JSON.stringify(details) }], details };
803
+ },
804
+ });
568
805
  const approveParameters = Type.Object({
569
- findings: Type.Array(findingSchema, {
806
+ findings: Type.Optional(Type.Array(findingSchema, {
570
807
  description: "Optional informational findings (Minor/Info only; no Critical/Important on approval)",
571
- }),
808
+ })),
572
809
  }, { additionalProperties: false });
573
810
  const requestParameters = Type.Object({
574
811
  issueCategory: Type.Enum({
@@ -578,16 +815,17 @@ export async function createFrontendReviewTerminalTools(input) {
578
815
  "contract-requirement-gap": "contract-requirement-gap",
579
816
  "unknown": "unknown",
580
817
  }, { description: "Typed issue category (five-value enum)" }),
581
- evidenceRefs: Type.Array(Type.String({}), {
582
- description: "Evidence refs (paths or artifact ids); at least one",
583
- }),
584
- findings: Type.Array(findingSchema, {
585
- description: "At least one finding",
818
+ evidenceRefs: Type.Array(Type.String({ minLength: 1 }), {
819
+ description: "Evidence refs (paths or artifact ids); at least one", minItems: 1,
586
820
  }),
821
+ findings: Type.Optional(Type.Array(findingSchema, {
822
+ description: "At least one finding", minItems: 1,
823
+ })),
587
824
  }, { additionalProperties: false });
588
825
  async function adoptReviewFact(kind, fact) {
589
826
  const requestId = randomUUID();
590
827
  try {
828
+ scopeProtocol.assertComplete();
591
829
  const parsed = kind === "approve_review"
592
830
  ? approveReviewFactSchema.parse(fact)
593
831
  : requestReviewChangesFactSchema.parse(fact);
@@ -649,7 +887,7 @@ export async function createFrontendReviewTerminalTools(input) {
649
887
  return adoptReviewFact("approve_review", {
650
888
  kind: "approve_review",
651
889
  verdict: "approve_review",
652
- findings: params?.findings ?? [],
890
+ findings: allFindings(params?.findings),
653
891
  });
654
892
  },
655
893
  });
@@ -665,17 +903,15 @@ export async function createFrontendReviewTerminalTools(input) {
665
903
  verdict: "request_review_changes",
666
904
  issueCategory: params?.issueCategory,
667
905
  evidenceRefs: params?.evidenceRefs,
668
- findings: params?.findings,
906
+ findings: allFindings(params?.findings),
669
907
  });
670
908
  },
671
909
  });
672
- return {
673
- customTools: [approveReviewTool, requestReviewChangesTool],
674
- flush: async () => {
675
- const committed = readCommittedEvents(store, attemptId);
676
- await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "review-typed-facts.jsonl"), committed);
677
- },
678
- };
910
+ const durable = await createDurableFrontendTools({
911
+ file: path.join(input.runDir, input.nodeId, "review-typed-facts.jsonl"), attemptId, store: input.store,
912
+ binding: { inputDigest: input.inputDigest, scopeDigest: input.inventory?.digest }, setWorkingStore: next => { store = next; }, tools: [...scopeProtocol.customTools, recordFindingTool, approveReviewTool, requestReviewChangesTool], validateRestored: async () => { await input.inventory?.validate(); },
913
+ });
914
+ return { ...durable, scopeProtocol };
679
915
  }
680
916
  /**
681
917
  * M8: build the two committed typed design terminal tools (approve_design /
@@ -692,21 +928,35 @@ export async function createFrontendDesignTerminalTools(input) {
692
928
  ]);
693
929
  const { approveDesignFactSchema, readCommittedEvents, requestDesignChangesFactSchema, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
694
930
  const { adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
695
- const store = input.store;
931
+ let store = input.store;
696
932
  const attemptId = input.attemptId;
697
933
  const findingSchema = Type.Object({
698
- severity: Type.String({
699
- description: "Critical | Important | Minor | Info",
700
- }),
701
- file: Type.Optional(Type.String({})),
702
- line: Type.Optional(Type.Number({})),
703
- issue: Type.String({}),
704
- requiredChange: Type.Optional(Type.String({})),
934
+ severity: Type.Enum({ Critical: "Critical", Important: "Important", Minor: "Minor", Info: "Info" }),
935
+ file: Type.Optional(Type.String({ minLength: 1 })),
936
+ line: Type.Optional(Type.Integer({ minimum: 1 })),
937
+ issue: Type.String({ minLength: 1 }),
938
+ requiredChange: Type.Optional(Type.String({ minLength: 1 })),
705
939
  }, { additionalProperties: false });
940
+ const scopeProtocol = createFrontendReviewScopeProtocol({ phase: "design", inventory: input.inventory, getStore: () => store, attemptId });
941
+ const savedFindings = () => [...new Map(readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "design-finding").map(r => [r.fact.id, r.fact.finding])).values()];
942
+ const allFindings = (direct) => [...new Map([...savedFindings(), ...(Array.isArray(direct) ? direct : [])].map(finding => [JSON.stringify(finding), finding])).values()];
943
+ const recordFindingTool = defineTool({
944
+ name: "record_design_finding", label: "record_design_finding",
945
+ description: "Save one finding with a stable id. Submit findings incrementally, then finalize without repeating the findings array. Saved blocking findings cannot be omitted from approval; correct a finding explicitly with replace:true.",
946
+ parameters: Type.Object({ id: Type.String({ minLength: 1 }), finding: findingSchema }, { additionalProperties: false }),
947
+ async execute(_callId, params) {
948
+ const fact = { kind: "design-finding", id: params.id, finding: params.finding };
949
+ const requestId = `${attemptId}:finding:${randomUUID()}`;
950
+ const staged = stageTypedEventFact({ store, requestId, attemptId, fact });
951
+ const adopted = await adoptTypedEventFact({ store, requestId, attemptId, fact, eventId: staged.eventId, expectedRevision: store.revision });
952
+ const details = { ok: true, eventId: adopted.eventId, revision: adopted.revision };
953
+ return { content: [{ type: "text", text: JSON.stringify(details) }], details };
954
+ },
955
+ });
706
956
  const approveParameters = Type.Object({
707
- findings: Type.Array(findingSchema, {
957
+ findings: Type.Optional(Type.Array(findingSchema, {
708
958
  description: "Optional informational findings (Minor/Info only; no Critical/Important on approval)",
709
- }),
959
+ })),
710
960
  }, { additionalProperties: false });
711
961
  const requestParameters = Type.Object({
712
962
  issueCategory: Type.Enum({
@@ -716,16 +966,17 @@ export async function createFrontendDesignTerminalTools(input) {
716
966
  "contract-requirement-gap": "contract-requirement-gap",
717
967
  "unknown": "unknown",
718
968
  }, { description: "Typed issue category (five-value enum)" }),
719
- evidenceRefs: Type.Array(Type.String({}), {
720
- description: "Evidence refs (paths or artifact ids); at least one",
721
- }),
722
- findings: Type.Array(findingSchema, {
723
- description: "At least one finding",
969
+ evidenceRefs: Type.Array(Type.String({ minLength: 1 }), {
970
+ description: "Evidence refs (paths or artifact ids); at least one", minItems: 1,
724
971
  }),
972
+ findings: Type.Optional(Type.Array(findingSchema, {
973
+ description: "At least one finding", minItems: 1,
974
+ })),
725
975
  }, { additionalProperties: false });
726
976
  async function adoptDesignFact(kind, fact) {
727
977
  const requestId = randomUUID();
728
978
  try {
979
+ scopeProtocol.assertComplete();
729
980
  const parsed = kind === "approve_design"
730
981
  ? approveDesignFactSchema.parse(fact)
731
982
  : requestDesignChangesFactSchema.parse(fact);
@@ -787,7 +1038,7 @@ export async function createFrontendDesignTerminalTools(input) {
787
1038
  return adoptDesignFact("approve_design", {
788
1039
  kind: "approve_design",
789
1040
  verdict: "approve_design",
790
- findings: params?.findings ?? [],
1041
+ findings: allFindings(params?.findings),
791
1042
  });
792
1043
  },
793
1044
  });
@@ -803,17 +1054,15 @@ export async function createFrontendDesignTerminalTools(input) {
803
1054
  verdict: "request_design_changes",
804
1055
  issueCategory: params?.issueCategory,
805
1056
  evidenceRefs: params?.evidenceRefs,
806
- findings: params?.findings,
1057
+ findings: allFindings(params?.findings),
807
1058
  });
808
1059
  },
809
1060
  });
810
- return {
811
- customTools: [approveDesignTool, requestDesignChangesTool],
812
- flush: async () => {
813
- const committed = readCommittedEvents(store, attemptId);
814
- await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "design-typed-facts.jsonl"), committed);
815
- },
816
- };
1061
+ const durable = await createDurableFrontendTools({
1062
+ file: path.join(input.runDir, input.nodeId, "design-typed-facts.jsonl"), attemptId, store: input.store,
1063
+ binding: { inputDigest: input.inputDigest, scopeDigest: input.inventory?.digest }, setWorkingStore: next => { store = next; }, tools: [...scopeProtocol.customTools, recordFindingTool, approveDesignTool, requestDesignChangesTool], validateRestored: async () => { await input.inventory?.validate(); },
1064
+ });
1065
+ return { ...durable, scopeProtocol };
817
1066
  }
818
1067
  /**
819
1068
  * Source fidelity ledger (AC-005/AC-006): load the contract node's committed
@@ -858,6 +1107,8 @@ async function loadContractRequirementInheritance(runDir) {
858
1107
  : undefined;
859
1108
  if (sourceFragmentIds || sourceRefs) {
860
1109
  byId.set(id, {
1110
+ ...(typeof recordFact.text === "string" ? { text: recordFact.text } : {}),
1111
+ ...(recordFact.execution !== undefined ? { execution: frontendExecutionSchema.parse(recordFact.execution) } : {}),
861
1112
  ...(sourceFragmentIds ? { sourceFragmentIds } : {}),
862
1113
  ...(sourceRefs ? { sourceRefs } : {}),
863
1114
  });
@@ -906,8 +1157,111 @@ async function resolveFrontendPlanNewComponentSourceReferences(input) {
906
1157
  }
907
1158
  }
908
1159
  /**
909
- * A+B: `frontend-plan-pi` now records its decision ledger through seven
910
- * incremental `record_*` tools (origin=plan) and closes with exactly one
1160
+ * Frozen canonical requirements keyed by id, resolved from the source-fidelity
1161
+ * ledger before the contract node starts. record_requirement commits these
1162
+ * runtime-owned values so the model can never rewrite authoritative requirement
1163
+ * text or stringify the fragment bindings.
1164
+ */
1165
+ async function resolveFrontendCanonicalRequirements(input) {
1166
+ const binding = input.sourceBinding;
1167
+ if (!binding || binding.schemaVersion !== 2 || !binding.ledgerPath) {
1168
+ return new Map();
1169
+ }
1170
+ const absolutePath = path.resolve(input.cwd, binding.ledgerPath);
1171
+ const workspaceRoot = path.resolve(input.cwd);
1172
+ if (absolutePath !== workspaceRoot &&
1173
+ !absolutePath.startsWith(`${workspaceRoot}${path.sep}`)) {
1174
+ throw new Error("canonical requirement ledger escapes workspace");
1175
+ }
1176
+ try {
1177
+ const raw = await readFile(absolutePath, "utf8");
1178
+ if (sha256Text(raw) !== binding.ledgerSha256) {
1179
+ throw new Error("source ledger is stale");
1180
+ }
1181
+ const ledger = parseLedgerJson(raw);
1182
+ const sourceFragments = new Map();
1183
+ const boundFragments = new Map(ledger.fragments.map((fragment) => [fragment.id, fragment]));
1184
+ for (const sourcePath of new Set(ledger.fragments.map((fragment) => fragment.path))) {
1185
+ const sourceAbsolute = path.resolve(workspaceRoot, sourcePath);
1186
+ if (!sourceAbsolute.startsWith(`${workspaceRoot}${path.sep}`)) {
1187
+ throw new Error("source fragment escapes workspace");
1188
+ }
1189
+ const content = await readFile(sourceAbsolute, "utf8");
1190
+ for (const fragment of splitDocumentIntoFragments({ path: sourcePath, content })) {
1191
+ const frozen = boundFragments.get(fragment.id);
1192
+ if (frozen && frozen.sha256 === sha256Text(fragment.text)) {
1193
+ sourceFragments.set(fragment.id, fragment.text);
1194
+ }
1195
+ }
1196
+ }
1197
+ return new Map(ledger.canonicalRequirements.map((requirement) => [
1198
+ requirement.id,
1199
+ {
1200
+ text: requirement.text,
1201
+ sourceFragmentIds: [...(requirement.sourceFragmentIds ?? [])],
1202
+ sourceFragments,
1203
+ },
1204
+ ]));
1205
+ }
1206
+ catch (error) {
1207
+ throw new Error(`canonical requirement sources unavailable: ${error instanceof Error ? error.message : String(error)}`);
1208
+ }
1209
+ }
1210
+ /**
1211
+ * Authoritative UI state ids declared by the contract node
1212
+ * (ui-state-declaration facts). Empty when the source declares no UI-state
1213
+ * table, in which case the plan's state vocabulary is registry-only.
1214
+ */
1215
+ async function resolveFrontendDeclaredUiStateIds(input) {
1216
+ try {
1217
+ const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
1218
+ return records.flatMap((record) => {
1219
+ const fact = record.fact;
1220
+ return record.phase === "committed" &&
1221
+ fact.kind === "ui-state-declaration" &&
1222
+ typeof fact.id === "string"
1223
+ ? [fact.id]
1224
+ : [];
1225
+ });
1226
+ }
1227
+ catch {
1228
+ // No contract facts (or no declared states): registry-only vocabulary.
1229
+ return [];
1230
+ }
1231
+ }
1232
+ /**
1233
+ * Resolve the frozen canonical behavior verification-target ids the PRD
1234
+ * declares. The requirement text is the same authority the design review reads
1235
+ * when it rejects `VT-SHARD-N-…` / variant ids as a contract-requirement gap,
1236
+ * so extracting `VT-…` tokens from the frozen contract requirement facts makes
1237
+ * that authority deterministic instead of prose-only. An empty result (no PRD
1238
+ * declared any behavior target id) disables the canonical check and keeps the
1239
+ * historical free-form path.
1240
+ */
1241
+ export async function resolveFrontendCanonicalVerificationTargetIds(input) {
1242
+ try {
1243
+ const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
1244
+ const ids = new Set();
1245
+ for (const record of records) {
1246
+ const fact = record.fact;
1247
+ if (record.phase !== "committed" || fact.kind !== "requirement") {
1248
+ continue;
1249
+ }
1250
+ const text = typeof fact.text === "string" ? fact.text : "";
1251
+ for (const match of text.matchAll(/\bVT-[A-Z0-9][A-Z0-9_-]*\b/g)) {
1252
+ ids.add(match[0]);
1253
+ }
1254
+ }
1255
+ return [...ids].sort();
1256
+ }
1257
+ catch {
1258
+ // No contract facts (or unreadable): fall back to no canonical set.
1259
+ return [];
1260
+ }
1261
+ }
1262
+ /**
1263
+ * A+B: `frontend-plan-pi` records its decision ledger through incremental
1264
+ * `record_*` tools (origin=plan) and closes with exactly one
911
1265
  * `finalize_plan` terminal. A later attempt may explicitly adopt a quarantined
912
1266
  * fact via `adopt_staged_fact`. Flush writes `plan-typed-facts.jsonl` for the
913
1267
  * node validator / compile authority.
@@ -920,26 +1274,12 @@ export async function createFrontendPlanLedgerTools(input) {
920
1274
  const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
921
1275
  const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
922
1276
  const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
923
- const store = input.store;
1277
+ let store = input.store;
924
1278
  const attemptId = input.attemptId;
925
1279
  let activeRequirementScope = [];
926
1280
  const scopedRequirementIds = () => [...activeRequirementScope];
927
- // A retry creates a fresh executor-local store, but the plan ledger is the
928
- // cross-attempt authority. Restore the committed prefix before registering
929
- // tools; otherwise the first flush of a retry can overwrite facts that the
930
- // previous attempt had already committed. The on-disk file contains only
931
- // committed records, so loading it is also fail-closed with respect to
932
- // staged/quarantined facts.
933
- const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
934
- if (persisted.records.length > 0) {
935
- const existingEventIds = new Set(store.records.map((record) => record.eventId));
936
- for (const record of persisted.records) {
937
- if (!existingEventIds.has(record.eventId)) {
938
- store.records.push(record);
939
- }
940
- }
941
- store.revision = Math.max(store.revision, persisted.revision);
942
- }
1281
+ const contractInheritance = await loadContractRequirementInheritance(input.runDir);
1282
+ const executionGroups = collectFrontendExecutionGroups([...contractInheritance].map(([id, r]) => ({ id, ...r })));
943
1283
  const stringArray = Type.Array(Type.String({}));
944
1284
  const optionalString = Type.Optional(Type.String({}));
945
1285
  const optionalStringArray = Type.Optional(stringArray);
@@ -974,9 +1314,7 @@ export async function createFrontendPlanLedgerTools(input) {
974
1314
  verificationTargetIds: stringArray,
975
1315
  }, { additionalProperties: false });
976
1316
  const mockEndpointSchema = Type.Object({
977
- method: Type.String({
978
- description: "GET | POST | PUT | PATCH | DELETE | HEAD | OPTIONS",
979
- }),
1317
+ method: Type.Enum({ GET: "GET", POST: "POST", PUT: "PUT", PATCH: "PATCH", DELETE: "DELETE", HEAD: "HEAD", OPTIONS: "OPTIONS" }),
980
1318
  path: Type.String({}),
981
1319
  fixture: optionalString,
982
1320
  consumer: optionalString,
@@ -998,23 +1336,24 @@ export async function createFrontendPlanLedgerTools(input) {
998
1336
  paths: stringArray,
999
1337
  conflicts: stringArray,
1000
1338
  }, { additionalProperties: false });
1001
- // Enum fields use literal unions, not advisory strings: a soft Type.String
1002
- // lets the model commit values like type="behavior" that pass the tool
1003
- // boundary, flush into the ledger, and only fail the compile-time zod enum
1004
- // — a deterministic attempt failure the model could have fixed in-node.
1005
- const verificationTargetTypeSchema = Type.Union([
1006
- Type.Literal("static"),
1339
+ // Verification mode is runtime-owned (contract v2): the plan submits a
1340
+ // commandId referencing the frozen command directory, never a type. A
1341
+ // soft Type.String would let the model commit values that only fail at
1342
+ // compile time — keep the reference a required string and validate it
1343
+ // against the directory at the tool boundary below.
1344
+ const verificationTargetScopeSchema = Type.Union([
1007
1345
  Type.Literal("unit"),
1008
1346
  Type.Literal("component"),
1009
1347
  Type.Literal("integration"),
1010
- Type.Literal("mock"),
1011
1348
  ]);
1012
1349
  const verificationTargetSchema = Type.Object({
1013
1350
  id: Type.String({}),
1014
- type: verificationTargetTypeSchema,
1015
- commandLabel: Type.String({}),
1351
+ commandId: Type.String({
1352
+ description: "Frozen command directory key (e.g. verify-npm-run-build); the runtime resolves mode and label from it.",
1353
+ }),
1016
1354
  file: Type.String({}),
1017
1355
  requirementIds: stringArray,
1356
+ scope: Type.Optional(verificationTargetScopeSchema),
1018
1357
  uiStates: Type.Optional(Type.Array(Type.String({}), {
1019
1358
  description: "Optional only at this tool boundary. An omitted value is deterministically recorded as []. Pass an explicit array for new calls.",
1020
1359
  })),
@@ -1037,9 +1376,15 @@ export async function createFrontendPlanLedgerTools(input) {
1037
1376
  specReference: Type.Optional(Type.Object({
1038
1377
  path: Type.String({}),
1039
1378
  section: Type.String({}),
1040
- line: Type.Optional(Type.Number({})),
1379
+ line: Type.Optional(Type.Integer({ minimum: 1 })),
1041
1380
  }, { additionalProperties: false })),
1042
- rationale: Type.String({}),
1381
+ rationale: Type.Optional(Type.String({ minLength: 1 })),
1382
+ covers: Type.Optional(Type.Array(Type.String({}), {
1383
+ description: "UI state and/or interaction names this single component choice covers (one choice may cover many ids).",
1384
+ })),
1385
+ evidencePath: Type.Optional(Type.String({
1386
+ description: "REQUIRED for decision=reuse-existing: repo-relative path whose existing file is the reuse evidence. Greenfield paths must use decision=new.",
1387
+ })),
1043
1388
  }, { additionalProperties: false });
1044
1389
  const stringList = (value) => Array.isArray(value)
1045
1390
  ? value.filter((item) => typeof item === "string")
@@ -1118,7 +1463,7 @@ export async function createFrontendPlanLedgerTools(input) {
1118
1463
  const recordComponentChoiceTool = defineTool({
1119
1464
  name: "record_component_choice",
1120
1465
  label: "record_component_choice",
1121
- description: "Record ONE component choice (origin=plan component-choice fact). Declare every UI purpose's component selection. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo (e.g. reusing ActiveRunBadge's styling convention). Omit rationale for reuse-existing; it is optional. Call up to 5 component choices per assistant message (batching reduces API round trips and rate-limit risk); never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\"}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"}",
1466
+ description: "Record ONE component choice (origin=plan component-choice fact). One choice may cover MULTIPLE UI states/interactions via covers: [\"<state-or-interaction name>\", ...]; do not emit one row per interaction. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo and REQUIRES evidencePath: a repo-relative path to the existing file that proves the reuse — the runtime verifies the file exists (fresh evidence); a path with no existing file is greenfield and must use decision=new instead. Call up to 5 component choices per assistant message; never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\", \"covers\": [\"<other interaction names this component also serves>\"]}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"} or {\"choice\": {\"purpose\": \"FocusQueuePanel\", \"component\": \"FocusQueuePanel\", \"decision\": \"reuse-existing\", \"evidencePath\": \"src/journal.ts\"}}",
1122
1467
  promptSnippet: "Record 1-5 component choices (up to 5 per message).",
1123
1468
  parameters: Type.Object({
1124
1469
  choice: uiComponentChoiceSchema,
@@ -1169,6 +1514,46 @@ export async function createFrontendPlanLedgerTools(input) {
1169
1514
  ...(citation.line !== undefined ? { line: citation.line } : {}),
1170
1515
  };
1171
1516
  }
1517
+ if (choice.decision === "reuse-existing") {
1518
+ // reuse-existing must cite a file that actually exists NOW (fresh
1519
+ // existence evidence, same semantics as Scout pathEvidence.fresh).
1520
+ // A path with no file on disk is greenfield: decision=new is the
1521
+ // only honest choice (dogfood run dag-1788504923861-0f7b17a9
1522
+ // claimed 18 reuse-existing conventions in a not-yet-written
1523
+ // src/planner.ts and every gate let it through).
1524
+ const evidencePath = typeof choice.evidencePath === "string"
1525
+ ? choice.evidencePath.trim()
1526
+ : "";
1527
+ if (!evidencePath) {
1528
+ return planToolReceipt({
1529
+ ok: false,
1530
+ kind: "component-choice",
1531
+ error: "decision=reuse-existing requires evidencePath naming the existing repo file that proves the reuse; if the file does not exist yet, use decision=new",
1532
+ });
1533
+ }
1534
+ if (input.workspaceRoot) {
1535
+ const absolute = path.resolve(input.workspaceRoot, evidencePath);
1536
+ const workspaceRoot = path.resolve(input.workspaceRoot);
1537
+ const contained = absolute === workspaceRoot ||
1538
+ absolute.startsWith(`${workspaceRoot}${path.sep}`);
1539
+ let exists = false;
1540
+ if (contained) {
1541
+ try {
1542
+ exists = (await stat(absolute)).isFile();
1543
+ }
1544
+ catch {
1545
+ exists = false;
1546
+ }
1547
+ }
1548
+ if (!exists) {
1549
+ return planToolReceipt({
1550
+ ok: false,
1551
+ kind: "component-choice",
1552
+ error: `decision=reuse-existing evidencePath "${evidencePath}" has no fresh existence evidence (file not found in the workspace); reuse requires an existing file — use decision=new for greenfield paths`,
1553
+ });
1554
+ }
1555
+ }
1556
+ }
1172
1557
  const components = typeof choice.component === "string" ? [choice.component] : [];
1173
1558
  const result = await adoptPlanFact("component-choice", `${attemptId}:record_component_choice:${randomUUID()}`, {
1174
1559
  kind: "component-choice",
@@ -1192,10 +1577,86 @@ export async function createFrontendPlanLedgerTools(input) {
1192
1577
  return planToolReceipt(echo);
1193
1578
  },
1194
1579
  });
1580
+ // Global UX vocabulary: one compact registry committed BEFORE any state
1581
+ // flow details. Coverage may slice by AC; UX must not — the registry is
1582
+ // the anti-duplication anchor that keeps every later slice on the same
1583
+ // named concepts instead of re-inventing them per AC chunk.
1584
+ const recordStateRegistryTool = defineTool({
1585
+ name: "record_state_registry",
1586
+ label: "record_state_registry",
1587
+ description: "Commit the GLOBAL UX vocabulary (origin=plan state-registry fact) BEFORE any record_state_flow call: uiStateNames (use the contract's declared authoritative state ids when provided) and interactionNames (stable behavior-domain kebab-case names, e.g. planner-task-edit / focus-queue-move — one name per behavior domain, never one per AC). Empty arrays explicitly mean the request has no UI state or interaction vocabulary. Re-record the full vocabulary to correct it; the latest commit wins, but it may not remove names still referenced by committed state-flow facts. Example: {\"uiStateNames\": [\"planner-empty\"], \"interactionNames\": [\"planner-task-create\", \"focus-queue-move\"]}",
1588
+ promptSnippet: "Record the global UI-state/interaction vocabulary once, before any state flow.",
1589
+ parameters: Type.Object({
1590
+ uiStateNames: stringArray,
1591
+ interactionNames: stringArray,
1592
+ }, { additionalProperties: false }),
1593
+ async execute(_toolCallId, params) {
1594
+ const uiStateNames = [...new Set(stringList(params?.uiStateNames).map((name) => name.trim()))];
1595
+ const interactionNames = [
1596
+ ...new Set(stringList(params?.interactionNames).map((name) => name.trim())),
1597
+ ];
1598
+ if (uiStateNames.some((name) => name.length === 0) || interactionNames.some((name) => name.length === 0)) {
1599
+ return planToolReceipt({
1600
+ ok: false,
1601
+ kind: "state-registry",
1602
+ error: "record_state_registry names must be non-empty strings",
1603
+ });
1604
+ }
1605
+ const invalidInteractionNames = interactionNames.filter((name) => !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(name));
1606
+ if (invalidInteractionNames.length > 0) {
1607
+ return planToolReceipt({
1608
+ ok: false,
1609
+ kind: "state-registry",
1610
+ error: `record_state_registry interactionNames must be stable kebab-case behavior-domain names: ${invalidInteractionNames.join(", ")}`,
1611
+ });
1612
+ }
1613
+ if (input.declaredUiStateIds &&
1614
+ input.declaredUiStateIds.length > 0) {
1615
+ const declared = new Set(input.declaredUiStateIds);
1616
+ const undeclared = uiStateNames.filter((name) => !declared.has(name));
1617
+ if (undeclared.length > 0) {
1618
+ return planToolReceipt({
1619
+ ok: false,
1620
+ kind: "state-registry",
1621
+ error: `record_state_registry uiStateNames are not declared by the contract's authoritative UI-state table: ${undeclared.join(", ")}; use the declared ids (${input.declaredUiStateIds.join(", ")})`,
1622
+ });
1623
+ }
1624
+ const missing = input.declaredUiStateIds.filter((name) => !uiStateNames.includes(name));
1625
+ if (missing.length > 0) {
1626
+ return planToolReceipt({
1627
+ ok: false,
1628
+ kind: "state-registry",
1629
+ error: `record_state_registry must include every state from the contract's authoritative UI-state table; missing: ${missing.join(", ")}`,
1630
+ });
1631
+ }
1632
+ }
1633
+ // A correction is deterministic only when the new last-wins registry
1634
+ // still contains every live name already committed by state-flow facts.
1635
+ // This permits adding a missed concept, while preventing a registry edit
1636
+ // from retroactively orphaning earlier slices.
1637
+ const liveNames = collectCanonicalStateFlowNames(readCommittedEvents(store, attemptId));
1638
+ const orphanedUiStates = [...liveNames.uiStateNames].filter((name) => !uiStateNames.includes(name));
1639
+ const orphanedInteractions = [...liveNames.interactionNames].filter((name) => !interactionNames.includes(name));
1640
+ if (orphanedUiStates.length > 0 || orphanedInteractions.length > 0) {
1641
+ return planToolReceipt({
1642
+ ok: false,
1643
+ kind: "state-registry",
1644
+ error: `record_state_registry cannot remove names still referenced by committed state-flow facts (uiStates: ${orphanedUiStates.join(", ") || "none"}; interactions: ${orphanedInteractions.join(", ") || "none"}); first correct/remove those state-flow entries, then re-record the full registry`,
1645
+ });
1646
+ }
1647
+ const result = await adoptPlanFact("state-registry", `${attemptId}:record_state_registry:${randomUUID()}`, {
1648
+ kind: "state-registry",
1649
+ origin: "plan",
1650
+ uiStateNames,
1651
+ interactionNames,
1652
+ });
1653
+ return planToolReceipt(result);
1654
+ },
1655
+ });
1195
1656
  const recordStateFlowTool = defineTool({
1196
1657
  name: "record_state_flow",
1197
1658
  label: "record_state_flow",
1198
- description: "Record UI states and interactions as an origin=plan state-flow fact. To correct stale named entries from an earlier UX batch, include removeUiStateNames and/or removeInteractionNames. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
1659
+ description: "Record UI states and interactions as an origin=plan state-flow fact. REQUIRES a committed record_state_registry vocabulary first, and every name here must be in that registry; uiState names must also be contract-declared authoritative ids when the contract declares them. To correct stale named entries from an earlier UX batch, include removeUiStateNames and/or removeInteractionNames. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
1199
1660
  promptSnippet: "Record the plan state-flow fact.",
1200
1661
  parameters: Type.Object({
1201
1662
  uiStates: Type.Array(uiStateSchema),
@@ -1299,9 +1760,101 @@ export async function createFrontendPlanLedgerTools(input) {
1299
1760
  }
1300
1761
  interactions.push({ ...interaction, name: resolvedName });
1301
1762
  }
1763
+ // UX vocabulary gate: every recorded state/interaction name must be
1764
+ // declared in the committed global registry first. This keeps UX out
1765
+ // of AC-number slicing — the model commits one compact vocabulary
1766
+ // (record_state_registry), then fills behavior-domain details, and a
1767
+ // later slice cannot silently rename an earlier concept (dogfood
1768
+ // dag-1788504923861-0f7b17a9: 5 renames of "prioritize" + 7 of
1769
+ // "focus queue" across AC chunks).
1302
1770
  const states = uiStates
1303
1771
  .map((state) => (typeof state?.name === "string" ? state.name : ""))
1304
1772
  .filter(Boolean);
1773
+ const committedRegistry = readCommittedEvents(store, attemptId)
1774
+ .map((event) => event.fact)
1775
+ .filter((fact) => isRecordObject(fact) && fact.kind === "state-registry")
1776
+ .at(-1);
1777
+ if (!committedRegistry) {
1778
+ return planToolReceipt({
1779
+ ok: false,
1780
+ kind: "state-flow",
1781
+ error: "record_state_flow requires a committed UX registry first: call record_state_registry with the full uiStateNames/interactionNames vocabulary, then record state flows against it",
1782
+ });
1783
+ }
1784
+ const registryUiStateNames = new Set(stringList(committedRegistry.uiStateNames));
1785
+ const registryInteractionNames = new Set(stringList(committedRegistry.interactionNames));
1786
+ const unknownStates = states.filter((name) => !registryUiStateNames.has(name));
1787
+ if (unknownStates.length > 0) {
1788
+ return planToolReceipt({
1789
+ ok: false,
1790
+ kind: "state-flow",
1791
+ error: `record_state_flow uiState names not in the committed registry: ${unknownStates.join(", ")}; re-record record_state_registry with the complete vocabulary first (registered: ${[...registryUiStateNames].join(", ") || "(none)"})`,
1792
+ });
1793
+ }
1794
+ const unknownInteractions = interactions
1795
+ .map((interaction) => typeof interaction?.name === "string" ? interaction.name : "")
1796
+ .filter((name) => name && !registryInteractionNames.has(name));
1797
+ if (unknownInteractions.length > 0) {
1798
+ return planToolReceipt({
1799
+ ok: false,
1800
+ kind: "state-flow",
1801
+ error: `record_state_flow interaction names not in the committed registry: ${unknownInteractions.join(", ")}; re-record record_state_registry with the complete vocabulary first (registered: ${[...registryInteractionNames].join(", ") || "(none)"})`,
1802
+ });
1803
+ }
1804
+ // Authoritative UI states: when the contract node declared the
1805
+ // source's UI-state table, planner states must bind those ids — no
1806
+ // invented variants (planner-empty/create-invalid/no-results/
1807
+ // editing/focus-full/storage-unavailable, not "planner-list").
1808
+ if (input.declaredUiStateIds &&
1809
+ input.declaredUiStateIds.length > 0) {
1810
+ const declared = new Set(input.declaredUiStateIds);
1811
+ const undeclaredStates = states.filter((name) => !declared.has(name));
1812
+ if (undeclaredStates.length > 0) {
1813
+ return planToolReceipt({
1814
+ ok: false,
1815
+ kind: "state-flow",
1816
+ error: `record_state_flow uiState names are not declared by the contract's authoritative UI-state table: ${undeclaredStates.join(", ")}; use the declared ids (${input.declaredUiStateIds.join(", ")}), or record a design deviation if the source table is genuinely incomplete`,
1817
+ });
1818
+ }
1819
+ }
1820
+ // Named entries are last-wins at assembly. A scoped correction may
1821
+ // replace its own bindings, but must retain other scopes' coverage.
1822
+ const scope = new Set(scopedRequirementIds());
1823
+ const committed = readCommittedEvents(store, attemptId).map(event => event.fact);
1824
+ const targetOwners = new Map();
1825
+ for (const fact of committed) {
1826
+ if (fact.kind === "plan-verification-target" && isRecordObject(fact.entry) && typeof fact.entry.id === "string") {
1827
+ targetOwners.set(fact.entry.id, stringList(fact.entry.requirementIds));
1828
+ }
1829
+ }
1830
+ for (const [field, removals] of [["uiStates", removeUiStateNames], ["interactions", removeInteractionNames]]) {
1831
+ const live = new Map();
1832
+ for (const fact of committed) {
1833
+ if (fact.kind !== "state-flow")
1834
+ continue;
1835
+ for (const name of stringList(fact[field === "uiStates" ? "removeUiStateNames" : "removeInteractionNames"]))
1836
+ live.delete(name);
1837
+ for (const entry of Array.isArray(fact[field]) ? fact[field] : []) {
1838
+ if (isRecordObject(entry) && typeof entry.name === "string")
1839
+ live.set(entry.name, entry);
1840
+ }
1841
+ }
1842
+ const foreignBindings = (entry) => scope.size === 0 ? [] : stringList(entry.verificationTargetIds).filter(id => {
1843
+ const owners = targetOwners.get(id);
1844
+ return !owners?.length || owners.some(owner => !scope.has(owner));
1845
+ });
1846
+ for (const name of removals) {
1847
+ const previous = live.get(name);
1848
+ if (previous && foreignBindings(previous).length)
1849
+ return planToolReceipt({ ok: false, kind: "state-flow",
1850
+ error: `Cannot remove shared ${name} from this requirement scope; retain it and update its scoped bindings instead` });
1851
+ }
1852
+ for (const entry of field === "uiStates" ? uiStates : interactions) {
1853
+ const previous = live.get(String(entry.name));
1854
+ if (previous)
1855
+ entry.verificationTargetIds = [...new Set([...foreignBindings(previous), ...stringList(entry.verificationTargetIds)])];
1856
+ }
1857
+ }
1305
1858
  const result = await adoptPlanFact("state-flow", `${attemptId}:record_state_flow:${randomUUID()}`, {
1306
1859
  kind: "state-flow",
1307
1860
  origin: "plan",
@@ -1320,13 +1873,14 @@ export async function createFrontendPlanLedgerTools(input) {
1320
1873
  label: "record_data_flow",
1321
1874
  description: "Record interaction/endpoint data flow as an origin=plan data-flow fact. Example: {\"interactions\": [\"<interaction name>\"], \"endpoints\": [\"GET <path>\"]}",
1322
1875
  promptSnippet: "Record the plan data-flow fact.",
1323
- parameters: Type.Object({ interactions: stringArray, endpoints: stringArray }, { additionalProperties: false }),
1876
+ parameters: Type.Object({ interactions: stringArray, endpoints: stringArray, replace: Type.Optional(Type.Boolean()) }, { additionalProperties: false }),
1324
1877
  async execute(_toolCallId, params) {
1878
+ const previous = params.replace ? undefined : readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "data-flow").at(-1)?.fact;
1325
1879
  const result = await adoptPlanFact("data-flow", `${attemptId}:record_data_flow:${randomUUID()}`, {
1326
1880
  kind: "data-flow",
1327
1881
  origin: "plan",
1328
- interactions: stringList(params?.interactions),
1329
- endpoints: stringList(params?.endpoints),
1882
+ interactions: [...new Set([...stringList(previous?.interactions), ...stringList(params?.interactions)])],
1883
+ endpoints: [...new Set([...stringList(previous?.endpoints), ...stringList(params?.endpoints)])],
1330
1884
  });
1331
1885
  return planToolReceipt(result);
1332
1886
  },
@@ -1354,6 +1908,16 @@ export async function createFrontendPlanLedgerTools(input) {
1354
1908
  return planToolReceipt(result);
1355
1909
  },
1356
1910
  });
1911
+ const recordMockEndpointTool = defineTool({
1912
+ name: "record_mock_endpoint", label: "record_mock_endpoint",
1913
+ description: "Record one Mock/API endpoint. First record_mock_api with the policy and endpoints: []; then submit each endpoint separately. Never regenerate the whole endpoint collection. Use replace:true to revise an existing method/path, or replace:true plus remove:true to withdraw it.",
1914
+ parameters: Type.Object({ endpoint: mockEndpointSchema, remove: Type.Optional(Type.Boolean()) }, { additionalProperties: false }),
1915
+ async execute(_callId, params) {
1916
+ if (!readCommittedEvents(store, attemptId).some(r => r.fact.kind === "mock-api"))
1917
+ return planToolReceipt({ ok: false, kind: "mock-endpoint", code: "MOCK_POLICY_MISSING", error: "Record the mock policy before its endpoints" });
1918
+ return planToolReceipt(await adoptPlanFact("mock-endpoint", `${attemptId}:endpoint:${randomUUID()}`, { kind: "mock-endpoint", origin: "plan", endpoint: params.endpoint, ...(params.remove ? { removed: true } : {}) }));
1919
+ },
1920
+ });
1357
1921
  const recordDesignDeviationTool = defineTool({
1358
1922
  name: "record_design_deviation",
1359
1923
  label: "record_design_deviation",
@@ -1429,6 +1993,8 @@ export async function createFrontendPlanLedgerTools(input) {
1429
1993
  // canonical-coverage gate rejects with no in-node cure. Reject here
1430
1994
  // and name the allowed ids.
1431
1995
  const id = typeof entry.id === "string" ? entry.id : "";
1996
+ if (activeRequirementScope.length && !activeRequirementScope.includes(id))
1997
+ return planToolReceipt({ ok: false, kind: "plan-requirement", code: "FRONTEND_INPUT_SCOPE_VIOLATION", error: `Requirement ${id} is outside this session` });
1432
1998
  if (id &&
1433
1999
  input.requirementIds &&
1434
2000
  input.requirementIds.length > 0 &&
@@ -1460,6 +2026,15 @@ export async function createFrontendPlanLedgerTools(input) {
1460
2026
  });
1461
2027
  }
1462
2028
  }
2029
+ if (id &&
2030
+ activeRequirementScope.length > 0 &&
2031
+ !activeRequirementScope.includes(id)) {
2032
+ return planToolReceipt({
2033
+ ok: false,
2034
+ kind: "plan-requirement",
2035
+ error: `record_plan_requirement id "${id}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
2036
+ });
2037
+ }
1463
2038
  const result = await adoptPlanFact("plan-requirement", `${attemptId}:record_plan_requirement:${randomUUID()}`, {
1464
2039
  kind: "plan-requirement",
1465
2040
  origin: "plan",
@@ -1469,10 +2044,37 @@ export async function createFrontendPlanLedgerTools(input) {
1469
2044
  return planToolReceipt(result);
1470
2045
  },
1471
2046
  });
2047
+ const recordPlanGroupCoverageTool = defineTool({
2048
+ name: "record_plan_group_coverage", label: "record_plan_group_coverage",
2049
+ description: "Submit shared implementation/verification references for one declared execution group. Runtime expands to every canonical member and retains its full outcome and source bindings. A shared VT must actually verify each independent condition. Use per-requirement records for differences; never create UI for constraints or exclusions. replace:true explicitly revises the group.",
2050
+ parameters: Type.Object({ id: Type.String({ minLength: 1 }), implementationTargets: stringArray, verificationTargetIds: stringArray, replace: Type.Optional(Type.Boolean()) }, { additionalProperties: false }),
2051
+ async execute(callId, params, signal, onUpdate, ctx) {
2052
+ const group = executionGroups.find(g => g.id === params.id && g.kind !== "unclassified");
2053
+ if (!group || (activeRequirementScope.length && group.requirementIds.some(id => !activeRequirementScope.includes(id))))
2054
+ return planToolReceipt({ ok: false, kind: "plan-requirement", code: "EXECUTION_GROUP_SCOPE_INVALID", error: "A known complete group must be present in this session; submit individual member records when the group spans scopes" });
2055
+ let last;
2056
+ for (const id of group.requirementIds) {
2057
+ const existing = readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "plan-requirement" && r.fact.entry?.id === id).at(-1)?.fact.entry;
2058
+ if (existing && !params.replace) {
2059
+ if (JSON.stringify(existing.implementationTargets) !== JSON.stringify(params.implementationTargets) || JSON.stringify(existing.verificationTargetIds) !== JSON.stringify(params.verificationTargetIds))
2060
+ return planToolReceipt({ ok: false, kind: "plan-requirement", code: "FACT_IDENTITY_CONFLICT", error: `${id}: existing coverage differs; use replace:true to revise explicitly` });
2061
+ continue;
2062
+ }
2063
+ last = await recordPlanRequirementTool.execute(`${callId}:${id}`, { entry: { id, implementationTargets: params.implementationTargets, verificationTargetIds: params.verificationTargetIds }, ...(params.replace ? { replace: true } : {}) }, signal, onUpdate, ctx);
2064
+ if (!last.details?.ok)
2065
+ return last;
2066
+ }
2067
+ return planToolReceipt(await adoptPlanFact("plan-group-coverage", `${attemptId}:group:${randomUUID()}`, { kind: "plan-group-coverage", origin: "plan", id: group.id, requirementIds: group.requirementIds }));
2068
+ },
2069
+ });
1472
2070
  const recordPlanVerificationTargetTool = defineTool({
1473
2071
  name: "record_plan_verification_target",
1474
2072
  label: "record_plan_verification_target",
1475
- description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). For non-static targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A non-static target id is the stable trace token that implementation must place in a real describe/it/test title; static targets are traced by file and command only. Entry carries id, type, commandLabel, file, requirementIds, and uiStates; free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"type\": \"unit\", \"commandLabel\": \"<frozen command label>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}",
2073
+ description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}" +
2074
+ (input.canonicalVerificationTargetIds &&
2075
+ input.canonicalVerificationTargetIds.length > 0
2076
+ ? ` Frozen canonical behavior target ids (use exactly for behavior targets): ${input.canonicalVerificationTargetIds.join(", ")}.`
2077
+ : ""),
1476
2078
  promptSnippet: "Commit 1-4 plan verification target entries (up to 4 per message).",
1477
2079
  parameters: Type.Object({
1478
2080
  entry: verificationTargetSchema,
@@ -1487,19 +2089,61 @@ export async function createFrontendPlanLedgerTools(input) {
1487
2089
  error: "record_plan_verification_target requires a non-empty entry object",
1488
2090
  });
1489
2091
  }
1490
- // Belt-and-braces for providers that do not strictly enforce the
1491
- // tool-schema enum: reject an invalid type here with the allowed
1492
- // values so the model can re-record in-node instead of the whole
1493
- // attempt dying at compile time on the strict zod enum.
1494
- const verificationTargetType = rawEntry.type;
1495
- if (typeof verificationTargetType !== "string" ||
1496
- !["static", "unit", "component", "integration", "mock"].includes(verificationTargetType)) {
2092
+ // Contract v2: the plan references a frozen command by commandId;
2093
+ // mode and label are resolved by the runtime from the frozen
2094
+ // command directory (run.json frontend-verify-shell bundle lanes).
2095
+ // Reject unknown ids and behavior commands bound to non-test
2096
+ // files here so the model fixes them in-node instead of the
2097
+ // attempt dying at materialization or at the writer focused-check.
2098
+ const verificationCommandId = typeof rawEntry.commandId === "string"
2099
+ ? rawEntry.commandId.trim()
2100
+ : "";
2101
+ if (!verificationCommandId) {
1497
2102
  return planToolReceipt({
1498
2103
  ok: false,
1499
2104
  kind: "plan-verification-target",
1500
- error: `record_plan_verification_target entry.type must be one of static | unit | component | integration | mock (received ${JSON.stringify(verificationTargetType ?? null)}); for a runtime behavior check use type=mock or type=integration`,
2105
+ error: `record_plan_verification_target entry.commandId is required (received ${JSON.stringify(rawEntry.commandId ?? null)}); pick one id from the frozen command directory in your prompt`,
1501
2106
  });
1502
2107
  }
2108
+ const { deriveFrontendVerifyCommandDirectoryFromRun, isFrontendTestFilePath, } = await import("../workflows/dag/frontend-implementation-contract.js");
2109
+ const verifyDirectory = await deriveFrontendVerifyCommandDirectoryFromRun(input.runDir);
2110
+ if (verifyDirectory.length > 0) {
2111
+ const directoryEntry = verifyDirectory.find((entry) => entry.commandId === verificationCommandId);
2112
+ if (!directoryEntry) {
2113
+ return planToolReceipt({
2114
+ ok: false,
2115
+ kind: "plan-verification-target",
2116
+ error: `record_plan_verification_target verification-target-unknown-command: unknown commandId "${verificationCommandId}"; available frozen commands: [${verifyDirectory.map((entry) => `${entry.commandId} (${entry.mode}: ${entry.label})`).join(", ")}]`,
2117
+ });
2118
+ }
2119
+ if (directoryEntry.mode === "behavior" &&
2120
+ typeof rawEntry.file === "string" &&
2121
+ !isFrontendTestFilePath(rawEntry.file)) {
2122
+ return planToolReceipt({
2123
+ ok: false,
2124
+ kind: "plan-verification-target",
2125
+ error: `record_plan_verification_target verification-target-phase-mismatch: behavior command "${directoryEntry.label}" (${directoryEntry.commandId}) must bind a test file (__tests__/, tests?/, e2e/, cypress/, *.test.*, *.spec.*, *.cy.*); received file "${rawEntry.file}"`,
2126
+ });
2127
+ }
2128
+ // Canonical-identity check for behavior targets: the PRD freezes
2129
+ // the exact behavior verification-target ids (e.g.
2130
+ // VT-SMOKE-COUNTER-BEHAVIOR). A committed non-canonical id is
2131
+ // immutable and design review rejects it as a
2132
+ // contract-requirement gap, so reject invented ids here.
2133
+ if (directoryEntry.mode === "behavior" &&
2134
+ input.canonicalVerificationTargetIds &&
2135
+ input.canonicalVerificationTargetIds.length > 0) {
2136
+ const canonicalTargetId = typeof rawEntry.id === "string" ? rawEntry.id.trim() : "";
2137
+ if (canonicalTargetId &&
2138
+ !input.canonicalVerificationTargetIds.includes(canonicalTargetId)) {
2139
+ return planToolReceipt({
2140
+ ok: false,
2141
+ kind: "plan-verification-target",
2142
+ error: `record_plan_verification_target id "${canonicalTargetId}" is not a frozen canonical behavior verification target; canonical ids are: ${input.canonicalVerificationTargetIds.join(", ")}`,
2143
+ });
2144
+ }
2145
+ }
2146
+ }
1503
2147
  // Duplicate-id rejection: committed typed facts are immutable, so
1504
2148
  // re-recording the same VT id would deadlock the compile by default.
1505
2149
  // replace=true is the explicit in-node correction path.
@@ -1568,6 +2212,15 @@ export async function createFrontendPlanLedgerTools(input) {
1568
2212
  : [];
1569
2213
  }));
1570
2214
  const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
2215
+ const outOfScopeRequirementIds = stringList(rawEntry.requirementIds).filter((id) => activeRequirementScope.length > 0 &&
2216
+ !activeRequirementScope.includes(id));
2217
+ if (outOfScopeRequirementIds.length > 0) {
2218
+ return planToolReceipt({
2219
+ ok: false,
2220
+ kind: "plan-verification-target",
2221
+ error: `record_plan_verification_target references requirements outside this session's scope [${activeRequirementScope.join(", ")}]: ${outOfScopeRequirementIds.join(", ")}`,
2222
+ });
2223
+ }
1571
2224
  if (unknownRequirementIds.length > 0) {
1572
2225
  return planToolReceipt({
1573
2226
  ok: false,
@@ -1614,6 +2267,18 @@ export async function createFrontendPlanLedgerTools(input) {
1614
2267
  error: "record_plan_evidence_gap requires a non-empty description describing the gap",
1615
2268
  });
1616
2269
  }
2270
+ const requirementId = typeof entry.requirementId === "string"
2271
+ ? entry.requirementId
2272
+ : undefined;
2273
+ if (requirementId &&
2274
+ activeRequirementScope.length > 0 &&
2275
+ !activeRequirementScope.includes(requirementId)) {
2276
+ return planToolReceipt({
2277
+ ok: false,
2278
+ kind: "plan-evidence-gap",
2279
+ error: `record_plan_evidence_gap requirementId "${requirementId}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
2280
+ });
2281
+ }
1617
2282
  const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
1618
2283
  return planToolReceipt(result);
1619
2284
  },
@@ -1621,7 +2286,7 @@ export async function createFrontendPlanLedgerTools(input) {
1621
2286
  const finalizePlanTool = defineTool({
1622
2287
  name: "finalize_plan",
1623
2288
  label: "finalize_plan",
1624
- description: "Commit the finalize_plan terminal. Requirements, verification targets, and evidence gaps (optional) were already committed incrementally through record_plan_requirement / record_plan_verification_target / record_plan_evidence_gap; finalize_plan assembles them from the ledger together with these optional remaining fields, publishes the canonical editable patch on a target-surface fact, and commits the terminal. Call exactly once.",
2289
+ description: "Commit the finalize_plan terminal. Requirements, verification targets, and evidence gaps (optional) were already committed incrementally through record_plan_requirement / record_plan_verification_target / record_plan_evidence_gap; finalize_plan assembles them from the ledger together with these optional remaining fields, publishes the canonical editable patch on a target-surface fact, and commits the terminal. A successful terminal commit occurs exactly once. If validation fails, correct only the reported facts and retry finalize.",
1625
2290
  promptSnippet: "Commit the finalize_plan terminal (ledger fields + optional residualRisks / realIntegrationGap).",
1626
2291
  parameters: Type.Object({
1627
2292
  residualRisks: optionalStringArray,
@@ -1636,6 +2301,9 @@ export async function createFrontendPlanLedgerTools(input) {
1636
2301
  // when the plan did not re-declare them. The contract node is the
1637
2302
  // sole synthesis point; the plan inherits by requirement id.
1638
2303
  const contractInheritance = await loadContractRequirementInheritance(input.runDir);
2304
+ const missingData = collectFrontendPlanPhaseMissingFacts({ phase: "global-mock-data", requirementIds: input.requirementIds ?? [...contractInheritance.keys()], committedFacts: committed }).filter(f => f.kind === "data-flow");
2305
+ if (missingData.length)
2306
+ return planToolReceipt({ ok: false, kind: "finalize_plan", code: "PLAN_DATA_FLOW_INCOMPLETE", error: missingData.map(f => f.reason).join("; ") });
1639
2307
  const fragment = assemblePlanPatchFromCommittedFacts(committed, contractInheritance) ?? {};
1640
2308
  const patch = {
1641
2309
  ...fragment,
@@ -1646,6 +2314,29 @@ export async function createFrontendPlanLedgerTools(input) {
1646
2314
  ? { realIntegrationGap: params.realIntegrationGap }
1647
2315
  : {}),
1648
2316
  };
2317
+ const latestRegistry = [...committed]
2318
+ .reverse()
2319
+ .map((record) => record.fact)
2320
+ .find((fact) => isRecordObject(fact) && fact.kind === "state-registry");
2321
+ if (latestRegistry) {
2322
+ const registryUiStates = new Set(stringList(latestRegistry.uiStateNames));
2323
+ const registryInteractions = new Set(stringList(latestRegistry.interactionNames));
2324
+ const live = collectCanonicalStateFlowNames(committed);
2325
+ const missingUiStates = [...registryUiStates].filter((name) => !live.uiStateNames.has(name));
2326
+ const missingInteractions = [...registryInteractions].filter((name) => !live.interactionNames.has(name));
2327
+ const undeclaredUiStates = [...live.uiStateNames].filter((name) => !registryUiStates.has(name));
2328
+ const undeclaredInteractions = [...live.interactionNames].filter((name) => !registryInteractions.has(name));
2329
+ if (missingUiStates.length > 0 ||
2330
+ missingInteractions.length > 0 ||
2331
+ undeclaredUiStates.length > 0 ||
2332
+ undeclaredInteractions.length > 0) {
2333
+ return planToolReceipt({
2334
+ ok: false,
2335
+ kind: "finalize_plan",
2336
+ error: `finalize_plan UX registry mismatch: every registered name must have one live state-flow entry and every live entry must be registered (missing uiStates: ${missingUiStates.join(", ") || "none"}; missing interactions: ${missingInteractions.join(", ") || "none"}; undeclared uiStates: ${undeclaredUiStates.join(", ") || "none"}; undeclared interactions: ${undeclaredInteractions.join(", ") || "none"})`,
2337
+ });
2338
+ }
2339
+ }
1649
2340
  // Front-load the node's compile + policy gates into the finalize
1650
2341
  // receipt (same pipeline the design-policy shell and the node
1651
2342
  // self-check run: merge the patch onto the runtime skeleton,
@@ -1672,9 +2363,10 @@ export async function createFrontendPlanLedgerTools(input) {
1672
2363
  catch (error) {
1673
2364
  if (error instanceof PlanPolicyPrecheckFailure) {
1674
2365
  // Template the fix: every uncovered interaction / state
1675
- // maps to a ready-to-submit record_component_choice
1676
- // call. One reuse-existing choice covers all
1677
- // behavioural interactions.
2366
+ // maps to a record_component_choice skeleton. Reuse is
2367
+ // never implied: the operator/model must fill an existing
2368
+ // evidencePath or switch the choice to decision=new with
2369
+ // the required frozen source citation.
1678
2370
  const suggestions = error.findings
1679
2371
  .filter((finding) => finding.code === "ui-design-coverage-missing" &&
1680
2372
  finding.path)
@@ -1685,6 +2377,7 @@ export async function createFrontendPlanLedgerTools(input) {
1685
2377
  purpose: finding.path,
1686
2378
  component: "<name the existing or new component>",
1687
2379
  decision: "reuse-existing",
2380
+ evidencePath: "<existing repo file that proves this reuse>",
1688
2381
  },
1689
2382
  },
1690
2383
  }));
@@ -1766,30 +2459,164 @@ export async function createFrontendPlanLedgerTools(input) {
1766
2459
  }
1767
2460
  },
1768
2461
  });
1769
- return {
1770
- customTools: [
2462
+ const readPlanFactsTool = defineTool({
2463
+ name: "read_plan_facts",
2464
+ label: "read_plan_facts",
2465
+ description: "Read the LIVE committed plan ledger without repository access. Filter by kind and optional entryId or eventId. Use field (dot path, e.g. entry.requirementIds or uiStates) to page large arrays/strings. offset/limit paginate results; follow nextOffset until null before replacing any full-array fact. expectedRevision pins all pages to one ledger revision; restart if it changes. No writes.",
2466
+ parameters: Type.Object({
2467
+ kind: Type.String(), entryId: Type.Optional(Type.String()), eventId: Type.Optional(Type.String()),
2468
+ field: Type.Optional(Type.String()), offset: Type.Optional(Type.Integer({ minimum: 0 })),
2469
+ limit: Type.Optional(Type.Integer({ minimum: 1, maximum: 50 })),
2470
+ stringOffset: Type.Optional(Type.Integer({ minimum: 0 })),
2471
+ expectedRevision: Type.Optional(Type.Integer({ minimum: 0 })),
2472
+ }, { additionalProperties: false }),
2473
+ async execute(_id, params) {
2474
+ const reply = (details) => ({ content: [{ type: "text", text: JSON.stringify(details) }], details });
2475
+ if (params.expectedRevision !== undefined && params.expectedRevision !== store.revision) {
2476
+ return reply({ ok: false, revision: store.revision, error: "ledger revision changed; restart pagination" });
2477
+ }
2478
+ let records = readCommittedEvents(store, attemptId).filter(record => {
2479
+ const fact = record.fact;
2480
+ return fact.kind === params.kind && (!params.eventId || record.eventId === params.eventId) &&
2481
+ (!params.entryId || fact.entry?.id === params.entryId);
2482
+ });
2483
+ // Full-entry corrections and registries are last-wins. Other kinds
2484
+ // remain an ordered event log (state removals must remain visible).
2485
+ if (params.entryId || params.kind === "state-registry")
2486
+ records = records.slice(-1);
2487
+ let values = records.map(record => ({ eventId: record.eventId, ...record.fact }));
2488
+ if (params.field) {
2489
+ values = records.flatMap(record => {
2490
+ let value = record.fact;
2491
+ for (const key of params.field.split(".")) {
2492
+ if (!value || typeof value !== "object" || !Object.hasOwn(value, key))
2493
+ return [];
2494
+ value = value[key];
2495
+ }
2496
+ return Array.isArray(value) ? value : [value];
2497
+ });
2498
+ }
2499
+ const offset = params.offset ?? 0;
2500
+ if (params.stringOffset !== undefined) {
2501
+ const value = values[offset];
2502
+ if (typeof value !== "string")
2503
+ return reply({ ok: false, error: "stringOffset requires a string field item" });
2504
+ // Unicode code points prevent a page boundary splitting a surrogate pair.
2505
+ const characters = Array.from(value);
2506
+ const chunk = characters.slice(params.stringOffset, params.stringOffset + 1000).join("");
2507
+ const next = params.stringOffset + Array.from(chunk).length;
2508
+ return reply({ ok: true, revision: store.revision, offset, stringOffset: params.stringOffset,
2509
+ chunk, totalCharacters: characters.length, nextStringOffset: next < characters.length ? next : null });
2510
+ }
2511
+ const items = [];
2512
+ let bytes = 0;
2513
+ for (const value of values.slice(offset, offset + (params.limit ?? 10))) {
2514
+ const size = Buffer.byteLength(JSON.stringify(value));
2515
+ if (bytes + size > 12_000)
2516
+ break;
2517
+ bytes += size;
2518
+ items.push(value);
2519
+ }
2520
+ if (!items.length && offset < values.length)
2521
+ return reply({
2522
+ ok: false, revision: store.revision, total: values.length, offset,
2523
+ error: "item exceeds the page budget; select an eventId/entryId and a narrower field path; for a string item set stringOffset: 0 and follow nextStringOffset",
2524
+ fields: typeof values[offset] === "object" && values[offset] !== null ? Object.keys(values[offset]) : [],
2525
+ });
2526
+ return reply({ ok: true, revision: store.revision, total: values.length, offset, items,
2527
+ nextOffset: offset + items.length < values.length ? offset + items.length : null });
2528
+ },
2529
+ });
2530
+ const durable = await createDurableFrontendTools({
2531
+ file: path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), attemptId, store: input.store,
2532
+ setWorkingStore: next => { store = next; },
2533
+ binding: { sourceBinding: input.sourceBinding, skeleton: input.skeleton, requirementIds: input.requirementIds, writeSet: input.writeSetPatterns, declaredUiStateIds: input.declaredUiStateIds, citations: input.componentNewSourceReferences, contractInheritance },
2534
+ tools: [
1771
2535
  recordRouteSelectionTool,
1772
2536
  recordComponentChoiceTool,
2537
+ recordStateRegistryTool,
1773
2538
  recordStateFlowTool,
1774
2539
  recordDataFlowTool,
1775
2540
  recordMockApiTool,
2541
+ recordMockEndpointTool,
1776
2542
  recordDesignDeviationTool,
1777
2543
  recordDependencyTool,
1778
2544
  recordPlanRequirementTool,
2545
+ recordPlanGroupCoverageTool,
1779
2546
  recordPlanVerificationTargetTool,
1780
2547
  recordPlanEvidenceGapTool,
1781
2548
  adoptStagedFactTool,
1782
2549
  finalizePlanTool,
1783
2550
  ],
1784
- setActiveRequirementScope: (requirementIds) => {
1785
- activeRequirementScope = [
1786
- ...new Set(requirementIds.filter((id) => id.trim().length > 0)),
1787
- ];
1788
- },
1789
- flush: async () => {
1790
- const committed = readCommittedEvents(store, attemptId);
1791
- await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
2551
+ });
2552
+ return {
2553
+ customTools: [...durable.customTools, { ...readPlanFactsTool, execute: async (...args) => { await durable.flush(); return readPlanFactsTool.execute(...args); } }],
2554
+ adoptCommittedFacts: async (records) => durable.commitExternal(async () => {
2555
+ for (const record of records) {
2556
+ if (record.phase !== "committed")
2557
+ continue;
2558
+ const fact = record.fact;
2559
+ if (!fact || typeof fact !== "object" || Array.isArray(fact))
2560
+ continue;
2561
+ const kind = typeof fact.kind === "string"
2562
+ ? fact.kind
2563
+ : "plan-fact";
2564
+ const entry = fact.entry;
2565
+ const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
2566
+ typeof entry?.id === "string"
2567
+ ? `${kind}:${entry.id}`
2568
+ : undefined;
2569
+ const sourceShardPrefix = `${attemptId}:parallel-merge:${record.attemptId}:`;
2570
+ const requestId = `${sourceShardPrefix}${record.eventId}`;
2571
+ const committed = readCommittedEvents(store, attemptId);
2572
+ const replay = committed.find((candidate) => candidate.requestId === requestId);
2573
+ if (replay) {
2574
+ if (replay.payloadSha256 !== record.payloadSha256) {
2575
+ throw new Error(`frontend plan shard fact merge replay conflict: source event ${record.eventId} changed payload`);
2576
+ }
2577
+ continue;
2578
+ }
2579
+ const explicitReplacement = typeof fact.replaces === "string" &&
2580
+ fact.replaces === entry?.id;
2581
+ const conflicting = committed.find((candidate) => {
2582
+ const candidateFact = candidate.fact;
2583
+ const candidateEntry = candidateFact.entry;
2584
+ const sameSourceShard = candidate.requestId.startsWith(sourceShardPrefix);
2585
+ return (candidateFact.kind === kind &&
2586
+ typeof candidateEntry?.id === "string" &&
2587
+ `${kind}:${candidateEntry.id}` === identity &&
2588
+ !(sameSourceShard && explicitReplacement));
2589
+ });
2590
+ if (conflicting) {
2591
+ // Two shards may honestly emit the same fact (e.g. both
2592
+ // derive the same frozen verification target). Identical
2593
+ // payloads are duplicates to skip; divergent payloads are
2594
+ // a real conflict the ladder must resolve.
2595
+ if (conflicting.payloadSha256 === record.payloadSha256) {
2596
+ continue;
2597
+ }
2598
+ const fromSameSourceShard = conflicting.requestId.startsWith(sourceShardPrefix);
2599
+ if (fromSameSourceShard) {
2600
+ throw new Error(`frontend plan shard fact merge conflict: ${identity} was already committed by another shard with a different payload`);
2601
+ }
2602
+ // Divergent payload from a different source attempt means a
2603
+ // ladder retry re-derived the coverage phase: the fresh
2604
+ // derivation supersedes the stored record.
2605
+ const storeWithRecords = store;
2606
+ storeWithRecords.records = storeWithRecords.records.filter((candidate) => candidate.eventId !== conflicting.eventId);
2607
+ }
2608
+ const result = await adoptPlanFact(kind, requestId, fact);
2609
+ if (!result.ok) {
2610
+ throw new Error(`frontend plan shard fact merge failed: ${result.error}`);
2611
+ }
2612
+ }
2613
+ }),
2614
+ setActiveRequirementScope: (requirementIds) => {
2615
+ activeRequirementScope = [
2616
+ ...new Set(requirementIds.filter((id) => id.trim().length > 0)),
2617
+ ];
1792
2618
  },
2619
+ flush: durable.flush,
1793
2620
  committedFactCount: () => readCommittedEvents(store, attemptId).length,
1794
2621
  committedRequirementIds: () => {
1795
2622
  const ids = new Set();
@@ -1819,6 +2646,10 @@ function contractFactSourceSpan(fact) {
1819
2646
  if (Array.isArray(fact.sourceRefs) && fact.sourceRefs.length > 0) {
1820
2647
  return fact.sourceRefs;
1821
2648
  }
2649
+ if (Array.isArray(fact.sourceFragmentIds) &&
2650
+ fact.sourceFragmentIds.some((value) => typeof value === "string" && value.trim().length > 0)) {
2651
+ return fact.sourceFragmentIds;
2652
+ }
1822
2653
  if (typeof fact.source === "string" && fact.source.trim().length > 0) {
1823
2654
  return fact.source;
1824
2655
  }
@@ -1868,6 +2699,7 @@ export function buildFrontendTaskContractVNext(records) {
1868
2699
  requirements,
1869
2700
  constraints: byKind("constraint"),
1870
2701
  evidenceExpectations: byKind("evidence-expectation"),
2702
+ deliverableDeclarations: byKind("required-deliverables").at(-1)?.items ?? [],
1871
2703
  handoffIntents: byKind("handoff-intent"),
1872
2704
  openQuestions: byKind("open-question"),
1873
2705
  splitProposals: byKind("split-proposal"),
@@ -1875,7 +2707,7 @@ export function buildFrontendTaskContractVNext(records) {
1875
2707
  };
1876
2708
  }
1877
2709
  /**
1878
- * A+B: `frontend-contract-pi` incremental contract tools. Six `record_*` tools
2710
+ * A+B: `frontend-contract-pi` incremental contract tools. The `record_*` tools
1879
2711
  * commit origin=contract facts and `finalize_contract` commits the terminal
1880
2712
  * disposition (ready | ready-with-assumptions | blocked + blockingOwner).
1881
2713
  */
@@ -1886,8 +2718,11 @@ export async function createFrontendContractTools(input) {
1886
2718
  ]);
1887
2719
  const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
1888
2720
  const { adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
1889
- const store = input.store;
2721
+ let store = input.store;
1890
2722
  const attemptId = input.attemptId;
2723
+ let activeScope = null;
2724
+ const committedRequirementIds = () => new Set(readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "requirement").map(r => String(r.fact.id)));
2725
+ const completedScopeRequirementIds = () => new Set(readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "contract-scope-completed").flatMap(r => Array.isArray(r.fact.requirementIds) ? r.fact.requirementIds.filter((id) => typeof id === "string") : []));
1891
2726
  const receipt = (details) => ({
1892
2727
  content: [{ type: "text", text: JSON.stringify(details) }],
1893
2728
  details,
@@ -1927,9 +2762,7 @@ export async function createFrontendContractTools(input) {
1927
2762
  }
1928
2763
  }
1929
2764
  const recordKinds = {
1930
- record_requirement: "requirement",
1931
2765
  record_constraint: "constraint",
1932
- record_evidence_expectation: "evidence-expectation",
1933
2766
  record_handoff_intent: "handoff-intent",
1934
2767
  record_open_question: "open-question",
1935
2768
  record_split_proposal: "split-proposal",
@@ -1939,16 +2772,176 @@ export async function createFrontendContractTools(input) {
1939
2772
  label: name,
1940
2773
  description: `Commit an origin=contract ${kind} fact. IMPORTANT: submit incrementally — batch up to 5 record_* calls per message, starting from the FIRST message; never attempt to emit the whole contract in one response (a single large dump will be truncated and rejected). Every message must make progress by committing at least one record_* fact.`,
1941
2774
  promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
1942
- parameters: Type.Object({}, { additionalProperties: true }),
2775
+ parameters: Type.Object({
2776
+ text: Type.String({ minLength: 1 }),
2777
+ requirementIds: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
2778
+ sourceFragmentIds: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
2779
+ ...(kind === "constraint" ? { category: Type.Optional(Type.Enum({ constraint: "constraint", "non-goal": "non-goal", risk: "risk" })) } : {}),
2780
+ ...(kind === "handoff-intent" ? { taskKind: Type.Literal("frontend-test"), blocking: Type.Optional(Type.Boolean()) } : {}),
2781
+ }, { additionalProperties: false }),
1943
2782
  async execute(_toolCallId, params) {
2783
+ const data = params;
2784
+ if (!data.text?.trim())
2785
+ return receipt({ ok: false, code: "TOOL_SCHEMA_INVALID", error: `${name}: text must be non-empty` });
2786
+ const knownFragments = new Set([...(input.canonicalRequirements?.values() ?? [])].flatMap(r => r.sourceFragmentIds));
2787
+ if (data.requirementIds?.some(id => !input.canonicalRequirements?.has(id)) || data.sourceFragmentIds?.some(id => !knownFragments.has(id)))
2788
+ return receipt({ ok: false, code: "CONTRACT_REFERENCE_UNKNOWN", error: `${name}: reference is outside the frozen source inventory` });
1944
2789
  const result = await adoptContractFact(kind, {
2790
+ ...(params ?? {}),
1945
2791
  kind,
1946
2792
  origin: "contract",
1947
- ...(params ?? {}),
1948
2793
  });
1949
2794
  return receipt(result);
1950
2795
  },
1951
2796
  }));
2797
+ // record_requirement is runtime-owned by design: the requirement TEXT and
2798
+ // sourceFragmentIds come from the frozen source-fidelity ledger, never from
2799
+ // model-authored prose. The model only names the canonical id it confirms.
2800
+ // This closes the free-shape hole (additionalProperties:true accepted
2801
+ // `statement` rewrites and JSON-stringified `sourceFragmentIds` arrays,
2802
+ // which then reached the planner as empty text + dead fragment bindings).
2803
+ const recordRequirementTool = defineTool({
2804
+ name: "record_requirement",
2805
+ label: "record_requirement",
2806
+ description: 'Confirm one canonical ledger requirement (origin=contract requirement fact). Pass the canonical id and optional execution:{groupId,kind,summary} only. Reuse a group only when its behavior and all permission/threshold/error conditions agree; retain separate groups for differences. Constraints/exclusions do not require invented UI. The canonical id is listed in the <frontend_contract_input> inventory, e.g. {"id": "AC-001"} — the runtime commits the authoritative text and sourceFragmentIds from the frozen ledger. Never pass text/statement/sourceFragmentIds yourself: free-form rewrites and JSON-stringified fragment arrays are rejected. IMPORTANT: batch up to 5 record_* calls per message, starting from the FIRST message.',
2807
+ promptSnippet: "Confirm 1-5 canonical requirements by id (up to 5 per message).",
2808
+ parameters: Type.Object({ id: Type.String({ description: "Canonical ledger requirement id (e.g. AC-001)" }), execution: Type.Optional(Type.Object({ groupId: Type.String({ minLength: 1 }), kind: Type.Union([Type.Literal("behavior"), Type.Literal("constraint"), Type.Literal("exclusion")]), summary: Type.String({ minLength: 1 }) }, { additionalProperties: false })) }, { additionalProperties: false }),
2809
+ async execute(_toolCallId, params) {
2810
+ const id = typeof params?.id === "string" ? params.id.trim() : "";
2811
+ if (!id) {
2812
+ return receipt({
2813
+ ok: false,
2814
+ kind: "requirement",
2815
+ error: "record_requirement requires the canonical requirement id",
2816
+ });
2817
+ }
2818
+ if (activeScope && !activeScope.has(id))
2819
+ return receipt({ ok: false, code: "FRONTEND_INPUT_SCOPE_VIOLATION", error: `requirement ${id} is outside the complete input scope of this session` });
2820
+ const canonical = input.canonicalRequirements?.get(id);
2821
+ if (!canonical) {
2822
+ const known = [...(input.canonicalRequirements?.keys() ?? [])];
2823
+ return receipt({
2824
+ ok: false,
2825
+ kind: "requirement",
2826
+ error: `record_requirement id "${id}" is not a canonical ledger requirement; canonical ids are: ${known.join(", ") || "(none)"}`,
2827
+ });
2828
+ }
2829
+ if (params.execution) {
2830
+ try {
2831
+ collectFrontendExecutionGroups([...readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "requirement" && r.fact.id !== id).map(r => ({ id: String(r.fact.id), execution: r.fact.execution })), { id, execution: params.execution }]);
2832
+ }
2833
+ catch (error) {
2834
+ return receipt({ ok: false, code: "EXECUTION_GROUP_CONFLICT", error: String(error) });
2835
+ }
2836
+ }
2837
+ const result = await adoptContractFact("requirement", {
2838
+ kind: "requirement",
2839
+ origin: "contract",
2840
+ disposition: "explicit",
2841
+ id,
2842
+ text: canonical.text,
2843
+ sourceFragmentIds: canonical.sourceFragmentIds,
2844
+ ...(params.execution ? { execution: params.execution } : {}),
2845
+ });
2846
+ return receipt(result);
2847
+ },
2848
+ });
2849
+ // Authoritative UI state declarations: the contract node extracts the
2850
+ // PRD/reference UI-state table into structured facts so the planner binds
2851
+ // uiStates to declared ids instead of inventing names (dogfood
2852
+ // dag-1788504923861-0f7b17a9: 10 invented list-visibility variants, 0 of
2853
+ // the 7 PRD states).
2854
+ const recordUiStateTool = defineTool({
2855
+ name: "record_ui_state",
2856
+ label: "record_ui_state",
2857
+ description: 'Declare ONE authoritative UI state extracted from the task source\'s UI-state table (origin=contract ui-state-declaration fact). Call once per declared state, exactly as the source names it: {"id": "<state id from the source table>", "trigger": "<when this state applies>", "observableOutcome": "<what the user can observe>"}. The planner must bind these ids later; do not rename or invent states.',
2858
+ promptSnippet: "Declare 1-5 authoritative UI states (up to 5 per message).",
2859
+ parameters: Type.Object({
2860
+ id: Type.String({ description: "UI state id exactly as the source table declares it" }),
2861
+ trigger: Type.String({ description: "When this state applies" }),
2862
+ observableOutcome: Type.String({ description: "Observable result for the user" }),
2863
+ }, { additionalProperties: false }),
2864
+ async execute(_toolCallId, params) {
2865
+ const id = typeof params?.id === "string" ? params.id.trim() : "";
2866
+ const trigger = typeof params?.trigger === "string" ? params.trigger.trim() : "";
2867
+ const observableOutcome = typeof params?.observableOutcome === "string"
2868
+ ? params.observableOutcome.trim()
2869
+ : "";
2870
+ if (!id || !trigger || !observableOutcome) {
2871
+ return receipt({
2872
+ ok: false,
2873
+ kind: "ui-state-declaration",
2874
+ error: "record_ui_state requires non-empty id, trigger, and observableOutcome",
2875
+ });
2876
+ }
2877
+ const result = await adoptContractFact("ui-state-declaration", {
2878
+ kind: "ui-state-declaration",
2879
+ origin: "contract",
2880
+ id,
2881
+ trigger,
2882
+ observableOutcome,
2883
+ });
2884
+ return receipt(result);
2885
+ },
2886
+ });
2887
+ const evidenceStatus = Type.Enum({
2888
+ required: "required", optional: "optional", "not-applicable": "not-applicable",
2889
+ });
2890
+ const recordEvidenceExpectationTool = defineTool({
2891
+ name: "record_evidence_expectation",
2892
+ label: "record_evidence_expectation",
2893
+ description: 'Record evidence requirements as {requirementId:"<canonical id>",evidence:{static:"required|optional|not-applicable",behavior:"required|optional|not-applicable",mock:"required|optional|not-applicable","real-integration":"required|optional|not-applicable"}}. Declare at least one lane; later calls replace only the lanes they name.',
2894
+ parameters: Type.Object({
2895
+ requirementId: Type.String(),
2896
+ evidence: Type.Object({
2897
+ static: Type.Optional(evidenceStatus),
2898
+ behavior: Type.Optional(evidenceStatus),
2899
+ mock: Type.Optional(evidenceStatus),
2900
+ "real-integration": Type.Optional(evidenceStatus),
2901
+ }, { additionalProperties: false }),
2902
+ }, { additionalProperties: false }),
2903
+ async execute(_toolCallId, params) {
2904
+ try {
2905
+ const fact = frontendEvidenceExpectationSchema.parse(params);
2906
+ if (!input.canonicalRequirements?.has(fact.requirementId)) {
2907
+ throw new Error(`unknown canonical requirement ${fact.requirementId}`);
2908
+ }
2909
+ return receipt(await adoptContractFact("evidence-expectation", {
2910
+ kind: "evidence-expectation", origin: "contract", ...fact,
2911
+ }));
2912
+ }
2913
+ catch (error) {
2914
+ return receipt({
2915
+ ok: false, kind: "evidence-expectation",
2916
+ error: error instanceof Error ? error.message : String(error),
2917
+ });
2918
+ }
2919
+ },
2920
+ });
2921
+ const recordRequiredDeliverablesTool = defineTool({
2922
+ name: "record_required_deliverables",
2923
+ label: "record_required_deliverables",
2924
+ description: 'Declare the complete source-required file deliverables as {items:[{path,requirementId,sourceFragmentId}]}. Interpret obligations from the original source, including lists/tables: permissions (allowedPaths/only allowed to modify), prohibitions, examples and read-only references are NOT delivery obligations. Each path must appear exactly in its frozen requirement-bound source fragment. Submit {items:[]} explicitly if no files are mandatory. Each call appends complete source-bound items; replace:true explicitly replaces the inventory. Required before finalize_contract ready.',
2925
+ parameters: Type.Object({ replace: Type.Optional(Type.Boolean()), items: Type.Array(Type.Object({
2926
+ path: Type.String(), requirementId: Type.String(), sourceFragmentId: Type.String(),
2927
+ }, { additionalProperties: false })) }, { additionalProperties: false }),
2928
+ async execute(_toolCallId, params) {
2929
+ try {
2930
+ const declaration = validateFrontendRequiredDeliverables({ items: params.items }, input.canonicalRequirements ?? new Map());
2931
+ const previous = params.replace ? [] : readCommittedEvents(store, attemptId).filter(r => r.fact.kind === "required-deliverables").at(-1)?.fact.items;
2932
+ const items = [...new Map([...(Array.isArray(previous) ? previous : []), ...declaration.items].map(item => [JSON.stringify(item), item])).values()];
2933
+ return receipt(await adoptContractFact("required-deliverables", {
2934
+ kind: "required-deliverables", origin: "contract", items,
2935
+ }));
2936
+ }
2937
+ catch (error) {
2938
+ return receipt({
2939
+ ok: false, kind: "required-deliverables",
2940
+ error: error instanceof Error ? error.message : String(error),
2941
+ });
2942
+ }
2943
+ },
2944
+ });
1952
2945
  // OpenSpec selection committed as individual typed facts (one path per
1953
2946
  // call) so a large candidate set never exceeds a single model output
1954
2947
  // budget: each tool call carries exactly one {path, disposition,
@@ -1994,7 +2987,7 @@ export async function createFrontendContractTools(input) {
1994
2987
  const finalizeContractTool = defineTool({
1995
2988
  name: "finalize_contract",
1996
2989
  label: "finalize_contract",
1997
- description: "Commit the contract-finalized terminal fact with a disposition of ready | ready-with-assumptions | blocked (blocked requires blockingOwner). Call exactly once.",
2990
+ description: "Commit the contract-finalized terminal fact with a disposition of ready | ready-with-assumptions | blocked (blocked requires blockingOwner). A successful terminal commit occurs exactly once. If validation fails, correct only the reported facts and retry finalize.",
1998
2991
  promptSnippet: "Commit the contract-finalized terminal (disposition + optional blockingOwner).",
1999
2992
  parameters: Type.Object({
2000
2993
  disposition: Type.Enum({
@@ -2019,6 +3012,16 @@ export async function createFrontendContractTools(input) {
2019
3012
  error: "blocked disposition requires blockingOwner (blocked-human | blocked-external)",
2020
3013
  });
2021
3014
  }
3015
+ if (disposition !== "blocked" && input.canonicalRequirements?.size &&
3016
+ !readCommittedEvents(store, attemptId).some((record) => record.fact.kind === "required-deliverables")) {
3017
+ return receipt({
3018
+ ok: false, kind: "finalize_contract",
3019
+ error: "call record_required_deliverables with the complete source-bound inventory (or items:[] when none) before finalizing",
3020
+ });
3021
+ }
3022
+ const missing = [...(input.canonicalRequirements?.keys() ?? [])].filter(id => !committedRequirementIds().has(id) || (activeScope !== null && !completedScopeRequirementIds().has(id)));
3023
+ if (disposition !== "blocked" && missing.length)
3024
+ return receipt({ ok: false, code: "CONTRACT_REQUIREMENT_COVERAGE_MISSING", error: `Confirm all complete source obligations before finalizing: ${missing.join(", ")}` });
2022
3025
  const blockedOwner = mapContractBlockedOwner({
2023
3026
  disposition: disposition ?? "",
2024
3027
  blockingOwner,
@@ -2039,15 +3042,41 @@ export async function createFrontendContractTools(input) {
2039
3042
  return receipt(result);
2040
3043
  },
2041
3044
  });
2042
- return {
2043
- customTools: [
3045
+ const completeScopeTool = defineTool({
3046
+ name: "complete_contract_scope", label: "complete_contract_scope",
3047
+ description: "After recording all requirements AND their evidence, constraints, questions and deliverables for this session, mark the scope complete. Confirming an ID alone does not complete its analysis. Do this before finalize_contract.",
3048
+ parameters: Type.Object({ requirementIds: Type.Array(Type.String({ minLength: 1 }), { uniqueItems: true }) }, { additionalProperties: false }),
3049
+ async execute(_id, params) {
3050
+ const ids = params.requirementIds;
3051
+ if (activeScope === null || ids.length !== activeScope.size || ids.some(id => !activeScope?.has(id) || !committedRequirementIds().has(id)))
3052
+ return receipt({ ok: false, code: "CONTRACT_SCOPE_INCOMPLETE", error: "Complete exactly the active scope after confirming all its obligations" });
3053
+ return receipt(await adoptContractFact("contract-scope-completed", { kind: "contract-scope-completed", origin: "contract", requirementIds: ids }));
3054
+ },
3055
+ });
3056
+ const durable = await createDurableFrontendTools({
3057
+ file: path.join(input.runDir, input.nodeId, "contract-typed-facts.jsonl"), attemptId, store: input.store,
3058
+ setWorkingStore: next => { store = next; }, binding: { canonicalRequirements: input.canonicalRequirements, sourceDigest: input.sourceDigest },
3059
+ tools: [
2044
3060
  ...recordTools,
3061
+ recordRequirementTool,
3062
+ recordEvidenceExpectationTool,
3063
+ recordUiStateTool,
3064
+ recordRequiredDeliverablesTool,
2045
3065
  recordOpenspecSelectionTool,
3066
+ completeScopeTool,
2046
3067
  finalizeContractTool,
2047
3068
  ],
3069
+ });
3070
+ return {
3071
+ customTools: durable.customTools,
3072
+ inputRequirements: () => [...(input.canonicalRequirements ?? [])].map(([id, value]) => ({ id, text: value.text, sourceFragmentIds: [...value.sourceFragmentIds] })),
3073
+ completedScopeRequirementIds,
3074
+ setActiveRequirementScope: ids => { activeScope = ids === null ? null : new Set(ids); },
3075
+ committedRequirementIds,
3076
+ committedFacts: () => readCommittedEvents(input.store, attemptId),
2048
3077
  flush: async () => {
2049
- const committed = readCommittedEvents(store, attemptId);
2050
- await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "contract-typed-facts.jsonl"), committed);
3078
+ await durable.flush();
3079
+ const committed = readCommittedEvents(input.store, attemptId);
2051
3080
  await writeJsonAtomic(path.join(input.runDir, input.nodeId, "frontend-task-contract-vNext.json"), buildFrontendTaskContractVNext(committed));
2052
3081
  },
2053
3082
  };
@@ -2099,7 +3128,7 @@ export async function createFrontendScoutEvidenceTools(input) {
2099
3128
  ]);
2100
3129
  const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
2101
3130
  const { adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
2102
- const store = input.store;
3131
+ let store = input.store;
2103
3132
  const attemptId = input.attemptId;
2104
3133
  const stringArray = Type.Array(Type.String({}));
2105
3134
  const optionalString = Type.Optional(Type.String({}));
@@ -2113,6 +3142,18 @@ export async function createFrontendScoutEvidenceTools(input) {
2113
3142
  });
2114
3143
  const sourceDeclaredPaths = (input.sourceDeclaredPaths ?? []).map((value) => value.replaceAll("\\", "/").replace(/^\.\//, "").replace(/\/$/, ""));
2115
3144
  const hasSourceDeclarations = input.sourceDeclaredPaths !== undefined;
3145
+ let activeScope;
3146
+ const scopeIdentity = (ids) => createHash("sha256").update(JSON.stringify([...ids].sort())).digest("hex");
3147
+ const latestScopes = () => {
3148
+ const byRequirement = new Map();
3149
+ for (const record of readCommittedEvents(store, attemptId))
3150
+ if (record.fact.kind === "scout-scope" && Array.isArray(record.fact.requirementIds)) {
3151
+ for (const id of record.fact.requirementIds)
3152
+ if (typeof id === "string")
3153
+ byRequirement.set(id, record.fact);
3154
+ }
3155
+ return byRequirement;
3156
+ };
2116
3157
  const isSourceDeclared = (candidate) => {
2117
3158
  const normalized = candidate.replaceAll("\\", "/").replace(/^\.\//, "").replace(/\/$/, "");
2118
3159
  return sourceDeclaredPaths.some((declared) => declared === normalized || declared.startsWith(`${normalized}/`));
@@ -2205,6 +3246,7 @@ export async function createFrontendScoutEvidenceTools(input) {
2205
3246
  description: "Commit an origin=scout target-surface fact with complete/blocked discovery status. A complete surface needs a proven target path and no unresolved paths; blocked surfaces name the unresolved paths instead of guessing. Example: {\"completeness\": \"complete\", \"entrypoint\": \"<file>\", \"implementationPaths\": [\"<dir or file>\"], \"testPaths\": [\"<file>\"], \"allowedPathConflicts\": [], \"unresolvedPaths\": []}",
2206
3247
  promptSnippet: "Commit an origin=scout target-surface fact.",
2207
3248
  parameters: Type.Object({
3249
+ scopeId: Type.Optional(Type.String({ minLength: 1 })),
2208
3250
  completeness: scoutCompleteness,
2209
3251
  entrypoint: optionalString,
2210
3252
  routeOrMount: optionalString,
@@ -2215,6 +3257,8 @@ export async function createFrontendScoutEvidenceTools(input) {
2215
3257
  unresolvedPaths: stringArray,
2216
3258
  }, { additionalProperties: false }),
2217
3259
  async execute(_toolCallId, params) {
3260
+ if (input.requirementIds && (!activeScope?.length || params.scopeId !== scopeIdentity(activeScope)))
3261
+ return receipt({ ok: false, code: "FRONTEND_INPUT_SCOPE_VIOLATION", error: "Use exactly the runtime Scout scopeId; discovery may complete only the supplied obligations" });
2218
3262
  const implementationPaths = params?.implementationPaths ?? [];
2219
3263
  const testPaths = params?.testPaths ?? [];
2220
3264
  const pathEvidence = await enrichScoutPathEvidence([
@@ -2222,7 +3266,7 @@ export async function createFrontendScoutEvidenceTools(input) {
2222
3266
  ...implementationPaths,
2223
3267
  ...testPaths,
2224
3268
  ]);
2225
- const result = await adoptScoutFact("target-surface", {
3269
+ const surface = {
2226
3270
  kind: "target-surface",
2227
3271
  origin: "scout",
2228
3272
  completeness: params?.completeness ?? "blocked",
@@ -2235,7 +3279,27 @@ export async function createFrontendScoutEvidenceTools(input) {
2235
3279
  unresolvedPaths: params?.unresolvedPaths ?? [],
2236
3280
  ...(hasSourceDeclarations ? { sourceDeclaredPaths } : {}),
2237
3281
  ...(pathEvidence.length > 0 ? { pathEvidence } : {}),
2238
- });
3282
+ };
3283
+ if (activeScope) {
3284
+ const { readCompleteScoutTargetSurface } = await import("../workflows/dag/frontend-shadow-dual-write.js");
3285
+ if (surface.completeness === "complete") {
3286
+ const check = readCompleteScoutTargetSurface([{ phase: "committed", fact: surface }]);
3287
+ if (!check.ok)
3288
+ return receipt({ ok: false, code: "SCOUT_SCOPE_INCOMPLETE", error: check.reason });
3289
+ }
3290
+ const saved = await adoptScoutFact("scout-scope", { kind: "scout-scope", origin: "scout", id: params.scopeId, requirementIds: activeScope, surface });
3291
+ if (!saved.ok)
3292
+ return receipt(saved);
3293
+ const current = latestScopes();
3294
+ if (input.requirementIds?.every(id => current.get(id)?.surface?.completeness === "complete")) {
3295
+ const surfaces = [...new Set(input.requirementIds.map(id => current.get(id)))].map(f => f.surface);
3296
+ const union = (key) => [...new Set(surfaces.flatMap(s => Array.isArray(s[key]) ? s[key] : []))];
3297
+ const entries = [...new Set(surfaces.map(s => String(s.entrypoint ?? "")).filter(Boolean))];
3298
+ return receipt(await adoptScoutFact("target-surface", { kind: "target-surface", origin: "scout", completeness: "complete", entrypoint: entries[0] ?? "", implementationPaths: [...new Set([...entries, ...union("implementationPaths")])], testPaths: union("testPaths"), allowedPathConflicts: union("allowedPathConflicts"), unresolvedPaths: union("unresolvedPaths"), routeOrMount: [...new Set(surfaces.map(s => s.routeOrMount).filter(Boolean))].join("\n"), dataSource: [...new Set(surfaces.map(s => s.dataSource).filter(Boolean))].join("\n"), pathEvidence: surfaces.flatMap(s => s.pathEvidence ?? []), ...(hasSourceDeclarations ? { sourceDeclaredPaths } : {}) }));
3299
+ }
3300
+ return receipt(saved);
3301
+ }
3302
+ const result = await adoptScoutFact("target-surface", surface);
2239
3303
  return receipt(result);
2240
3304
  },
2241
3305
  });
@@ -2259,12 +3323,53 @@ export async function createFrontendScoutEvidenceTools(input) {
2259
3323
  return receipt(result);
2260
3324
  },
2261
3325
  });
3326
+ const durable = await createDurableFrontendTools({
3327
+ file: path.join(input.runDir, input.nodeId, "scout-typed-facts.jsonl"), attemptId, store: input.store,
3328
+ setWorkingStore: next => { store = next; }, binding: { requirementIds: input.requirementIds, sourceDeclaredPaths: input.sourceDeclaredPaths, sourceDigest: input.sourceDigest, workspaceRoot: input.workspaceRoot },
3329
+ tools: [recordTargetSurfaceTool, recordDesignEvidenceTool],
3330
+ validateRestored: async (records) => {
3331
+ if (!input.workspaceRoot)
3332
+ return;
3333
+ for (const record of records) {
3334
+ const evidence = record.fact.kind === "scout-scope" ? record.fact.surface?.pathEvidence : record.fact.pathEvidence;
3335
+ if (!Array.isArray(evidence))
3336
+ continue;
3337
+ for (const previous of evidence) {
3338
+ if (!isRecordObject(previous) || typeof previous.path !== "string")
3339
+ throw Error("scout path evidence is malformed");
3340
+ const current = (await enrichScoutPathEvidence([previous.path]))[0];
3341
+ if (!current || current.sha256 !== previous.sha256 || current.fresh !== previous.fresh)
3342
+ throw Error(`scout evidence drift: ${previous.path}; refresh Scout before reusing facts`);
3343
+ }
3344
+ }
3345
+ },
3346
+ });
2262
3347
  return {
2263
- customTools: [recordTargetSurfaceTool, recordDesignEvidenceTool],
2264
- flush: async () => {
2265
- const committed = readCommittedEvents(store, attemptId);
2266
- await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "scout-typed-facts.jsonl"), committed);
3348
+ ...durable,
3349
+ adoptCommittedFacts: async (records) => durable.commitExternal(async () => {
3350
+ for (const record of records) {
3351
+ if (record.phase !== "committed")
3352
+ continue;
3353
+ const fact = record.fact;
3354
+ if (!fact || typeof fact !== "object" || Array.isArray(fact))
3355
+ continue;
3356
+ const result = await adoptScoutFact(typeof fact.kind === "string"
3357
+ ? fact.kind
3358
+ : "scout-fact", fact);
3359
+ if (!result.ok) {
3360
+ throw new Error(`frontend scout shard fact merge failed: ${result.error}`);
3361
+ }
3362
+ }
3363
+ }),
3364
+ setActiveScope: ids => {
3365
+ if (!ids.length || ids.some(id => !input.requirementIds?.includes(id)))
3366
+ throw Error("FRONTEND_INPUT_SCOPE_VIOLATION");
3367
+ activeScope = [...new Set(ids)];
3368
+ return scopeIdentity(activeScope);
2267
3369
  },
3370
+ completedRequirementIds: () => new Set([...latestScopes()].filter(([, fact]) => fact.surface?.completeness === "complete").map(([id]) => id)),
3371
+ completedScopeFacts: () => [...new Set(latestScopes().values())].filter(f => f.surface?.completeness === "complete"),
3372
+ committedFacts: () => readCommittedEvents(store, attemptId),
2268
3373
  };
2269
3374
  }
2270
3375
  export function buildDagPiUserMessage(task, persona, step) {
@@ -2515,15 +3620,14 @@ async function runFrontendReviewTerminalShadow(input) {
2515
3620
  return input.mapped;
2516
3621
  const { compareTypedReviewToLegacyJsonVerdict } = await import("../workflows/dag/frontend-review-context.js");
2517
3622
  const { parseJsonReviewVerdict } = await import("../workflows/dag/output-protocol.js");
2518
- const sessionEventsPath = path.join(input.meta.runDir, input.task.id, "session-events.jsonl");
2519
3623
  let typedKinds = [];
2520
3624
  try {
2521
- const content = await readFile(sessionEventsPath, "utf8");
2522
- typedKinds = scanReviewTerminalKindsFromSessionEvents(content);
3625
+ await input.tools?.flush();
3626
+ input.tools?.scopeProtocol?.assertComplete();
3627
+ typedKinds = input.tools?.scopeProtocol?.committedFacts().filter(r => ["approve_review", "request_review_changes"].includes(String(r.fact.kind))).map(r => String(r.fact.kind)) ?? [];
2523
3628
  }
2524
- catch {
2525
- // Missing/unreadable session log → fail-closed at zero terminal facts.
2526
- typedKinds = [];
3629
+ catch (error) {
3630
+ return { ...input.mapped, ok: false, failureCategory: "frontend-ledger-invalid", stderr: `FRONTEND_REVIEW_SCOPE_INCOMPLETE: ${error instanceof Error ? error.message : String(error)}` };
2527
3631
  }
2528
3632
  let legacyVerdict;
2529
3633
  try {
@@ -2549,13 +3653,7 @@ async function runFrontendReviewTerminalShadow(input) {
2549
3653
  reason: `typed review equivalence comparison crashed: ${error instanceof Error ? error.message : String(error)}`,
2550
3654
  };
2551
3655
  }
2552
- // Audit-only flush + artifact. Neither blocks the node.
2553
- try {
2554
- await input.tools?.flush?.();
2555
- }
2556
- catch {
2557
- // best-effort
2558
- }
3656
+ // The durable ledger was validated above; this artifact is audit-only.
2559
3657
  try {
2560
3658
  await writeDagNodeJsonArtifact(input.meta.runDir, input.task.id, "fact-review-status.json", {
2561
3659
  schemaVersion: 1,
@@ -2597,23 +3695,16 @@ async function runFrontendDesignTerminalShadow(input) {
2597
3695
  // diagnosed as an omitted terminal tool call.
2598
3696
  if (!input.mapped.ok)
2599
3697
  return input.mapped;
2600
- const sessionEventsPath = path.join(input.meta.runDir, input.task.id, "session-events.jsonl");
2601
3698
  let typedKinds = [];
2602
3699
  try {
2603
- const content = await readFile(sessionEventsPath, "utf8");
2604
- typedKinds = scanDesignTerminalKindsFromSessionEvents(content);
2605
- }
2606
- catch {
2607
- // Missing/unreadable session log → fail-closed at zero terminal facts.
2608
- typedKinds = [];
2609
- }
2610
- // Audit-only flush + artifact. Neither blocks the node.
2611
- try {
2612
- await input.tools?.flush?.();
3700
+ await input.tools?.flush();
3701
+ input.tools?.scopeProtocol?.assertComplete();
3702
+ typedKinds = input.tools?.scopeProtocol?.committedFacts().filter(r => ["approve_design", "request_design_changes"].includes(String(r.fact.kind))).map(r => String(r.fact.kind)) ?? [];
2613
3703
  }
2614
- catch {
2615
- // best-effort
3704
+ catch (error) {
3705
+ return { ...input.mapped, ok: false, failureCategory: "frontend-ledger-invalid", stderr: `FRONTEND_REVIEW_SCOPE_INCOMPLETE: ${error instanceof Error ? error.message : String(error)}` };
2616
3706
  }
3707
+ // The durable ledger was validated above; this artifact is audit-only.
2617
3708
  try {
2618
3709
  await writeDagNodeJsonArtifact(input.meta.runDir, input.task.id, "fact-design-status.json", {
2619
3710
  schemaVersion: 1,
@@ -2646,6 +3737,7 @@ const FRONTEND_PLAN_SEGMENTS = [
2646
3737
  id: "coverage",
2647
3738
  toolNames: new Set([
2648
3739
  "record_plan_requirement",
3740
+ "record_plan_group_coverage",
2649
3741
  "record_plan_verification_target",
2650
3742
  "record_plan_evidence_gap",
2651
3743
  "adopt_staged_fact",
@@ -2653,19 +3745,31 @@ const FRONTEND_PLAN_SEGMENTS = [
2653
3745
  instruction: [
2654
3746
  "PLAN PHASE — requirement coverage only.",
2655
3747
  "Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target facts (verification target bound to requirement ids and files). Group related requirements under one non-static behavior target when one observable test behavior proves them together; do not mechanically create one target per requirement. A non-static target id is the stable machine trace token; never submit prose as a symbol. Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
3748
+ "A coverage session is complete only when EVERY requirement assigned to this session (the full inventory, or the exact COVERAGE BATCH / shard list when present) has committed coverage facts: a record_plan_requirement entry plus verification targets, or a committed evidence gap. Keep committing in batches of up to 4 record_* calls per assistant message until then; do not write a concluding summary while any assigned requirement is still uncommitted — an early stop strands the remainder into a MISSING-FACT repair session and doubles the sessions needed.",
2656
3749
  "If a requirement genuinely cannot have a verification target, record a non-empty record_plan_evidence_gap. Do not call finalize_plan; it is not available in this phase.",
2657
3750
  ].join(" "),
2658
3751
  },
3752
+ {
3753
+ id: "ux-registry",
3754
+ toolNames: new Set(["record_state_registry", "adopt_staged_fact"]),
3755
+ instruction: [
3756
+ "PLAN PHASE — global UX vocabulary.",
3757
+ "Bootstrap the global UX vocabulary from the execution-group index and authoritative declared states. This is navigation, not permission to decide unseen behavior. Detailed complete scopes may extend the registry with replace:true while preserving live names. UI states use declaredUiStates ids when present. Interaction names are stable kebab-case behavior domains; merge requirements that describe the same behavior instead of renaming it per AC slice. Empty arrays explicitly declare that no UX vocabulary applies. Do not record component choices or state-flow details in this phase.",
3758
+ "Do not call finalize_plan; it is not available in this phase.",
3759
+ ].join(" "),
3760
+ },
2659
3761
  {
2660
3762
  id: "ux-local",
2661
3763
  toolNames: new Set([
3764
+ "record_state_registry",
2662
3765
  "record_component_choice",
2663
3766
  "record_state_flow",
3767
+ "record_plan_verification_target",
2664
3768
  "adopt_staged_fact",
2665
3769
  ]),
2666
3770
  instruction: [
2667
- "PLAN PHASE — requirement-local UX decisions.",
2668
- "Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). For ONLY this requirement slice, record component choices and UI state flow. Keep these facts for this slice together. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
3771
+ "PLAN PHASE — global UX decisions.",
3772
+ "Requirements and verification targets are already committed in the ledger; do not re-record unchanged facts. Review the current complete execution-group scope and the committed global UX registry together, then record each component choice, UI state and interaction exactly once. Bind each applicable state to its verificationTargetIds; the runtime derives the reverse VT.uiStates relation. If a VT requires correction, record_plan_verification_target with replace:true is available after declaring its states; preserve its requirement coverage. Multiple requirements describing one behavior share one registry name and state-flow entry; never repeat or rename it per AC. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
2669
3773
  "Do not call finalize_plan; it is not available in this phase.",
2670
3774
  ].join(" "),
2671
3775
  },
@@ -2679,7 +3783,7 @@ const FRONTEND_PLAN_SEGMENTS = [
2679
3783
  },
2680
3784
  {
2681
3785
  id: "global-mock-data",
2682
- toolNames: new Set(["record_data_flow", "record_mock_api", "adopt_staged_fact"]),
3786
+ toolNames: new Set(["record_data_flow", "record_mock_api", "record_mock_endpoint", "adopt_staged_fact"]),
2683
3787
  instruction: [
2684
3788
  "PLAN PHASE — global Mock/API and data policy.",
2685
3789
  "Record the cross-cutting interaction-to-endpoint data flow and Mock/API strategy only. Keep this decision set separate from route, component, state, dependency, and deviation facts. Do not call finalize_plan.",
@@ -2702,196 +3806,6 @@ const FRONTEND_PLAN_SEGMENTS = [
2702
3806
  ].join(" "),
2703
3807
  },
2704
3808
  ];
2705
- function committedFactFromPlanRecord(value) {
2706
- if (!value || typeof value !== "object" || Array.isArray(value))
2707
- return undefined;
2708
- const record = value;
2709
- if (record.phase !== undefined && record.phase !== "committed")
2710
- return undefined;
2711
- const fact = record.fact;
2712
- return fact && typeof fact === "object" && !Array.isArray(fact)
2713
- ? fact
2714
- : typeof record.kind === "string"
2715
- ? record
2716
- : undefined;
2717
- }
2718
- function planFactStringList(value) {
2719
- if (!Array.isArray(value))
2720
- return [];
2721
- return value.filter((item) => typeof item === "string" && item.trim().length > 0);
2722
- }
2723
- function planFactScopeIntersects(fact, requirementIds) {
2724
- return planFactStringList(fact.scopeRequirementIds).some((id) => requirementIds.has(id));
2725
- }
2726
- /** Apply state-flow replacement/removal semantics exactly as the plan compiler does. */
2727
- function collectCanonicalStateFlowNames(committedFacts) {
2728
- const uiStateNames = new Set();
2729
- const interactionNames = new Set();
2730
- for (const value of committedFacts) {
2731
- const fact = committedFactFromPlanRecord(value);
2732
- if (!fact || fact.origin !== "plan" || fact.kind !== "state-flow")
2733
- continue;
2734
- for (const name of planFactStringList(fact.removeUiStateNames)) {
2735
- uiStateNames.delete(name);
2736
- }
2737
- for (const name of planFactStringList(fact.removeInteractionNames)) {
2738
- interactionNames.delete(name);
2739
- }
2740
- for (const state of Array.isArray(fact.uiStates) ? fact.uiStates : []) {
2741
- if (!state || typeof state !== "object" || Array.isArray(state))
2742
- continue;
2743
- const name = state.name;
2744
- if (typeof name === "string" && name.trim())
2745
- uiStateNames.add(name);
2746
- }
2747
- for (const interaction of Array.isArray(fact.interactions)
2748
- ? fact.interactions
2749
- : []) {
2750
- if (!interaction || typeof interaction !== "object" || Array.isArray(interaction)) {
2751
- continue;
2752
- }
2753
- const name = interaction.name;
2754
- if (typeof name === "string" && name.trim())
2755
- interactionNames.add(name);
2756
- }
2757
- }
2758
- return { uiStateNames, interactionNames };
2759
- }
2760
- /** Compute the authoritative coverage queue from the committed plan ledger. */
2761
- export function collectFrontendPlanMissingFacts(input) {
2762
- const requirements = new Map();
2763
- const standaloneEvidenceGaps = new Set();
2764
- const verificationTargetIds = new Set();
2765
- const verificationTargetRequirements = new Map();
2766
- for (const value of input.committedFacts) {
2767
- const fact = committedFactFromPlanRecord(value);
2768
- if (!fact || fact.origin !== "plan")
2769
- continue;
2770
- if (fact.kind === "plan-requirement" && fact.entry && typeof fact.entry === "object") {
2771
- const entry = fact.entry;
2772
- if (typeof entry.id === "string" && entry.id.trim())
2773
- requirements.set(entry.id, entry);
2774
- }
2775
- if (fact.kind === "plan-verification-target" && fact.entry && typeof fact.entry === "object") {
2776
- const entry = fact.entry;
2777
- const id = entry.id;
2778
- if (typeof id === "string" && id.trim()) {
2779
- verificationTargetIds.add(id);
2780
- verificationTargetRequirements.set(id, new Set(Array.isArray(entry.requirementIds)
2781
- ? entry.requirementIds.filter((value) => typeof value === "string")
2782
- : []));
2783
- }
2784
- }
2785
- if (fact.kind === "plan-evidence-gap" && fact.entry && typeof fact.entry === "object") {
2786
- const entry = fact.entry;
2787
- const requirementId = entry.requirementId;
2788
- const description = entry.description;
2789
- if (typeof requirementId === "string" && requirementId.trim() && typeof description === "string" && description.trim()) {
2790
- standaloneEvidenceGaps.add(requirementId);
2791
- }
2792
- }
2793
- }
2794
- const missing = [];
2795
- for (const id of input.requirementIds) {
2796
- const entry = requirements.get(id);
2797
- if (!entry) {
2798
- missing.push({
2799
- kind: "plan-requirement",
2800
- id,
2801
- requirementIds: [id],
2802
- reason: `requirement ${id} has no committed plan-requirement fact`,
2803
- });
2804
- continue;
2805
- }
2806
- const targetIds = Array.isArray(entry.verificationTargetIds)
2807
- ? entry.verificationTargetIds.filter((value) => typeof value === "string" && value.trim().length > 0)
2808
- : [];
2809
- const gap = entry.evidenceGap && typeof entry.evidenceGap === "object"
2810
- ? entry.evidenceGap
2811
- : undefined;
2812
- const hasEvidenceGap = (typeof gap?.description === "string" && gap.description.trim().length > 0) ||
2813
- standaloneEvidenceGaps.has(id);
2814
- if (targetIds.length === 0 && !hasEvidenceGap) {
2815
- missing.push({
2816
- kind: "plan-verification-target",
2817
- requirementIds: [id],
2818
- reason: `requirement ${id} declares neither a verification target nor a non-empty evidenceGap`,
2819
- });
2820
- continue;
2821
- }
2822
- for (const targetId of targetIds) {
2823
- if (!verificationTargetIds.has(targetId) ||
2824
- !verificationTargetRequirements.get(targetId)?.has(id)) {
2825
- missing.push({
2826
- kind: "plan-verification-target",
2827
- id: targetId,
2828
- requirementIds: [id],
2829
- reason: `requirement ${id} references verification target ${targetId}, but that target is not committed`,
2830
- });
2831
- }
2832
- }
2833
- }
2834
- return missing;
2835
- }
2836
- /** Completeness checks for phases whose facts are committed incrementally. */
2837
- export function collectFrontendPlanPhaseMissingFacts(input) {
2838
- const facts = input.committedFacts
2839
- .map(committedFactFromPlanRecord)
2840
- .filter((fact) => Boolean(fact && fact.origin === "plan"));
2841
- if (input.phase === "ux-local") {
2842
- const needsUx = input.requirementIds.some((id) => input.behaviorRequiredRequirementIds?.includes(id));
2843
- if (!needsUx)
2844
- return [];
2845
- const requirementSlice = new Set(input.requirementIds);
2846
- const scopedFacts = facts.filter((fact) => planFactScopeIntersects(fact, requirementSlice));
2847
- const hasChoice = scopedFacts.some((fact) => fact.kind === "component-choice" &&
2848
- Array.isArray(fact.uiComponentChoices) &&
2849
- fact.uiComponentChoices.length > 0);
2850
- const canonicalStateFlow = collectCanonicalStateFlowNames(scopedFacts);
2851
- const hasStateFlow = canonicalStateFlow.uiStateNames.size > 0 ||
2852
- canonicalStateFlow.interactionNames.size > 0;
2853
- const missing = [];
2854
- if (!hasChoice) {
2855
- missing.push({
2856
- kind: "component-choice",
2857
- requirementIds: [...input.requirementIds],
2858
- reason: "behaviour-required UX slice has no committed component-choice fact",
2859
- });
2860
- }
2861
- if (!hasStateFlow) {
2862
- missing.push({
2863
- kind: "state-flow",
2864
- requirementIds: [...input.requirementIds],
2865
- reason: "behaviour-required UX slice has no committed state-flow fact",
2866
- });
2867
- }
2868
- return missing;
2869
- }
2870
- const hasMockApi = facts.some((fact) => fact.kind === "mock-api");
2871
- const liveInteractions = collectCanonicalStateFlowNames(input.committedFacts).interactionNames;
2872
- const coveredInteractions = new Set(facts
2873
- .filter((fact) => fact.kind === "data-flow")
2874
- .flatMap((fact) => planFactStringList(fact.interactions)));
2875
- const missing = [];
2876
- if (!hasMockApi) {
2877
- missing.push({
2878
- kind: "mock-api",
2879
- requirementIds: [...input.requirementIds],
2880
- reason: "global Mock/data phase has no committed mock-api fact",
2881
- });
2882
- }
2883
- for (const interaction of liveInteractions) {
2884
- if (coveredInteractions.has(interaction))
2885
- continue;
2886
- missing.push({
2887
- kind: "data-flow",
2888
- id: interaction,
2889
- requirementIds: [...input.requirementIds],
2890
- reason: `interaction ${interaction} has no committed data-flow fact`,
2891
- });
2892
- }
2893
- return missing;
2894
- }
2895
3809
  /** Estimate calls conservatively: requirement + one VT, with a second VT
2896
3810
  * reserved for behaviour-required requirements. Explicit declarations win. */
2897
3811
  export function estimateFrontendPlanRequirementRecordCalls(fact) {
@@ -2932,28 +3846,48 @@ export function batchFrontendPlanRequirements(input) {
2932
3846
  return batches;
2933
3847
  }
2934
3848
  const FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS = 10;
2935
- const FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS = 6;
2936
- // A large requirement set creates both coverage and UX-local shards. Keep a
2937
- // safety bound, but do not let the old 32-session ceiling skip finalize for a
2938
- // legitimate large plan.
3849
+ const FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY = 4;
3850
+ // A single registry owns global names; large UX work uses bounded serial
3851
+ // scopes against that registry. Keep a safety bound for adaptive retries
3852
+ // without letting the old 32-session ceiling skip finalize.
2939
3853
  const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
2940
3854
  const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
2941
3855
  const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
2942
- function compactPromptString(value, maxChars) {
2943
- if (typeof value !== "string" || value.trim().length === 0)
2944
- return undefined;
2945
- const normalized = value.trim();
2946
- return normalized.length <= maxChars
2947
- ? normalized
2948
- : `${normalized.slice(0, maxChars - 1)}…`;
3856
+ async function mapWithConcurrency(items, limit, worker, shouldReduceConcurrency) {
3857
+ const results = new Array(items.length);
3858
+ let nextIndex = 0;
3859
+ let concurrency = Math.max(1, limit);
3860
+ const running = new Map();
3861
+ try {
3862
+ while (nextIndex < items.length || running.size > 0) {
3863
+ while (nextIndex < items.length && running.size < concurrency) {
3864
+ const index = nextIndex++;
3865
+ running.set(index, worker(items[index], index).then(result => ({ index, result })));
3866
+ }
3867
+ const { index, result } = await Promise.race(running.values());
3868
+ running.delete(index);
3869
+ results[index] = result;
3870
+ // Drain existing work; only pending shards use the reduced cap.
3871
+ // Failed shards remain failed and receive no extra retry allowance.
3872
+ if (shouldReduceConcurrency(result))
3873
+ concurrency = Math.max(1, Math.floor(concurrency / 2));
3874
+ }
3875
+ }
3876
+ catch (error) {
3877
+ await Promise.allSettled(running.values());
3878
+ throw error;
3879
+ }
3880
+ return results;
2949
3881
  }
2950
- function compactPromptStringArray(value, maxEntries = 12, maxChars = 180) {
3882
+ function compactPromptString(value, _maxChars) {
3883
+ return typeof value === "string" && value.trim().length ? value.trim() : undefined;
3884
+ }
3885
+ function compactPromptStringArray(value, _maxEntries = 12, maxChars = 180) {
2951
3886
  if (!Array.isArray(value))
2952
3887
  return [];
2953
3888
  return value
2954
3889
  .map((item) => compactPromptString(item, maxChars))
2955
- .filter((item) => item !== undefined)
2956
- .slice(0, maxEntries);
3890
+ .filter((item) => item !== undefined);
2957
3891
  }
2958
3892
  function countFrontendPlanTargetSurfaces(basePrompt) {
2959
3893
  const match = /<frontend_plan_input>[\s\S]*?<\/frontend_plan_input>/.exec(basePrompt);
@@ -3026,7 +3960,7 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3026
3960
  }
3027
3961
  }
3028
3962
  if (!payload || !Array.isArray(payload.requirements))
3029
- return basePrompt;
3963
+ throw Error("FRONTEND_INPUT_INVALID: plan inventory is not parseable");
3030
3964
  const requirementsById = new Map();
3031
3965
  for (const value of payload.requirements) {
3032
3966
  if (!value || typeof value !== "object" || Array.isArray(value))
@@ -3038,7 +3972,7 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3038
3972
  }
3039
3973
  const requirements = slice.map((id) => requirementsById.get(id));
3040
3974
  if (requirements.some((requirement) => requirement === undefined)) {
3041
- return basePrompt;
3975
+ throw Error(`FRONTEND_INPUT_SCOPE_MISSING: ${slice.filter(id => !requirementsById.has(id)).join(", ")}`);
3042
3976
  }
3043
3977
  const compactRequirements = requirements.map((requirement) => ({
3044
3978
  id: compactPromptString(requirement.id, 80),
@@ -3055,26 +3989,28 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3055
3989
  if (!value || typeof value !== "object" || Array.isArray(value))
3056
3990
  return [];
3057
3991
  const target = value;
3058
- const targetRequirementIds = compactPromptStringArray(target.requirementIds, 20, 80);
3992
+ if (options.verificationTargetIds && !options.verificationTargetIds.includes(String(target.id)))
3993
+ return [];
3994
+ const targetRequirementIds = planFactStringList(target.requirementIds);
3059
3995
  const relatedRequirementIds = targetRequirementIds.filter((id) => sliceSet.has(id));
3060
3996
  if (relatedRequirementIds.length === 0)
3061
3997
  return [];
3062
3998
  return [
3063
3999
  {
3064
- ...(compactPromptString(target.id, 80)
3065
- ? { id: compactPromptString(target.id, 80) }
4000
+ ...(typeof target.id === "string"
4001
+ ? { id: target.id }
3066
4002
  : {}),
3067
- ...(compactPromptString(target.type, 40)
3068
- ? { type: compactPromptString(target.type, 40) }
4003
+ ...(typeof target.commandId === "string"
4004
+ ? { commandId: target.commandId }
3069
4005
  : {}),
3070
4006
  ...(compactPromptString(target.commandLabel, 180)
3071
4007
  ? { commandLabel: compactPromptString(target.commandLabel, 180) }
3072
4008
  : {}),
3073
- ...(compactPromptString(target.file, 180)
3074
- ? { file: compactPromptString(target.file, 180) }
4009
+ ...(typeof target.file === "string"
4010
+ ? { file: target.file }
3075
4011
  : {}),
3076
4012
  requirementIds: relatedRequirementIds,
3077
- uiStates: compactPromptStringArray(target.uiStates, 12, 100),
4013
+ uiStates: planFactStringList(target.uiStates),
3078
4014
  },
3079
4015
  ];
3080
4016
  })
@@ -3121,8 +4057,36 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3121
4057
  }];
3122
4058
  })
3123
4059
  : [];
4060
+ const compactDeclaredUiStates = Array.isArray(payload.declaredUiStates)
4061
+ ? payload.declaredUiStates.flatMap((value) => {
4062
+ if (!value || typeof value !== "object" || Array.isArray(value))
4063
+ return [];
4064
+ const state = value;
4065
+ const id = typeof state.id === "string" ? state.id : undefined;
4066
+ if (!id)
4067
+ return [];
4068
+ return [{
4069
+ id,
4070
+ ...(compactPromptString(state.trigger, 180)
4071
+ ? { trigger: compactPromptString(state.trigger, 180) }
4072
+ : {}),
4073
+ ...(compactPromptString(state.observableOutcome, 240)
4074
+ ? { observableOutcome: compactPromptString(state.observableOutcome, 240) }
4075
+ : {}),
4076
+ }];
4077
+ })
4078
+ : [];
4079
+ const committedUx = payload.committedUx &&
4080
+ typeof payload.committedUx === "object" &&
4081
+ !Array.isArray(payload.committedUx)
4082
+ ? payload.committedUx
4083
+ : undefined;
3124
4084
  const compactPayload = {
4085
+ inputManifest: { ...projectFrontendInputScope({ ...payload, requirements: [...requirementsById.values()] }, slice).inputManifest, semantics: options.includeRequirementText === false ? "navigation-only" : "full" },
4086
+ constraints: payload.constraints,
4087
+ executionGroups: Array.isArray(payload.executionGroups) ? payload.executionGroups.filter(g => isRecordObject(g) && Array.isArray(g.requirementIds) && g.requirementIds.some(id => slice.includes(String(id)))) : [],
3125
4088
  requirements: compactRequirements,
4089
+ requiredDeliverables: Array.isArray(payload.requiredDeliverables) ? payload.requiredDeliverables : [],
3126
4090
  ...(compactTargetSurface.length > 0
3127
4091
  ? { targetSurface: compactTargetSurface }
3128
4092
  : {}),
@@ -3132,6 +4096,17 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3132
4096
  ...(compactDesignEvidence.length > 0
3133
4097
  ? { designEvidence: compactDesignEvidence }
3134
4098
  : {}),
4099
+ ...(compactDeclaredUiStates.length > 0
4100
+ ? { declaredUiStates: compactDeclaredUiStates }
4101
+ : {}),
4102
+ ...(committedUx
4103
+ ? {
4104
+ committedUx: {
4105
+ uiStateNames: compactPromptStringArray(committedUx.uiStateNames, 40, 100),
4106
+ interactionNames: compactPromptStringArray(committedUx.interactionNames, 60, 120),
4107
+ },
4108
+ }
4109
+ : {}),
3135
4110
  };
3136
4111
  const compactBlock = [
3137
4112
  "<frontend_plan_input>",
@@ -3187,9 +4162,9 @@ function compactFrontendPlanLedgerContext(input) {
3187
4162
  compactFacts.push({
3188
4163
  kind: fact.kind,
3189
4164
  entry: {
3190
- id: compactPromptString(entry.id, 80),
3191
- implementationTargets: compactPromptStringArray(entry.implementationTargets, 12, 180),
3192
- verificationTargetIds: compactPromptStringArray(entry.verificationTargetIds, 12, 80),
4165
+ id: entry.id,
4166
+ implementationTargets: planFactStringList(entry.implementationTargets).slice(0, 12),
4167
+ verificationTargetIds: planFactStringList(entry.verificationTargetIds).slice(0, 12),
3193
4168
  ...(compactPromptString(entry.expectedOutcome, 240)
3194
4169
  ? { expectedOutcome: compactPromptString(entry.expectedOutcome, 240) }
3195
4170
  : {}),
@@ -3203,18 +4178,19 @@ function compactFrontendPlanLedgerContext(input) {
3203
4178
  if (fact.kind === "plan-verification-target") {
3204
4179
  if (!entry)
3205
4180
  continue;
3206
- const requirementIds = compactPromptStringArray(entry.requirementIds, 20, 80);
3207
- if (!requirementIds.some((id) => slice.has(id)))
4181
+ // Scope against the complete canonical binding before compacting.
4182
+ // A shared target may bind a late requirement beyond the preview cap.
4183
+ const requirementIds = planFactStringList(entry.requirementIds).filter((id) => slice.has(id));
4184
+ if (requirementIds.length === 0)
3208
4185
  continue;
3209
4186
  compactFacts.push({
3210
4187
  kind: fact.kind,
3211
4188
  entry: {
3212
- id: compactPromptString(entry.id, 80),
3213
- type: compactPromptString(entry.type, 40),
3214
- commandLabel: compactPromptString(entry.commandLabel, 180),
3215
- file: compactPromptString(entry.file, 180),
4189
+ id: entry.id,
4190
+ commandId: entry.commandId,
4191
+ file: entry.file,
3216
4192
  requirementIds,
3217
- uiStates: compactPromptStringArray(entry.uiStates, 12, 100),
4193
+ uiStates: planFactStringList(entry.uiStates).slice(0, 12),
3218
4194
  },
3219
4195
  });
3220
4196
  continue;
@@ -3229,18 +4205,28 @@ function compactFrontendPlanLedgerContext(input) {
3229
4205
  purpose: compactPromptString(item.purpose, 120),
3230
4206
  component: compactPromptString(item.component, 120),
3231
4207
  decision: compactPromptString(item.decision, 40),
4208
+ covers: compactPromptStringArray(item.covers, 40, 120),
4209
+ evidencePath: compactPromptString(item.evidencePath, 180),
3232
4210
  }];
3233
- }).slice(0, 24)
4211
+ })
3234
4212
  : [];
3235
4213
  if (choices.length > 0)
3236
4214
  compactFacts.push({ kind: fact.kind, uiComponentChoices: choices });
3237
4215
  continue;
3238
4216
  }
4217
+ if (fact.kind === "state-registry") {
4218
+ compactFacts.push({
4219
+ kind: fact.kind,
4220
+ uiStateNames: planFactStringList(fact.uiStateNames).slice(0, 40),
4221
+ interactionNames: planFactStringList(fact.interactionNames).slice(0, 60),
4222
+ });
4223
+ continue;
4224
+ }
3239
4225
  if (fact.kind === "state-flow") {
3240
4226
  compactFacts.push({
3241
4227
  kind: fact.kind,
3242
- uiStates: Array.isArray(fact.uiStates) ? fact.uiStates.slice(0, 24) : [],
3243
- interactions: Array.isArray(fact.interactions) ? fact.interactions.slice(0, 24) : [],
4228
+ uiStates: Array.isArray(fact.uiStates) ? fact.uiStates : [],
4229
+ interactions: Array.isArray(fact.interactions) ? fact.interactions : [],
3244
4230
  removeUiStateNames: compactPromptStringArray(fact.removeUiStateNames, 24, 100),
3245
4231
  removeInteractionNames: compactPromptStringArray(fact.removeInteractionNames, 24, 100),
3246
4232
  });
@@ -3263,7 +4249,7 @@ function compactFrontendPlanLedgerContext(input) {
3263
4249
  mockApi: {
3264
4250
  strategy: compactPromptString(mockApi.strategy, 40),
3265
4251
  activation: compactPromptString(mockApi.activation, 180),
3266
- endpoints: Array.isArray(mockApi.endpoints) ? mockApi.endpoints.slice(0, 24) : [],
4252
+ endpoints: Array.isArray(mockApi.endpoints) ? mockApi.endpoints : [],
3267
4253
  },
3268
4254
  });
3269
4255
  continue;
@@ -3284,63 +4270,379 @@ function compactFrontendPlanLedgerContext(input) {
3284
4270
  return "";
3285
4271
  const priorityFacts = compactFacts.filter((fact) => fact.kind === "plan-requirement" || fact.kind === "plan-verification-target");
3286
4272
  const otherFacts = compactFacts.filter((fact) => fact.kind !== "plan-requirement" && fact.kind !== "plan-verification-target");
4273
+ let previewBytes = 0;
3287
4274
  const boundedFacts = [
3288
4275
  ...priorityFacts.slice(0, 64),
3289
4276
  ...otherFacts.slice(-32),
3290
- ].slice(0, 96);
4277
+ ].slice(0, 96).filter(fact => {
4278
+ const bytes = Buffer.byteLength(JSON.stringify(fact));
4279
+ if (previewBytes + bytes > 24_000)
4280
+ return false;
4281
+ previewBytes += bytes;
4282
+ return true;
4283
+ });
3291
4284
  return [
3292
4285
  "<frontend_plan_ledger>",
3293
- "Committed plan facts from earlier sessions. Treat these as authoritative; correct them only with the allowed replacement/removal fields.",
4286
+ "Preview of committed plan facts; not a replacement payload. Fields and lists may be omitted or shortened. Use read_plan_facts with kind, entryId/eventId and optional field to retrieve complete values in bounded pages before corrections; preserve all other bindings. Absence from this preview is not absence from the ledger.",
4287
+ JSON.stringify({ complete: false, matchingFacts: compactFacts.length, omittedFacts: compactFacts.length - boundedFacts.length }),
3294
4288
  JSON.stringify(boundedFacts),
3295
4289
  "</frontend_plan_ledger>",
3296
4290
  ].join("\n");
3297
4291
  }
3298
- export async function runFrontendPlanSegmentedSessions(input) {
3299
- const queue = [];
3300
- const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
3301
- const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
3302
- const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
3303
- const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
3304
- const allRequirementIds = input.requirementIds ?? [];
3305
- const buildPhasePrompt = (segment, missing = []) => {
3306
- const compact = allRequirementIds.length > 0
3307
- ? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds, {
3308
- includeRequirementText: segment.id === "global-mock-data" || segment.id === "global-dependency-deviation",
3309
- includeVerificationTargets: false,
3310
- includeDesignEvidence: segment.id === "global-dependency-deviation" || segment.id === "finalize",
3311
- includeChecklist: false,
3312
- })
3313
- : input.basePrompt;
3314
- const kinds = segment.id === "global-route"
3315
- ? ["target-surface"]
3316
- : segment.id === "global-mock-data"
3317
- ? ["plan-requirement", "plan-verification-target", "state-flow", "data-flow", "mock-api"]
3318
- : segment.id === "global-dependency-deviation"
3319
- ? ["dependency", "design-deviation"]
3320
- : ["plan-requirement", "plan-verification-target", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
3321
- const ledger = input.committedFacts
3322
- ? compactFrontendPlanLedgerContext({
3323
- committedFacts: input.committedFacts(),
3324
- requirementIds: allRequirementIds,
3325
- kinds,
3326
- })
3327
- : "";
3328
- return [
3329
- compact,
3330
- segment.instruction,
3331
- ledger,
3332
- ...(missing.length > 0
3333
- ? [
3334
- "MISSING-FACT QUEUE: repair ONLY these items, then re-check the phase:",
3335
- ...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
3336
- ]
3337
- : []),
3338
- ].filter(Boolean).join("\n\n");
4292
+ export async function runFrontendReviewSegmentedSessions(input) {
4293
+ const protocol = input.tools.scopeProtocol;
4294
+ const terminalKinds = input.phase === "review" ? ["approve_review", "request_review_changes"] : ["approve_design", "request_design_changes"];
4295
+ const terminal = () => protocol.committedFacts().some(r => terminalKinds.includes(String(r.fact.kind)));
4296
+ const targetBytes = input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES;
4297
+ const queue = packFrontendInputUnits(input.inventory.scopes.filter(s => !protocol.completedScopeIds().has(s.id)), { targetBytes, maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits }).map(scopes => ({ scopes, repairs: 0 }));
4298
+ if (!queue.length)
4299
+ queue.push({ scopes: [], repairs: 0 });
4300
+ let calls = 0;
4301
+ let last = { ok: true, assistantText: "", command: [], durationMs: 0, exitCode: 0, failureCategory: "success", modelDisplay: "unknown", parsedEvents: 0, stderr: "", stdout: "", timedOut: false, attemptedModels: [], fallbackUsed: false, tokensUsed: 0 };
4302
+ for (let index = 0; index < queue.length; index++) {
4303
+ await input.tools.flush();
4304
+ await input.inventory.validate();
4305
+ if (terminal()) {
4306
+ protocol.assertComplete();
4307
+ return last;
4308
+ }
4309
+ const item = queue[index];
4310
+ const scopes = item.scopes.filter(s => !protocol.completedScopeIds().has(s.id));
4311
+ protocol.setActiveScope(scopes.map(s => s.id));
4312
+ const finalScope = input.inventory.scopes.every(s => protocol.completedScopeIds().has(s.id) || scopes.some(current => current.id === s.id));
4313
+ const customTools = finalScope ? input.customTools : input.customTools.filter(t => !terminalKinds.includes(String(t.name)));
4314
+ const prompt = `${input.basePrompt}\n<frontend_review_scope>\n${JSON.stringify({ semantics: "full", inventoryDigest: input.inventory.digest, scopes, previouslyCompleted: [...protocol.completedScopeIds()], savedFindings: protocol.committedFacts().filter(r => String(r.fact.kind).endsWith("-finding")).map(r => ({ id: r.fact.id, finding: r.fact.finding })) })}\n</frontend_review_scope>\nReview the complete supplied scopes, saving each finding immediately. Call complete_review_scope for each exact id only after all its independent permissions, thresholds, errors and evidence have been checked. ${finalScope ? "After every scope is complete, make one independent overall approve/request_changes decision; persisted findings cannot be omitted." : "More scopes remain. Do not finalize or reread already completed scopes unless resolving a cross-scope issue."}`;
4315
+ if (Buffer.byteLength(prompt) > targetBytes && scopes.length > 1) {
4316
+ const at = Math.ceil(scopes.length / 2);
4317
+ queue.splice(index, 1, { scopes: scopes.slice(0, at), repairs: 0 }, { scopes: scopes.slice(at), repairs: 0 });
4318
+ index--;
4319
+ continue;
4320
+ }
4321
+ if (++calls > 128)
4322
+ return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_REVIEW_RECOVERY_EXHAUSTED: session quota reached" };
4323
+ last = await observeFrontendSession({ ...input.observation, phase: `${input.phase}/scope`, scopeIds: scopes.map(s => s.id), prompt, userMessage: input.sessionOptions.userMessage, customTools,
4324
+ artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${calls}.json` : undefined,
4325
+ committedCount: () => protocol.committedFacts().length, durableCommittedCount: () => protocol.committedFacts().length,
4326
+ }, observer => input.piStepFn({ ...input.sessionOptions, prompt, onAttemptObservation: observer, writerToolPolicy: { requireSdk: true, customTools } }));
4327
+ await input.tools.flush();
4328
+ await input.inventory.validate();
4329
+ if (last.timedOut || /interrupt|termination-unconfirmed|budget_breach/.test(last.failureCategory))
4330
+ return { ...last, ok: false };
4331
+ const capacity = readWriterThinkingExhaustionEvidence(last).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(last.failureCategory);
4332
+ const missing = scopes.filter(s => !protocol.completedScopeIds().has(s.id));
4333
+ if (capacity && (missing.length || !terminal() && finalScope)) {
4334
+ if (missing.length === 1 && scopes.length === 1 || !missing.length && !scopes.length)
4335
+ return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_INPUT_UNIT_TOO_LARGE: complete review unit or terminal exhausted" };
4336
+ const at = Math.ceil(missing.length / 2);
4337
+ const smaller = missing.length ? [missing.slice(0, at), missing.slice(at)].filter(s => s.length) : [[]];
4338
+ queue.splice(index, 1, ...smaller.map(scopes => ({ scopes, repairs: 0 })));
4339
+ index--;
4340
+ continue;
4341
+ }
4342
+ if (!last.ok && !capacity)
4343
+ return last;
4344
+ if (missing.length || finalScope && !terminal()) {
4345
+ if (item.repairs >= 1)
4346
+ return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_REVIEW_SCOPE_INCOMPLETE: required checkpoint or verdict missing" };
4347
+ queue.splice(index, 1, { scopes: missing, repairs: item.repairs + 1 });
4348
+ index--;
4349
+ }
4350
+ }
4351
+ protocol.assertComplete();
4352
+ return terminal() ? { ...last, ok: true, failureCategory: "success" } : { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_REVIEW_SCOPE_INCOMPLETE: missing independent verdict" };
4353
+ }
4354
+ export async function runFrontendScoutSegmentedSessions(input) {
4355
+ const inventory = parseFrontendInputBlock(input.basePrompt, "scout");
4356
+ if (!inventory)
4357
+ throw Error("FRONTEND_INPUT_MISSING: Scout compiled inventory unavailable");
4358
+ const units = collectFrontendExecutionGroups(inventory.payload.requirements).map(group => ({ ...group, id: `${group.kind === "unclassified" ? "requirement" : "group"}:${group.id}`, requirements: inventory.payload.requirements.filter(r => group.requirementIds.includes(r.id)) }));
4359
+ const targetBytes = input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES;
4360
+ const queue = packFrontendInputUnits(units, { targetBytes, maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits }).map(batch => ({ groups: batch, repairs: 0 }));
4361
+ let last = { ok: true, assistantText: "", command: [], durationMs: 0, exitCode: 0, failureCategory: "success", modelDisplay: input.sessionOptions.modelConfig?.model ?? "unknown", parsedEvents: 0, stderr: "", stdout: "", timedOut: false, attemptedModels: [], fallbackUsed: false, tokensUsed: 0 };
4362
+ let calls = 0;
4363
+ for (let index = 0; index < queue.length; index++) {
4364
+ await input.tools.flush();
4365
+ const item = queue[index];
4366
+ const completed = input.tools.completedRequirementIds();
4367
+ const groups = item.groups.filter(g => g.requirementIds.some(id => !completed.has(id)));
4368
+ if (!groups.length)
4369
+ continue;
4370
+ const ids = groups.flatMap(g => g.requirementIds);
4371
+ const scopeId = input.tools.setActiveScope(ids);
4372
+ const prompt = input.basePrompt.replace(inventory.block, `<frontend_scout_input>\n${JSON.stringify(projectFrontendInputScope(inventory.payload, ids))}\n</frontend_scout_input>`) +
4373
+ `\nSCOUT SCOPE ${scopeId}: discover the related surfaces for exactly these complete obligations: ${ids.join(", ")}. Reuse proven paths from completed scope navigation; do not reread their content unless relevant new evidence is needed. Submit record_target_surface with scopeId="${scopeId}" after all discovery/design evidence for this scope. completeness=complete closes only this scope; runtime merges all scopes.\n` +
4374
+ JSON.stringify({ completedScopePaths: input.tools.completedScopeFacts().map(f => ({ id: f.id, requirementIds: f.requirementIds, paths: f.surface?.implementationPaths })) });
4375
+ const envelopeBytes = Buffer.byteLength(prompt + input.sessionOptions.userMessage + JSON.stringify(input.customTools.map(t => { const tool = t; return { name: tool.name, description: tool.description, parameters: tool.parameters }; })));
4376
+ if (envelopeBytes > targetBytes && groups.length > 1) {
4377
+ const at = Math.ceil(groups.length / 2);
4378
+ queue.splice(index, 1, { groups: groups.slice(0, at), repairs: 0 }, { groups: groups.slice(at), repairs: 0 });
4379
+ index--;
4380
+ continue;
4381
+ }
4382
+ if (++calls > 128)
4383
+ return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "SCOUT_RECOVERY_EXHAUSTED: shared session quota reached" };
4384
+ last = await observeFrontendSession({ ...input.observation, phase: "scout/scope", scopeIds: ids, prompt, userMessage: input.sessionOptions.userMessage, customTools: input.customTools,
4385
+ artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${calls}.json` : undefined,
4386
+ committedCount: () => input.tools.committedFacts().length, durableCommittedCount: () => input.tools.committedFacts().length,
4387
+ }, observer => input.piStepFn({ ...input.sessionOptions, prompt, onAttemptObservation: observer, writerToolPolicy: { requireSdk: true, customTools: input.customTools } }));
4388
+ await input.tools.flush();
4389
+ const missing = groups.filter(g => g.requirementIds.some(id => !input.tools.completedRequirementIds().has(id)));
4390
+ if (last.timedOut || /interrupt|termination-unconfirmed|budget_breach/.test(last.failureCategory))
4391
+ return { ...last, ok: false };
4392
+ const capacity = readWriterThinkingExhaustionEvidence(last).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(last.failureCategory);
4393
+ if (capacity && missing.length) {
4394
+ if (missing.length === 1 && groups.length === 1)
4395
+ return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: `${last.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: Scout ${missing[0].id}; refine the complete source unit; unchanged retries disabled` };
4396
+ const at = Math.ceil(missing.length / 2);
4397
+ queue.splice(index, 1, ...[missing.slice(0, at), missing.slice(at)].filter(batch => batch.length).map(batch => ({ groups: batch, repairs: 0 })));
4398
+ index--;
4399
+ continue;
4400
+ }
4401
+ if (!last.ok && !capacity)
4402
+ return last;
4403
+ if (missing.length) {
4404
+ if (item.repairs >= 1)
4405
+ return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "SCOUT_SCOPE_INCOMPLETE: unresolved discovery remains after local correction" };
4406
+ queue.splice(index, 1, { groups: missing, repairs: item.repairs + 1 });
4407
+ index--;
4408
+ continue;
4409
+ }
4410
+ }
4411
+ await input.tools.flush();
4412
+ const { readCompleteScoutTargetSurface } = await import("../workflows/dag/frontend-shadow-dual-write.js");
4413
+ const closure = readCompleteScoutTargetSurface(input.tools.committedFacts());
4414
+ return closure.ok ? { ...last, ok: true, failureCategory: "success" } : { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: closure.reason };
4415
+ }
4416
+ export async function runFrontendContractSegmentedSessions(input) {
4417
+ // Build from the frozen runtime inventory if the caller has not rendered it yet.
4418
+ const basePrompt = parseFrontendInputBlock(input.basePrompt, "contract") ? input.basePrompt : input.basePrompt +
4419
+ `\n<frontend_contract_input>\nFrozen complete source obligations.\n${JSON.stringify({ requirements: input.tools.inputRequirements() })}\n</frontend_contract_input>`;
4420
+ const inventory = parseFrontendInputBlock(basePrompt, "contract");
4421
+ const pending = inventory.payload.requirements.filter(r => !input.tools.completedScopeRequirementIds().has(r.id));
4422
+ const batches = packFrontendInputUnits(pending, { targetBytes: input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes, maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits });
4423
+ if (!batches.length)
4424
+ batches.push([]);
4425
+ let last;
4426
+ let invocation = 0;
4427
+ const maxSessions = batches.length * 3 + 2;
4428
+ const terminal = () => input.tools.committedFacts().some(r => r.fact.kind === "contract-finalized");
4429
+ try {
4430
+ for (let index = 0; index < batches.length; index += 1) {
4431
+ let scopeIds = batches[index].map(r => r.id);
4432
+ const finalScope = index === batches.length - 1;
4433
+ for (let repair = 0; repair < 2; repair += 1) {
4434
+ // Confirmed IDs remain visible until the model commits a scope checkpoint.
4435
+ const completed = input.tools.completedScopeRequirementIds();
4436
+ scopeIds = scopeIds.filter(id => !completed.has(id));
4437
+ input.tools.setActiveRequirementScope(scopeIds);
4438
+ const groupIndex = collectFrontendExecutionGroups(input.tools.committedFacts().filter(r => r.fact.kind === "requirement").map(r => ({ id: String(r.fact.id), execution: r.fact.execution })));
4439
+ const shared = input.tools.committedFacts().filter(r => !["requirement", "contract-finalized", "contract-scope-completed"].includes(String(r.fact.kind))).map(r => r.fact);
4440
+ const prompt = projectFrontendContractPrompt(basePrompt, scopeIds) +
4441
+ `\nCONTRACT SCOPE: analyze only ${scopeIds.join(", ") || "(all scopes complete; verify global facts and terminal)"}. Each obligation is complete. Submit small records immediately, then call complete_contract_scope after ALL decisions for this scope. ` +
4442
+ (finalScope ? "After complete scope coverage and source-bound deliverables, call finalize_contract. Correct rejected calls and retry." : "Do not finalize; subsequent complete scopes remain.") +
4443
+ `\n<committed_contract_facts>\n${JSON.stringify({ facts: shared, executionGroups: groupIndex })}\n</committed_contract_facts>`;
4444
+ const customTools = input.tools.customTools.filter(t => finalScope || t.name !== "finalize_contract");
4445
+ invocation += 1;
4446
+ if (invocation > maxSessions)
4447
+ return { ...last, ok: false, failureCategory: "invalid-output", stderr: "CONTRACT_RECOVERY_EXHAUSTED: session quota exceeded" };
4448
+ last = await observeFrontendSession({
4449
+ ...input.observation, phase: "contract/scope", scopeIds, prompt, userMessage: input.sessionOptions.userMessage, customTools,
4450
+ artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${invocation}.json` : undefined,
4451
+ committedCount: () => input.tools.committedFacts().length, durableCommittedCount: () => input.tools.committedFacts().length,
4452
+ }, observer => input.piStepFn({ ...input.sessionOptions, prompt, onAttemptObservation: observer, writerToolPolicy: { requireSdk: true, customTools } }));
4453
+ await input.tools.flush();
4454
+ if (last.timedOut || /interrupt|termination-unconfirmed|budget_breach/.test(last.failureCategory))
4455
+ return { ...last, ok: false };
4456
+ const capacityExhausted = readWriterThinkingExhaustionEvidence(last).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(last.failureCategory);
4457
+ if (capacityExhausted && !last.timedOut) {
4458
+ const missing = scopeIds.filter(id => !input.tools.completedScopeRequirementIds().has(id));
4459
+ if (!missing.length) {
4460
+ if (finalScope && !terminal()) {
4461
+ if (!scopeIds.length)
4462
+ return { ...last, ok: false, failureCategory: "invalid-output", stderr: `${last.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: contract terminal exhausted; unchanged retry disabled` };
4463
+ batches.push([]);
4464
+ }
4465
+ break;
4466
+ }
4467
+ if (missing.length === 1 && scopeIds.length > 1) {
4468
+ batches.splice(index, 1, inventory.payload.requirements.filter(r => missing.includes(r.id)));
4469
+ index -= 1;
4470
+ break;
4471
+ }
4472
+ if (missing.length <= 1)
4473
+ return { ...last, ok: false, stderr: `${last.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${missing[0] ?? "contract terminal"}; atom could not complete; refine the source without dropping conditions` };
4474
+ const smaller = packFrontendInputUnits(inventory.payload.requirements.filter(r => missing.includes(r.id)), { maxUnits: Math.ceil(missing.length / 2) });
4475
+ batches.splice(index, 1, ...smaller);
4476
+ index -= 1;
4477
+ break;
4478
+ }
4479
+ if (!last.ok)
4480
+ return last;
4481
+ const missing = scopeIds.filter(id => !input.tools.completedScopeRequirementIds().has(id));
4482
+ if (!missing.length && (!finalScope || terminal()))
4483
+ break;
4484
+ if (repair === 1)
4485
+ return { ...last, ok: false, failureCategory: "invalid-output", stderr: `CONTRACT_SCOPE_INCOMPLETE: ${missing.join(", ") || "missing finalize_contract"}` };
4486
+ }
4487
+ }
4488
+ return last;
4489
+ }
4490
+ finally {
4491
+ input.tools.setActiveRequirementScope(null);
4492
+ }
4493
+ }
4494
+ /** One workload view for both the outer parallel map and inner session queue.
4495
+ * A declared execution group is an indivisible unit during initial packing. */
4496
+ function buildFrontendPlanWorkload(input) {
4497
+ const allRequirementIds = input.requirementIds;
4498
+ const compiledInput = parseFrontendInputBlock(input.basePrompt, "plan")?.payload;
4499
+ const fullUnits = new Map(compiledInput?.requirements.map(r => [r.id, r]) ?? []);
4500
+ const declaredGroups = Array.isArray(compiledInput?.executionGroups)
4501
+ ? compiledInput.executionGroups.filter(isRecordObject) : [];
4502
+ const workGroups = declaredGroups.map(g => ({
4503
+ id: String(g.id), kind: String(g.kind),
4504
+ requirementIds: Array.isArray(g.requirementIds)
4505
+ ? g.requirementIds.filter((id) => typeof id === "string" && allRequirementIds.includes(id)) : [],
4506
+ })).filter(g => g.requirementIds.length);
4507
+ const groupedIds = new Set(workGroups.flatMap(g => g.requirementIds));
4508
+ for (const id of allRequirementIds) {
4509
+ if (!groupedIds.has(id))
4510
+ workGroups.push({ id, kind: "unclassified", requirementIds: [id] });
4511
+ }
4512
+ const workCost = (ids) => 1 + ids.reduce((total, id) => total + Math.max(1, (input.requirementCosts?.get(id) ?? 2) - 1), 0);
4513
+ const policy = input.sessionOptions.frontendExecutionPolicy;
4514
+ const targetBytes = policy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES;
4515
+ const userMessageBytes = Buffer.byteLength(input.sessionOptions.userMessage ?? "");
4516
+ const buildWorkBatches = (ids) => {
4517
+ const work = workGroups.flatMap((g, index) => {
4518
+ const members = g.requirementIds.filter(id => ids.includes(id));
4519
+ return members.length ? [{ id: `${index}:${g.id}`, requirementIds: members,
4520
+ requirements: members.map(id => fullUnits.get(id) ?? { id }), estimatedCalls: workCost(members) }] : [];
4521
+ });
4522
+ const scaffoldBytes = Buffer.byteLength(input.basePrompt.replace(/<frontend_plan_input>[\s\S]*?<\/frontend_plan_input>/, JSON.stringify({ ...compiledInput, requirements: [] }))) + userMessageBytes + Buffer.byteLength(JSON.stringify(input.coverageTools));
4523
+ return packFrontendInputUnits(work, {
4524
+ targetBytes: Math.max(1, targetBytes - scaffoldBytes),
4525
+ maxUnits: policy?.maxScopeUnits ?? 4,
4526
+ maxCost: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS, cost: g => g.estimatedCalls,
4527
+ }).map(batch => batch.flatMap(g => g.requirementIds));
4528
+ };
4529
+ // This renderer also carries the Plan's evolving UX checkpoint on resume.
4530
+ // Only the Contract/Scout-derived input participates in ownership binding.
4531
+ const { committedUx: _committedUx, ...frozenInput } = compiledInput ?? {};
4532
+ return {
4533
+ frozenInput, workGroups, buildWorkBatches,
4534
+ compactEligible: input.requirementCosts !== undefined &&
4535
+ workGroups.length <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
4536
+ Buffer.byteLength(input.basePrompt) + userMessageBytes + Buffer.byteLength(JSON.stringify(input.allTools)) <= targetBytes * 2 &&
4537
+ workGroups.reduce((total, group) => total + workCost(group.requirementIds), 0) <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
4538
+ countFrontendPlanTargetSurfaces(input.basePrompt) === 1,
4539
+ };
4540
+ }
4541
+ const frontendPlanCoverageLayoutSchema = z.object({
4542
+ schemaVersion: z.literal(1),
4543
+ bindingSha256: z.string(),
4544
+ coverageBatches: z.array(z.array(z.string().min(1)).min(1)).min(1),
4545
+ layoutSha256: z.string(),
4546
+ }).strict();
4547
+ /** Freeze ledger ownership before any provider call. Session packing remains
4548
+ * adaptive inside each owner, but retry instructions cannot rename its scope. */
4549
+ async function loadOrCreateFrontendPlanCoverageLayout(input) {
4550
+ const fileName = "coverage-layout.json";
4551
+ const file = path.join(input.runDir, input.nodeId, fileName);
4552
+ const bindingSha256 = sha256OfCanonicalJson(input.binding);
4553
+ let raw;
4554
+ try {
4555
+ raw = await readFile(file, "utf8");
4556
+ }
4557
+ catch (error) {
4558
+ if (error.code !== "ENOENT")
4559
+ throw error;
4560
+ }
4561
+ let layout;
4562
+ if (raw !== undefined) {
4563
+ layout = frontendPlanCoverageLayoutSchema.parse(JSON.parse(raw));
4564
+ }
4565
+ else {
4566
+ // Losing the ownership receipt must never repartition acknowledged facts.
4567
+ for (const relative of ["plan-typed-facts.jsonl", "parallel"]) {
4568
+ const existing = await stat(path.join(input.runDir, input.nodeId, relative)).catch(error => {
4569
+ if (error.code !== "ENOENT")
4570
+ throw error;
4571
+ return undefined;
4572
+ });
4573
+ if (existing && (existing.isDirectory() || existing.size > 0)) {
4574
+ throw Error("FRONTEND_PLAN_LAYOUT_MISSING: preserve the existing ledger and restart its owning phase");
4575
+ }
4576
+ }
4577
+ const descriptor = {
4578
+ schemaVersion: 1, bindingSha256,
4579
+ coverageBatches: input.workload.compactEligible ? [input.requirementIds] : input.workload.buildWorkBatches(input.requirementIds),
4580
+ };
4581
+ layout = { ...descriptor, layoutSha256: sha256OfCanonicalJson(descriptor) };
4582
+ }
4583
+ const { layoutSha256, ...descriptor } = layout;
4584
+ if (layout.bindingSha256 !== bindingSha256 || layoutSha256 !== sha256OfCanonicalJson(descriptor)) {
4585
+ throw Error("FRONTEND_PLAN_LAYOUT_MISMATCH: frozen coverage ownership or input binding changed");
4586
+ }
4587
+ const members = layout.coverageBatches.flat();
4588
+ const owners = new Map(layout.coverageBatches.flatMap((batch, index) => batch.map(id => [id, index])));
4589
+ if (members.length !== owners.size || members.length !== input.requirementIds.length ||
4590
+ input.requirementIds.some(id => !owners.has(id)) ||
4591
+ input.workload.workGroups.some(group => new Set(group.requirementIds.map(id => owners.get(id))).size !== 1)) {
4592
+ throw Error("FRONTEND_PLAN_LAYOUT_INVALID: coverage must partition complete execution groups exactly once");
4593
+ }
4594
+ if (raw === undefined)
4595
+ await writeDagNodeJsonArtifact(input.runDir, input.nodeId, fileName, layout);
4596
+ return layout.coverageBatches;
4597
+ }
4598
+ export async function runFrontendPlanSegmentedSessions(input) {
4599
+ const queue = [];
4600
+ const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
4601
+ const uxRegistrySegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-registry");
4602
+ const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
4603
+ const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
4604
+ const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
4605
+ const allRequirementIds = input.requirementIds ?? [];
4606
+ const buildPhasePrompt = (segment, missing = [], scopeIds = allRequirementIds) => {
4607
+ const compact = allRequirementIds.length > 0
4608
+ ? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, scopeIds, {
4609
+ includeRequirementText: segment.id === "global-mock-data",
4610
+ includeVerificationTargets: false,
4611
+ includeDesignEvidence: segment.id === "global-dependency-deviation" || segment.id === "finalize",
4612
+ includeChecklist: false,
4613
+ })
4614
+ : input.basePrompt;
4615
+ const kinds = segment.id === "global-route"
4616
+ ? ["target-surface"]
4617
+ : segment.id === "global-mock-data"
4618
+ ? ["plan-requirement", "plan-verification-target", "state-flow", "data-flow", "mock-api"]
4619
+ : segment.id === "global-dependency-deviation"
4620
+ ? ["dependency", "design-deviation"]
4621
+ : ["plan-requirement", "plan-verification-target", "state-registry", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
4622
+ const ledger = input.committedFacts
4623
+ ? compactFrontendPlanLedgerContext({
4624
+ committedFacts: input.committedFacts(),
4625
+ requirementIds: scopeIds,
4626
+ kinds,
4627
+ ...(segment.id === "global-mock-data" ? { scopedKinds: ["state-flow"] } : {}),
4628
+ })
4629
+ : "";
4630
+ return [
4631
+ compact,
4632
+ segment.instruction,
4633
+ ledger,
4634
+ ...(missing.length > 0
4635
+ ? [
4636
+ "MISSING-FACT QUEUE: repair ONLY these items, then re-check the phase:",
4637
+ ...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
4638
+ ]
4639
+ : []),
4640
+ ].filter(Boolean).join("\n\n");
3339
4641
  };
3340
4642
  const buildCoveragePrompt = (slice, missing = []) => [
3341
4643
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
3342
4644
  coverageSegment.instruction,
3343
- `COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
4645
+ `COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them. Finish the whole list before concluding: commit every listed requirement's coverage facts (verification targets or an evidence gap), batching up to 4 record_* calls per message; an early stop re-queues the remainder as a MISSING-FACT repair session.`,
3344
4646
  ...(missing.length > 0
3345
4647
  ? [
3346
4648
  "MISSING-FACT QUEUE: the previous session did not establish complete coverage. Repair ONLY these items, then re-check the slice:",
@@ -3353,14 +4655,20 @@ export async function runFrontendPlanSegmentedSessions(input) {
3353
4655
  ? compactFrontendPlanLedgerContext({
3354
4656
  committedFacts: input.committedFacts(),
3355
4657
  requirementIds: slice,
3356
- kinds: ["plan-requirement", "plan-verification-target", "component-choice", "state-flow"],
3357
- scopedKinds: ["component-choice", "state-flow"],
4658
+ kinds: [
4659
+ "plan-requirement",
4660
+ "plan-verification-target",
4661
+ "state-registry",
4662
+ "component-choice",
4663
+ "state-flow",
4664
+ ],
4665
+ // Include shared state bindings from other scopes; page full values before correcting.
3358
4666
  })
3359
4667
  : "";
3360
4668
  return [
3361
4669
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
3362
4670
  uxSegment.instruction,
3363
- `UX LOCAL BATCH: process ONLY these requirements: ${slice.join(", ")}.`,
4671
+ `UX SCOPE: process these complete behavior groups together: ${slice.join(", ")}. Reuse shared registry/component ownership; do not rename a behavior by requirement id. Constraint/exclusion groups do not create UI; preserve genuine verification gaps.`,
3364
4672
  ledger,
3365
4673
  ...(missing.length > 0
3366
4674
  ? [
@@ -3370,6 +4678,38 @@ export async function runFrontendPlanSegmentedSessions(input) {
3370
4678
  : []),
3371
4679
  ].filter(Boolean).join("\n\n");
3372
4680
  };
4681
+ const buildUxRegistryPrompt = (missing = []) => {
4682
+ const ledger = input.committedFacts
4683
+ ? compactFrontendPlanLedgerContext({
4684
+ committedFacts: input.committedFacts(),
4685
+ requirementIds: allRequirementIds,
4686
+ kinds: [
4687
+ "plan-requirement",
4688
+ "plan-verification-target",
4689
+ "state-registry",
4690
+ ],
4691
+ })
4692
+ : "";
4693
+ const compact = allRequirementIds.length > 0
4694
+ ? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds, {
4695
+ includeRequirementText: false,
4696
+ includeVerificationTargets: false,
4697
+ includeDesignEvidence: false,
4698
+ includeChecklist: false,
4699
+ })
4700
+ : input.basePrompt;
4701
+ return [
4702
+ compact,
4703
+ uxRegistrySegment.instruction,
4704
+ ledger,
4705
+ ...(missing.length > 0
4706
+ ? [
4707
+ "MISSING-FACT QUEUE: commit the global UX registry, then re-check this phase:",
4708
+ ...missing.map((item) => `- ${item.reason}`),
4709
+ ]
4710
+ : []),
4711
+ ].filter(Boolean).join("\n\n");
4712
+ };
3373
4713
  const buildCompactLocalPrompt = (missing = []) => {
3374
4714
  const ledger = input.committedFacts
3375
4715
  ? compactFrontendPlanLedgerContext({
@@ -3378,6 +4718,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3378
4718
  kinds: [
3379
4719
  "plan-requirement",
3380
4720
  "plan-verification-target",
4721
+ "state-registry",
3381
4722
  "component-choice",
3382
4723
  "state-flow",
3383
4724
  ],
@@ -3386,7 +4727,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3386
4727
  return [
3387
4728
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds),
3388
4729
  "PLAN PHASE — compact local planning for a small frontend request.",
3389
- "For every listed requirement, record plan-requirement and verification-target facts, then record the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
4730
+ "Review all listed requirements together and record_state_registry first with one global UX vocabulary. Then record every plan-requirement and verification-target fact, followed by the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
3390
4731
  "TOOL-FIRST: your first assistant actions must be record_* tool calls, at most 2-3 facts per message. Do not draft the whole analysis before recording; if a fact is uncertain, record it with an evidence gap instead of reasoning longer.",
3391
4732
  ledger,
3392
4733
  ...(missing.length > 0
@@ -3397,13 +4738,13 @@ export async function runFrontendPlanSegmentedSessions(input) {
3397
4738
  : []),
3398
4739
  ].filter(Boolean).join("\n\n");
3399
4740
  };
3400
- const compactFinalizeInstruction = "This is a small-request compact pass. Reconcile the committed local facts with route, data-flow, Mock/API, dependency and deviation policy, then call finalize_plan exactly once.";
4741
+ const compactFinalizeInstruction = "This is a small-request compact pass. Reconcile the committed local facts with route, data-flow, Mock/API, dependency and deviation policy, then call finalize_plan; correct rejected facts and retry until exactly one successful terminal commit.";
3401
4742
  const buildCompactFinalizePrompt = (missing = []) => [buildPhasePrompt(finalizeSegment, missing), compactFinalizeInstruction].join("\n\n");
3402
4743
  const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
3403
4744
  ? {
3404
4745
  ...r,
3405
- failureCategory: PLANNER_THINKING_EXHAUSTED_CATEGORY,
3406
- stderr: `${r.stderr}\n${PLANNER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 typed facts committed; the batch ladder degraded the scope without converging — set thinking=off for this tier or switch to a non-thinking model`.trim(),
4746
+ failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
4747
+ stderr: `${r.stderr}\n${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before the next typed fact; preserve committed facts and retry only the unfinished phase`.trim(),
3407
4748
  }
3408
4749
  : r;
3409
4750
  // An empty list means "ledger unreadable / unknown" and falls back to one
@@ -3418,8 +4759,35 @@ export async function runFrontendPlanSegmentedSessions(input) {
3418
4759
  : [];
3419
4760
  const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
3420
4761
  const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
3421
- const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
3422
- const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
4762
+ if (input.parallelCoverageOnly === true &&
4763
+ requirementIdsProvided &&
4764
+ coverageWorkIds.length === 0) {
4765
+ // A retry may reopen a shard whose committed facts are already complete.
4766
+ // Treat that shard as an idempotent no-op; returning the normal empty
4767
+ // session failure would make a partially failed map impossible to resume.
4768
+ return {
4769
+ ok: true,
4770
+ assistantText: "",
4771
+ command: [],
4772
+ durationMs: 0,
4773
+ exitCode: 0,
4774
+ failureCategory: "success",
4775
+ modelDisplay: "reused-coverage-facts",
4776
+ parsedEvents: 0,
4777
+ stderr: "",
4778
+ stdout: "",
4779
+ timedOut: false,
4780
+ attemptedModels: [],
4781
+ fallbackUsed: false,
4782
+ tokensUsed: 0,
4783
+ };
4784
+ }
4785
+ const { workGroups, buildWorkBatches, compactEligible } = buildFrontendPlanWorkload({
4786
+ basePrompt: input.basePrompt, requirementIds: allRequirementIds,
4787
+ requirementCosts: input.requirementCosts, sessionOptions: input.sessionOptions,
4788
+ coverageTools: input.segmentCustomTools(coverageSegment.toolNames),
4789
+ allTools: input.segmentCustomTools(null),
4790
+ });
3423
4791
  // Small, single-surface requests do not benefit from six isolated Pi
3424
4792
  // sessions. Keep the typed ledger as the authority, but let one local
3425
4793
  // session establish requirement/UX facts and one final session establish
@@ -3427,16 +4795,14 @@ export async function runFrontendPlanSegmentedSessions(input) {
3427
4795
  // for larger plans and for the unscoped compatibility path.
3428
4796
  const useCompactSmallPlan = requirementIdsProvided &&
3429
4797
  input.compactSmallPlan === true &&
3430
- input.requirementCosts !== undefined &&
3431
- estimatedCalls > 12 &&
3432
- (input.requirementIds?.length ?? 0) <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
3433
- estimatedCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
3434
- targetSurfaceCount === 1;
4798
+ compactEligible;
3435
4799
  if (useCompactSmallPlan) {
3436
4800
  const compactLocalTools = new Set([
3437
4801
  "record_plan_requirement",
4802
+ "record_plan_group_coverage",
3438
4803
  "record_plan_verification_target",
3439
4804
  "record_plan_evidence_gap",
4805
+ "record_state_registry",
3440
4806
  "record_component_choice",
3441
4807
  "record_state_flow",
3442
4808
  "adopt_staged_fact",
@@ -3461,13 +4827,8 @@ export async function runFrontendPlanSegmentedSessions(input) {
3461
4827
  });
3462
4828
  }
3463
4829
  else if (coverageWorkIds.length > 0) {
3464
- const coverageBatches = batchFrontendPlanRequirements({
3465
- requirementIds: coverageWorkIds,
3466
- maxEstimatedRecordCalls: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS,
3467
- maxRequirements: 4,
3468
- requirementCosts: input.requirementCosts,
3469
- });
3470
- coverageBatches.forEach((slice, batchIndex) => queue.push({
4830
+ const completeBatches = buildWorkBatches(coverageWorkIds);
4831
+ completeBatches.forEach((slice, batchIndex) => queue.push({
3471
4832
  id: `coverage-batch-${batchIndex + 1}`,
3472
4833
  toolNames: coverageSegment.toolNames,
3473
4834
  coverageSlice: slice,
@@ -3478,35 +4839,35 @@ export async function runFrontendPlanSegmentedSessions(input) {
3478
4839
  if (useCompactSmallPlan) {
3479
4840
  // Compact mode already queued both sessions above.
3480
4841
  }
3481
- else {
3482
- const uxCosts = new Map((input.requirementIds ?? []).map((id) => [
3483
- id,
3484
- Math.max(2, Math.min(3, input.requirementCosts?.get(id) ?? 2)),
3485
- ]));
3486
- const uxSlices = requirementIdsProvided
3487
- ? batchFrontendPlanRequirements({
3488
- requirementIds: input.requirementIds,
3489
- maxEstimatedRecordCalls: FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS,
3490
- maxRequirements: 3,
3491
- requirementCosts: uxCosts,
3492
- })
3493
- : [[]];
3494
- uxSlices.forEach((slice, batchIndex) => queue.push({
4842
+ else if (!input.parallelCoverageOnly) {
4843
+ if (requirementIdsProvided) {
4844
+ queue.push({
4845
+ id: uxRegistrySegment.id,
4846
+ toolNames: uxRegistrySegment.toolNames,
4847
+ prompt: buildUxRegistryPrompt(),
4848
+ });
4849
+ }
4850
+ const uxWorkIds = workGroups.filter(g => !["constraint", "exclusion"].includes(g.kind) || g.requirementIds.some(id => input.behaviorRequiredRequirementIds?.includes(id))).flatMap(g => g.requirementIds);
4851
+ const uxBatches = requirementIdsProvided ? buildWorkBatches(uxWorkIds) : [[]];
4852
+ uxBatches.forEach((slice, batchIndex) => queue.push({
3495
4853
  id: `ux-local-${batchIndex + 1}`,
3496
- toolNames: uxSegment.toolNames,
3497
- ...(slice.length > 0 ? { requirementSlice: slice } : {}),
3498
- prompt: slice.length > 0
3499
- ? buildUxPrompt(slice)
3500
- : buildPhasePrompt(uxSegment),
4854
+ toolNames: requirementIdsProvided ? uxSegment.toolNames : new Set([...(uxSegment.toolNames ?? []), "record_state_registry"]),
4855
+ ...(slice.length ? { requirementSlice: slice } : {}),
4856
+ prompt: slice.length ? buildUxPrompt(slice) : buildPhasePrompt(uxSegment),
3501
4857
  }));
3502
4858
  for (const segment of FRONTEND_PLAN_SEGMENTS) {
3503
- if (["coverage", "ux-local", "finalize"].includes(segment.id))
4859
+ if (["coverage", "ux-registry", "ux-local", "finalize"].includes(segment.id))
3504
4860
  continue;
3505
- queue.push({
3506
- id: segment.id,
3507
- toolNames: segment.toolNames,
3508
- prompt: buildPhasePrompt(segment),
3509
- });
4861
+ if (segment.id === "global-mock-data" && requirementIdsProvided) {
4862
+ buildWorkBatches(allRequirementIds).forEach((slice, i) => queue.push({ id: `global-mock-data-${i + 1}`, toolNames: segment.toolNames, requirementSlice: slice, prompt: buildPhasePrompt(segment, [], slice) }));
4863
+ }
4864
+ else {
4865
+ queue.push({
4866
+ id: segment.id,
4867
+ toolNames: segment.toolNames,
4868
+ prompt: buildPhasePrompt(segment),
4869
+ });
4870
+ }
3510
4871
  }
3511
4872
  queue.push({
3512
4873
  id: finalizeSegment.id,
@@ -3514,11 +4875,33 @@ export async function runFrontendPlanSegmentedSessions(input) {
3514
4875
  prompt: buildPhasePrompt(finalizeSegment),
3515
4876
  });
3516
4877
  }
3517
- let last;
4878
+ let accumulated;
3518
4879
  let index = 0;
3519
4880
  let invocationCount = 0;
4881
+ let lastDurableCount = input.committedFactCount();
3520
4882
  while (index < queue.length) {
3521
4883
  const session = queue[index];
4884
+ if (input.attempt > 1 && input.committedFacts && session.id === "ux-registry" &&
4885
+ collectFrontendPlanPhaseMissingFacts({ phase: "ux-registry", requirementIds: allRequirementIds, committedFacts: input.committedFacts() }).length === 0) {
4886
+ index += 1;
4887
+ continue;
4888
+ }
4889
+ // Reuse complete UX scopes on node retry. Require explicit per-requirement
4890
+ // ownership; an unrelated/global fact must not prove a slice complete.
4891
+ if (input.attempt > 1 && input.committedFacts && session.id.startsWith("ux-local-") && session.requirementSlice?.length) {
4892
+ const facts = input.committedFacts();
4893
+ const ownedIds = new Set(facts.flatMap(value => {
4894
+ const fact = committedFactFromPlanRecord(value);
4895
+ return fact ? planFactStringList(fact.scopeRequirementIds) : [];
4896
+ }));
4897
+ const complete = session.requirementSlice.every(id => ownedIds.has(id)) &&
4898
+ collectFrontendPlanPhaseMissingFacts({ phase: "ux-local", requirementIds: session.requirementSlice,
4899
+ committedFacts: facts, behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds }).length === 0;
4900
+ if (complete) {
4901
+ index += 1;
4902
+ continue;
4903
+ }
4904
+ }
3522
4905
  const remaining = (session.coverageOnly ? session.coverageSlice ?? [] : []).filter((id) => !input.committedRequirementIds?.().has(id));
3523
4906
  const preexistingMissing = session.coverageOnly && session.coverageSlice && input.committedFacts
3524
4907
  ? collectFrontendPlanMissingFacts({
@@ -3536,12 +4919,15 @@ export async function runFrontendPlanSegmentedSessions(input) {
3536
4919
  }
3537
4920
  let prompt = session.prompt;
3538
4921
  if (session.coverageOnly && session.coverageSlice) {
3539
- const promptSlice = remaining.length > 0 ? remaining : session.coverageSlice;
4922
+ const promptSlice = [...new Set([...remaining, ...preexistingMissing.flatMap(item => item.requirementIds)])];
3540
4923
  prompt = buildCoveragePrompt(promptSlice, session.missingFacts ?? preexistingMissing);
3541
4924
  }
3542
4925
  else if (session.id === "compact-local") {
3543
4926
  prompt = buildCompactLocalPrompt(session.missingFacts);
3544
4927
  }
4928
+ else if (session.id === "ux-registry") {
4929
+ prompt = buildUxRegistryPrompt(session.missingFacts);
4930
+ }
3545
4931
  else if (session.id.startsWith("ux-local-")) {
3546
4932
  // Build this at execution time: coverage facts are committed by the
3547
4933
  // preceding sessions and must be visible to the UX-local model.
@@ -3553,41 +4939,79 @@ export async function runFrontendPlanSegmentedSessions(input) {
3553
4939
  // Global phases and finalize also consume the latest committed ledger;
3554
4940
  // constructing their prompt only when the session starts prevents a
3555
4941
  // stale queue entry from dropping facts written by earlier phases.
3556
- const segment = FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === session.id);
4942
+ const segment = FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === session.id || (candidate.id === "global-mock-data" && session.id.startsWith("global-mock-data-")));
3557
4943
  if (segment) {
3558
4944
  prompt =
3559
4945
  session.id === "finalize" && useCompactSmallPlan
3560
4946
  ? buildCompactFinalizePrompt(session.missingFacts)
3561
- : buildPhasePrompt(segment, session.missingFacts);
4947
+ : buildPhasePrompt(segment, session.missingFacts, session.requirementSlice ?? allRequirementIds);
3562
4948
  }
3563
4949
  }
4950
+ const atomicFocus = session.atomicRecovery ? session.missingFacts?.[0] : undefined;
4951
+ if (atomicFocus) {
4952
+ prompt = [
4953
+ compactFrontendPlanPromptForRequirementSlice(input.basePrompt, atomicFocus.requirementIds, {
4954
+ includeChecklist: false,
4955
+ ...(atomicFocus.kind === "plan-verification-target" && atomicFocus.id
4956
+ ? { verificationTargetIds: [atomicFocus.id] } : {}),
4957
+ }),
4958
+ `ATOMIC FACT: ${atomicFocus.kind}${atomicFocus.id ? ` ${atomicFocus.id}` : ""}`,
4959
+ atomicFocus.reason,
4960
+ "Commit ONLY this missing fact with one record_* call, then end this session. Other facts are queued separately. Preserve canonical IDs and shared bindings; use read_plan_facts for existing values. Do not summarize or plan the entire requirement. Capacity exhaustion is not evidence of a requirement gap.",
4961
+ ].join("\n\n");
4962
+ }
3564
4963
  const committedBefore = input.committedFactCount();
3565
- input.setActiveRequirementScope?.(session.id === "compact-local" || session.id.startsWith("ux-local-")
3566
- ? session.requirementSlice ?? []
3567
- : []);
4964
+ input.setActiveRequirementScope?.(session.coverageOnly ? session.coverageSlice ?? [] : session.requirementSlice ?? []);
3568
4965
  if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
3569
4966
  break;
3570
4967
  invocationCount += 1;
3571
- const customTools = input.segmentCustomTools(session.toolNames);
3572
- const result = await input.piStepFn({
3573
- ...input.sessionOptions,
3574
- prompt,
3575
- ...(customTools.length > 0
3576
- ? {
3577
- writerToolPolicy: {
3578
- requireSdk: true,
3579
- customTools,
3580
- },
3581
- }
3582
- : {}),
3583
- });
3584
- last = result;
3585
- try {
3586
- await input.flushLedger();
3587
- }
3588
- catch {
3589
- // best-effort: the node-level flush runs again after the attempt
4968
+ const atomicTools = {
4969
+ "plan-requirement": "record_plan_requirement", "plan-verification-target": "record_plan_verification_target",
4970
+ "state-registry": "record_state_registry", "component-choice": "record_component_choice",
4971
+ "state-flow": "record_state_flow", "data-flow": "record_data_flow", "mock-api": "record_mock_api",
4972
+ };
4973
+ const customTools = input.segmentCustomTools(atomicFocus ? new Set([atomicTools[atomicFocus.kind]]) : session.toolNames);
4974
+ const scopeForPacking = session.coverageSlice ?? session.requirementSlice;
4975
+ const envelopeBytes = Buffer.byteLength(prompt) + Buffer.byteLength(input.sessionOptions.userMessage ?? "") + Buffer.byteLength(JSON.stringify(customTools));
4976
+ if (envelopeBytes > (input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES) && scopeForPacking && session.id !== "compact-local") {
4977
+ const groups = workGroups.map(g => g.requirementIds.filter(id => scopeForPacking.includes(id))).filter(g => g.length);
4978
+ if (groups.length > 1) {
4979
+ const half = Math.ceil(groups.length / 2);
4980
+ queue.splice(index, 1, ...[groups.slice(0, half).flat(), groups.slice(half).flat()].map(slice => ({ ...session, ...(session.coverageOnly ? { coverageSlice: slice } : { requirementSlice: slice }) })));
4981
+ invocationCount -= 1;
4982
+ continue;
4983
+ }
3590
4984
  }
4985
+ const result = await observeFrontendSession({
4986
+ ...input.observation, phase: `plan/${session.id}`, scopeIds: session.requirementSlice ?? session.coverageSlice ?? allRequirementIds,
4987
+ prompt, userMessage: input.sessionOptions.userMessage, customTools, committedCount: input.committedFactCount, durableCommittedCount: () => lastDurableCount,
4988
+ artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${invocationCount}.json` : undefined,
4989
+ }, async (observer) => {
4990
+ const result = await input.piStepFn({
4991
+ ...input.sessionOptions,
4992
+ onAttemptObservation: observer,
4993
+ prompt,
4994
+ ...(customTools.length > 0
4995
+ ? {
4996
+ writerToolPolicy: {
4997
+ requireSdk: true,
4998
+ customTools,
4999
+ },
5000
+ }
5001
+ : {}),
5002
+ });
5003
+ try {
5004
+ await input.flushLedger();
5005
+ lastDurableCount = input.committedFactCount();
5006
+ }
5007
+ catch {
5008
+ // best-effort: the node-level flush runs again after the attempt
5009
+ }
5010
+ return result;
5011
+ });
5012
+ accumulated = accumulated
5013
+ ? combineSequentialPiResults(accumulated, result)
5014
+ : result;
3591
5015
  const committedAfter = input.committedFactCount();
3592
5016
  const committedFactsOnlySuccess = session.id !== "finalize" &&
3593
5017
  !(result.assistantText ?? "").trim() &&
@@ -3601,6 +5025,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3601
5025
  })
3602
5026
  : [];
3603
5027
  const isCompactLocalSession = session.id === "compact-local";
5028
+ const isUxRegistrySession = session.id === "ux-registry";
3604
5029
  const isUxLocalSession = session.id.startsWith("ux-local-");
3605
5030
  const missingPhase = isCompactLocalSession && input.committedFacts
3606
5031
  ? [
@@ -3608,6 +5033,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
3608
5033
  requirementIds: allRequirementIds,
3609
5034
  committedFacts: input.committedFacts(),
3610
5035
  }),
5036
+ ...collectFrontendPlanPhaseMissingFacts({
5037
+ phase: "ux-registry",
5038
+ requirementIds: allRequirementIds,
5039
+ committedFacts: input.committedFacts(),
5040
+ }),
3611
5041
  ...collectFrontendPlanPhaseMissingFacts({
3612
5042
  phase: "ux-local",
3613
5043
  requirementIds: allRequirementIds,
@@ -3615,20 +5045,26 @@ export async function runFrontendPlanSegmentedSessions(input) {
3615
5045
  behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3616
5046
  }),
3617
5047
  ]
3618
- : isUxLocalSession && session.requirementSlice && input.committedFacts
5048
+ : isUxRegistrySession && input.committedFacts
3619
5049
  ? collectFrontendPlanPhaseMissingFacts({
3620
- phase: "ux-local",
3621
- requirementIds: session.requirementSlice,
5050
+ phase: "ux-registry",
5051
+ requirementIds: allRequirementIds,
3622
5052
  committedFacts: input.committedFacts(),
3623
- behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3624
5053
  })
3625
- : session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
5054
+ : isUxLocalSession && session.requirementSlice && input.committedFacts
3626
5055
  ? collectFrontendPlanPhaseMissingFacts({
3627
- phase: "global-mock-data",
3628
- requirementIds: allRequirementIds,
5056
+ phase: "ux-local",
5057
+ requirementIds: session.requirementSlice,
3629
5058
  committedFacts: input.committedFacts(),
5059
+ behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3630
5060
  })
3631
- : [];
5061
+ : session.id.startsWith("global-mock-data") && allRequirementIds.length > 0 && input.committedFacts
5062
+ ? collectFrontendPlanPhaseMissingFacts({
5063
+ phase: "global-mock-data",
5064
+ requirementIds: session.requirementSlice ?? allRequirementIds,
5065
+ committedFacts: input.committedFacts(),
5066
+ })
5067
+ : [];
3632
5068
  const missingPhaseFacts = [...missingCoverage, ...missingPhase];
3633
5069
  // Frozen verification commands must operate on files the planner has
3634
5070
  // committed verification targets for; otherwise the admission writeSet
@@ -3656,11 +5092,102 @@ export async function runFrontendPlanSegmentedSessions(input) {
3656
5092
  ? buildCoveragePrompt(session.coverageSlice ?? [], missing)
3657
5093
  : isCompactLocalSession
3658
5094
  ? buildCompactLocalPrompt(missing)
3659
- : session.id === "finalize" && useCompactSmallPlan
3660
- ? buildCompactFinalizePrompt(missing)
3661
- : isUxLocalSession
3662
- ? buildUxPrompt(session.requirementSlice ?? [], missing)
3663
- : buildPhasePrompt(globalMockDataSegment, missing);
5095
+ : isUxRegistrySession
5096
+ ? buildUxRegistryPrompt(missing)
5097
+ : session.id === "finalize" && useCompactSmallPlan
5098
+ ? buildCompactFinalizePrompt(missing)
5099
+ : isUxLocalSession
5100
+ ? buildUxPrompt(session.requirementSlice ?? [], missing)
5101
+ : buildPhasePrompt(globalMockDataSegment, missing, session.requirementSlice ?? allRequirementIds);
5102
+ const recovery = classifyFrontendPlanRecovery({ ...result, stopReason: readWriterThinkingExhaustionEvidence(result).stopReason });
5103
+ if (recovery === "stop")
5104
+ return { ...result, ok: false };
5105
+ if ((recovery === "output" || (session.atomicRecovery && result.ok)) && input.committedFacts &&
5106
+ !(missingPhaseFacts.length === 1 && missingPhaseFacts[0].kind === "plan-requirement") &&
5107
+ (session.coverageOnly || isUxRegistrySession || isUxLocalSession || session.id.startsWith("global-mock-data"))) {
5108
+ if (missingPhaseFacts.length === 0) {
5109
+ // A complete validated slice does not need a successful prose turn.
5110
+ // This never accepts finalize or bypasses the final contract validator.
5111
+ accumulated = { ...accumulated, ok: true, failureCategory: "success" };
5112
+ index += 1;
5113
+ continue;
5114
+ }
5115
+ const missingIds = [...new Set(missingPhaseFacts.flatMap(item => item.requirementIds))];
5116
+ const scope = session.coverageSlice ?? session.requirementSlice;
5117
+ if (scope && missingIds.length < scope.length) {
5118
+ queue[index] = { ...session, missingFacts: missingPhaseFacts,
5119
+ ...(session.coverageOnly ? { coverageSlice: missingIds } : { requirementSlice: missingIds }),
5120
+ retryCount: 0 };
5121
+ continue;
5122
+ }
5123
+ if (scope && missingIds.length > 1 && (session.coverageOnly || isUxLocalSession)) {
5124
+ const half = Math.ceil(missingIds.length / 2);
5125
+ queue.splice(index, 1, ...[missingIds.slice(0, half), missingIds.slice(half)].map(ids => ({
5126
+ ...session, retryCount: 0,
5127
+ ...(session.coverageOnly ? { coverageSlice: ids } : { requirementSlice: ids }),
5128
+ missingFacts: missingPhaseFacts.filter(item => item.requirementIds.some(id => ids.includes(id))),
5129
+ })));
5130
+ continue;
5131
+ }
5132
+ const focusStillMissing = atomicFocus && missingPhaseFacts.some(item => item.kind === atomicFocus.kind && item.id === atomicFocus.id &&
5133
+ item.requirementIds.join("\0") === atomicFocus.requirementIds.join("\0"));
5134
+ const retries = focusStillMissing ? (session.retryCount ?? 0) + 1 : 0;
5135
+ if (retries < 2) {
5136
+ queue[index] = { ...session, atomicRecovery: true, missingFacts: missingPhaseFacts, retryCount: retries };
5137
+ continue;
5138
+ }
5139
+ return { ...accumulated, ok: false, failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
5140
+ stderr: `${accumulated.stderr}\nfrontend plan capacity recovery exhausted: preserve committed facts; phase=${session.id}; scope=${missingIds.join(",")}; strategy=atomic-fact; missing=${JSON.stringify(missingPhaseFacts)}`.trim() };
5141
+ }
5142
+ const capacityExhausted = recovery === "output" || recovery === "context";
5143
+ if (capacityExhausted && !result.timedOut) {
5144
+ const failure = () => mapPlannerExhaustion({ ...result, ok: false, stderr: `${result.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${session.id}; no smaller complete scope can finish; unchanged retries are disabled` }, committedAfter > committedBefore);
5145
+ const missingCoverageIds = input.committedFacts ? [...new Set(collectFrontendPlanMissingFacts({ requirementIds: session.coverageSlice ?? session.requirementSlice ?? allRequirementIds, committedFacts: input.committedFacts() }).flatMap(f => f.requirementIds))] : [...(session.coverageSlice ?? session.requirementSlice ?? allRequirementIds)];
5146
+ const splitScope = (ids, allowMemberSplit) => {
5147
+ const groups = workGroups.map(g => g.requirementIds.filter(id => ids.includes(id))).filter(g => g.length);
5148
+ if (groups.length > 1) {
5149
+ const half = Math.ceil(groups.length / 2);
5150
+ return [groups.slice(0, half).flat(), groups.slice(half).flat()];
5151
+ }
5152
+ if (allowMemberSplit && ids.length > 1) {
5153
+ const half = Math.ceil(ids.length / 2);
5154
+ return [ids.slice(0, half), ids.slice(half)];
5155
+ }
5156
+ return [];
5157
+ };
5158
+ if (isCompactLocalSession) {
5159
+ let scopes = splitScope(missingCoverageIds, true);
5160
+ if (!scopes.length && missingCoverageIds.length) {
5161
+ if (missingCoverageIds.length === allRequirementIds.length && committedAfter === committedBefore)
5162
+ return failure();
5163
+ scopes = [missingCoverageIds];
5164
+ }
5165
+ // UX ownership is independent of coverage completion. Preserve all groups.
5166
+ const recovery = scopes.map((slice, i) => ({ id: `coverage-capacity-${invocationCount}-${i}`, coverageOnly: true, coverageSlice: slice, toolNames: coverageSegment.toolNames, prompt: buildCoveragePrompt(slice) }));
5167
+ recovery.push({ id: "ux-registry", toolNames: uxRegistrySegment.toolNames, prompt: buildUxRegistryPrompt() });
5168
+ recovery.push(...buildWorkBatches([...allRequirementIds]).map((slice, i) => ({ id: `ux-local-capacity-${invocationCount}-${i}`, requirementSlice: slice, toolNames: uxSegment.toolNames, prompt: buildUxPrompt(slice) })));
5169
+ queue.splice(index, 1, ...recovery);
5170
+ continue;
5171
+ }
5172
+ if (session.coverageOnly || isUxLocalSession) {
5173
+ const missingIds = session.coverageOnly ? missingCoverageIds : [...new Set(missingPhase.flatMap(f => f.requirementIds))];
5174
+ if (!missingIds.length) {
5175
+ index += 1;
5176
+ continue;
5177
+ }
5178
+ const scopes = splitScope(missingIds, Boolean(session.coverageOnly));
5179
+ if (scopes.length) {
5180
+ queue.splice(index, 1, ...scopes.map(slice => ({ ...session, ...(session.coverageOnly ? { coverageSlice: slice } : { requirementSlice: slice }), missingFacts: missingPhaseFacts.filter(f => f.requirementIds.some(id => slice.includes(id))), retryCount: 0 })));
5181
+ continue;
5182
+ }
5183
+ const originalIds = session.coverageSlice ?? session.requirementSlice ?? [];
5184
+ if ((missingIds.length < originalIds.length || committedAfter > committedBefore) && (session.retryCount ?? 0) < 2) {
5185
+ queue[index] = { ...session, ...(session.coverageOnly ? { coverageSlice: missingIds } : { requirementSlice: missingIds }), missingFacts: missingPhaseFacts, retryCount: (session.retryCount ?? 0) + 1 };
5186
+ continue;
5187
+ }
5188
+ }
5189
+ return failure();
5190
+ }
3664
5191
  if (result.ok) {
3665
5192
  if (missingPhaseFacts.length === 0) {
3666
5193
  index += 1;
@@ -3676,9 +5203,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3676
5203
  continue;
3677
5204
  }
3678
5205
  return {
3679
- ...result,
5206
+ ...accumulated,
3680
5207
  ok: false,
3681
- stderr: `${result.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
5208
+ stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
3682
5209
  failureCategory: "invalid-output",
3683
5210
  };
3684
5211
  }
@@ -3695,33 +5222,15 @@ export async function runFrontendPlanSegmentedSessions(input) {
3695
5222
  }
3696
5223
  if (missingPhaseFacts.length > 0) {
3697
5224
  return {
3698
- ...result,
5225
+ ...accumulated,
3699
5226
  ok: false,
3700
- stderr: `frontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`,
5227
+ stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
3701
5228
  failureCategory: "invalid-output",
3702
5229
  };
3703
5230
  }
3704
5231
  index += 1;
3705
5232
  continue;
3706
5233
  }
3707
- // A length-stopped, fact-less compact session is the planner variant of
3708
- // writer-thinking-exhausted. Give the same scope exactly one tool-first
3709
- // retry before the split below re-batches the requirements, because one
3710
- // reinforced full-scope pass is cheaper than re-planning split halves.
3711
- const plannerThinkingBurn = isCompactLocalSession &&
3712
- committedAfter === committedBefore &&
3713
- !(result.assistantText ?? "").trim() &&
3714
- !result.stderr.trim() &&
3715
- !result.timedOut &&
3716
- readWriterThinkingExhaustionEvidence(result).stopReason === "length";
3717
- if (plannerThinkingBurn && (session.retryCount ?? 0) < 1) {
3718
- queue[index] = {
3719
- ...session,
3720
- retryCount: (session.retryCount ?? 0) + 1,
3721
- prompt: buildCompactLocalPrompt(),
3722
- };
3723
- continue;
3724
- }
3725
5234
  // Option 5: a multi-requirement coverage batch that failed with ZERO
3726
5235
  // new facts and no provider stderr is the upfront-reasoning burn —
3727
5236
  // halve the slice and retry instead of failing the attempt. The compact
@@ -3736,20 +5245,24 @@ export async function runFrontendPlanSegmentedSessions(input) {
3736
5245
  : undefined;
3737
5246
  const zeroProgressBurn = coverageSlice !== undefined &&
3738
5247
  coverageSlice.length > 1 &&
3739
- committedAfter === committedBefore &&
3740
- !(result.assistantText ?? "").trim() &&
3741
- !result.stderr.trim() &&
5248
+ (capacityExhausted || (["empty-output", "unknown"].includes(result.failureCategory) && committedAfter === committedBefore && !(result.assistantText ?? "").trim() && !result.stderr.trim())) &&
3742
5249
  !result.timedOut;
3743
5250
  if (zeroProgressBurn && coverageSlice) {
3744
- const half = Math.ceil(coverageSlice.length / 2);
3745
- const firstSlice = coverageSlice.slice(0, half);
3746
- const secondSlice = coverageSlice.slice(half);
5251
+ const missingIds = input.committedFacts ? new Set(collectFrontendPlanMissingFacts({ requirementIds: coverageSlice, committedFacts: input.committedFacts() }).flatMap(f => f.requirementIds)) : undefined;
5252
+ const unfinished = coverageSlice.filter(id => !missingIds || missingIds.has(id));
5253
+ if (unfinished.length <= 1)
5254
+ return mapPlannerExhaustion({ ...result, ok: false, stderr: `${result.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${unfinished[0] ?? session.id}; no smaller complete scope can finish` }, committedAfter > committedBefore);
5255
+ const half = Math.ceil(unfinished.length / 2);
5256
+ const firstSlice = unfinished.slice(0, half);
5257
+ const secondSlice = unfinished.slice(half);
3747
5258
  if (isCompactLocalSession) {
5259
+ queue.splice(index + 1, 0, { id: "ux-registry", toolNames: uxRegistrySegment.toolNames, prompt: buildUxRegistryPrompt() }, ...buildWorkBatches([...firstSlice, ...secondSlice]).map((slice, i) => ({ id: `ux-local-recovery-${i}`, toolNames: uxSegment.toolNames, requirementSlice: slice, prompt: buildUxPrompt(slice) })));
3748
5260
  // The split halves leave compact mode: continue them as ordinary
3749
5261
  // coverage sessions so every downstream ladder branch applies.
3750
5262
  queue.splice(index, 1, {
3751
5263
  ...session,
3752
5264
  id: `coverage-compact-split-1`,
5265
+ toolNames: coverageSegment.toolNames, retryCount: 0,
3753
5266
  coverageOnly: true,
3754
5267
  coverageSlice: firstSlice,
3755
5268
  requirementSlice: firstSlice,
@@ -3757,6 +5270,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3757
5270
  }, {
3758
5271
  ...session,
3759
5272
  id: `coverage-compact-split-2`,
5273
+ toolNames: coverageSegment.toolNames, retryCount: 0,
3760
5274
  coverageOnly: true,
3761
5275
  coverageSlice: secondSlice,
3762
5276
  requirementSlice: secondSlice,
@@ -3775,7 +5289,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3775
5289
  });
3776
5290
  continue;
3777
5291
  }
3778
- if ((session.coverageOnly || isUxLocalSession || session.id === "global-mock-data") &&
5292
+ if (readWriterThinkingExhaustionEvidence(result).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(result.failureCategory))
5293
+ return mapPlannerExhaustion({ ...result, ok: false, stderr: `${result.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${session.id}; refine the remaining complete scope; unchanged retries are disabled` }, committedAfter > committedBefore);
5294
+ if ((session.coverageOnly || isUxRegistrySession || isUxLocalSession || session.id.startsWith("global-mock-data")) &&
3779
5295
  missingPhaseFacts.length > 0 &&
3780
5296
  !result.stderr.trim() &&
3781
5297
  !result.timedOut &&
@@ -3788,11 +5304,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
3788
5304
  };
3789
5305
  continue;
3790
5306
  }
3791
- return mapPlannerExhaustion(result, committedAfter > committedBefore);
5307
+ return mapPlannerExhaustion(accumulated, committedAfter > committedBefore);
3792
5308
  }
3793
5309
  if (index < queue.length) {
3794
5310
  return {
3795
- ...(last ?? {
5311
+ ...(accumulated ?? {
3796
5312
  ok: false,
3797
5313
  stdout: "",
3798
5314
  stderr: "",
@@ -3809,11 +5325,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
3809
5325
  tokensUsed: 0,
3810
5326
  }),
3811
5327
  ok: false,
3812
- stderr: `${last?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
5328
+ stderr: `${accumulated?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
3813
5329
  failureCategory: "invalid-output",
3814
5330
  };
3815
5331
  }
3816
- return mapPlannerExhaustion(last ?? {
5332
+ return mapPlannerExhaustion(accumulated ?? {
3817
5333
  ok: false,
3818
5334
  stdout: "",
3819
5335
  stderr: "frontend plan segmentation produced no session",
@@ -3821,6 +5337,197 @@ export async function runFrontendPlanSegmentedSessions(input) {
3821
5337
  durationMs: 0,
3822
5338
  }, false);
3823
5339
  }
5340
+ /**
5341
+ * Run the two independent Scout evidence surfaces concurrently while keeping
5342
+ * their typed-event stores isolated. The main Scout store is the only store
5343
+ * visible to Plan; shard facts are merged in completion-order-independent
5344
+ * order after both sessions settle. This gives discovery real parallelism
5345
+ * without allowing sibling models to race a shared revision counter.
5346
+ */
5347
+ function aggregateParallelPiResults(results) {
5348
+ const first = results[0];
5349
+ const failed = results.find((result) => !result.ok);
5350
+ const representative = failed ?? first;
5351
+ return {
5352
+ ...representative,
5353
+ ok: failed === undefined,
5354
+ failureCategory: failed?.failureCategory ?? "success",
5355
+ durationMs: Math.max(...results.map((result) => result.durationMs), 0),
5356
+ exitCode: failed ? failed.exitCode : 0,
5357
+ stderr: results.map((result) => result.stderr).filter(Boolean).join("\n"),
5358
+ tokensUsed: results.reduce((total, result) => total + result.tokensUsed, 0),
5359
+ parsedEvents: results.reduce((total, result) => total + result.parsedEvents, 0),
5360
+ attemptedModels: [
5361
+ ...new Set(results.flatMap((result) => result.attemptedModels)),
5362
+ ],
5363
+ fallbackUsed: results.some((result) => result.fallbackUsed),
5364
+ timedOut: results.some((result) => result.timedOut),
5365
+ };
5366
+ }
5367
+ function combineSequentialPiResults(first, second) {
5368
+ return {
5369
+ ...second,
5370
+ durationMs: first.durationMs + second.durationMs,
5371
+ stderr: [first.stderr, second.stderr].filter(Boolean).join("\n"),
5372
+ tokensUsed: first.tokensUsed + second.tokensUsed,
5373
+ parsedEvents: first.parsedEvents + second.parsedEvents,
5374
+ attemptedModels: [
5375
+ ...new Set([...first.attemptedModels, ...second.attemptedModels]),
5376
+ ],
5377
+ fallbackUsed: first.fallbackUsed || second.fallbackUsed,
5378
+ timedOut: first.timedOut || second.timedOut,
5379
+ };
5380
+ }
5381
+ async function runFrontendScoutParallelSessions(input) {
5382
+ const [{ createTypedEventStore }] = await Promise.all([
5383
+ import("../workflows/dag/frontend-typed-event-store.js"),
5384
+ ]);
5385
+ const shards = [
5386
+ {
5387
+ id: "surface",
5388
+ toolName: "record_target_surface",
5389
+ instruction: [
5390
+ "PARALLEL SCOUT SHARD — target surface only.",
5391
+ "Inspect routes, entrypoints, implementation ownership, data source, and applicable test paths.",
5392
+ "Call record_target_surface exactly once with the complete runtime-evidenced surface. Do not call record_design_evidence.",
5393
+ ].join(" "),
5394
+ },
5395
+ {
5396
+ id: "design",
5397
+ toolName: "record_design_evidence",
5398
+ instruction: [
5399
+ "PARALLEL SCOUT SHARD — design evidence only.",
5400
+ "Inspect the frontend framework, styling/theme conventions, reusable components, and relevant design/spec files.",
5401
+ "Call record_design_evidence for the evidence you actually read. Do not call record_target_surface.",
5402
+ ].join(" "),
5403
+ },
5404
+ ];
5405
+ const outcomes = await Promise.all(shards.map(async (shard) => {
5406
+ let shardTools;
5407
+ let shardResult;
5408
+ try {
5409
+ const store = createTypedEventStore();
5410
+ const shardNodeId = `${input.nodeId}/parallel/${shard.id}`;
5411
+ shardTools = await createFrontendScoutEvidenceTools({
5412
+ attemptId: `${input.attemptId}:parallel:${shard.id}`,
5413
+ store,
5414
+ runDir: input.runDir,
5415
+ nodeId: shardNodeId,
5416
+ workspaceRoot: input.workspaceRoot,
5417
+ sourceDeclaredPaths: input.sourceDeclaredPaths,
5418
+ });
5419
+ const customTools = shardTools.customTools.filter((tool) => typeof tool === "object" &&
5420
+ tool !== null &&
5421
+ tool.name === shard.toolName);
5422
+ const sessionOptions = {
5423
+ ...input.sessionOptions,
5424
+ sessionEventsPath: path.join(input.runDir, shardNodeId, "session-events.jsonl"),
5425
+ };
5426
+ const prompt = `${input.basePrompt}\n\n${shard.instruction}`;
5427
+ const tools = [...customTools, ...(input.readBudgetTools ?? [])];
5428
+ shardResult = await observeFrontendSession({
5429
+ ...input.observation,
5430
+ phase: `scout/parallel/${shard.id}`,
5431
+ scopeIds: [shard.id],
5432
+ prompt,
5433
+ userMessage: sessionOptions.userMessage,
5434
+ customTools: tools,
5435
+ artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${shard.id}.json` : undefined,
5436
+ committedCount: () => shardTools?.committedFacts().length ?? 0,
5437
+ durableCommittedCount: () => shardTools?.committedFacts().length ?? 0,
5438
+ }, observer => input.piStepFn({
5439
+ ...sessionOptions,
5440
+ onAttemptObservation: observer,
5441
+ prompt,
5442
+ writerToolPolicy: {
5443
+ requireSdk: true,
5444
+ customTools: tools,
5445
+ },
5446
+ }));
5447
+ await shardTools.flush();
5448
+ return { shard, result: shardResult, tools: shardTools };
5449
+ }
5450
+ catch (error) {
5451
+ const crashMessage = `frontend scout parallel shard ${shard.id} crashed: ${error instanceof Error ? error.message : String(error)}`;
5452
+ return {
5453
+ shard,
5454
+ tools: shardTools,
5455
+ result: shardResult
5456
+ ? {
5457
+ ...shardResult,
5458
+ ok: false,
5459
+ failureCategory: shardResult.ok
5460
+ ? "invalid-output"
5461
+ : shardResult.failureCategory,
5462
+ stderr: [shardResult.stderr, crashMessage]
5463
+ .filter(Boolean)
5464
+ .join("\n"),
5465
+ }
5466
+ : {
5467
+ ok: false,
5468
+ assistantText: "",
5469
+ command: [],
5470
+ durationMs: 0,
5471
+ exitCode: null,
5472
+ failureCategory: "tool-policy",
5473
+ modelDisplay: "unknown",
5474
+ parsedEvents: 0,
5475
+ stderr: crashMessage,
5476
+ stdout: "",
5477
+ timedOut: false,
5478
+ attemptedModels: [],
5479
+ fallbackUsed: false,
5480
+ tokensUsed: 0,
5481
+ },
5482
+ };
5483
+ }
5484
+ }));
5485
+ const ordered = [...outcomes].sort((left, right) => left.shard.id.localeCompare(right.shard.id));
5486
+ try {
5487
+ for (const outcome of ordered) {
5488
+ if (outcome.result.ok && outcome.tools) {
5489
+ await input.mainTools.adoptCommittedFacts(outcome.tools.committedFacts());
5490
+ }
5491
+ }
5492
+ }
5493
+ catch (error) {
5494
+ const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
5495
+ return {
5496
+ ...aggregate,
5497
+ ok: false,
5498
+ failureCategory: "invalid-output",
5499
+ stderr: [
5500
+ aggregate.stderr,
5501
+ `frontend scout parallel fact merge failed: ${error instanceof Error ? error.message : String(error)}`,
5502
+ ]
5503
+ .filter(Boolean)
5504
+ .join("\n"),
5505
+ };
5506
+ }
5507
+ const failed = ordered.find((outcome) => !outcome.result.ok);
5508
+ if (failed) {
5509
+ const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
5510
+ return {
5511
+ ...aggregate,
5512
+ ok: false,
5513
+ stderr: `${aggregate.stderr}\nfrontend scout parallel shard failed: ${failed.shard.id}`.trim(),
5514
+ };
5515
+ }
5516
+ const committedKinds = new Set(ordered.flatMap((outcome) => (outcome.tools?.committedFacts() ?? [])
5517
+ .map((record) => record.fact?.kind)
5518
+ .filter((kind) => typeof kind === "string")));
5519
+ const missing = ["target-surface", "design-evidence"].filter((kind) => !committedKinds.has(kind));
5520
+ if (missing.length > 0) {
5521
+ const fallback = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
5522
+ return {
5523
+ ...fallback,
5524
+ ok: false,
5525
+ failureCategory: "invalid-output",
5526
+ stderr: `${fallback.stderr}\nfrontend scout parallel shards committed no ${missing.join(" or ")} fact`.trim(),
5527
+ };
5528
+ }
5529
+ return aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
5530
+ }
3824
5531
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
3825
5532
  const started = Date.now();
3826
5533
  const persona = resolveDagPiPersona(input.task);
@@ -3912,9 +5619,12 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
3912
5619
  let writerToolPolicy;
3913
5620
  let reviewTerminalTools;
3914
5621
  let designTerminalTools;
5622
+ let reviewInventory;
5623
+ let designInventory;
3915
5624
  let planLedgerTools;
3916
5625
  let contractTools;
3917
5626
  let scoutEvidenceTools;
5627
+ let scoutSourceDeclaredPaths;
3918
5628
  let readBudgetTools;
3919
5629
  const commandPolicy = resolveDagCommandPolicy(input.task.commandPolicy);
3920
5630
  const allowsPlaywrightCli = dagCommandPolicyAllows(input.task.commandPolicy, "playwright-cli");
@@ -4004,6 +5714,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4004
5714
  const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
4005
5715
  const store = createTypedEventStore();
4006
5716
  reviewTerminalTools = await createFrontendReviewTerminalTools({
5717
+ inventory: reviewInventory = await loadFrontendReviewScopes(meta.runDir, "review", meta.spec?.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES),
5718
+ inputDigest: reviewInventory?.digest ?? createHash("sha256").update(input.prompt).digest("hex"),
4007
5719
  attemptId: `${meta.runId}:${input.task.id}`,
4008
5720
  store,
4009
5721
  runDir: meta.runDir,
@@ -4029,6 +5741,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4029
5741
  const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
4030
5742
  const store = createTypedEventStore();
4031
5743
  designTerminalTools = await createFrontendDesignTerminalTools({
5744
+ inventory: designInventory = await loadFrontendReviewScopes(meta.runDir, "design", meta.spec?.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES),
5745
+ inputDigest: designInventory?.digest ?? createHash("sha256").update(input.prompt).digest("hex"),
4032
5746
  attemptId: `${meta.runId}:${input.task.id}`,
4033
5747
  store,
4034
5748
  runDir: meta.runDir,
@@ -4054,10 +5768,15 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4054
5768
  const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
4055
5769
  const store = createTypedEventStore();
4056
5770
  contractTools = await createFrontendContractTools({
5771
+ sourceDigest: (meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined),
4057
5772
  attemptId: `${meta.runId}:${input.task.id}`,
4058
5773
  store,
4059
5774
  runDir: meta.runDir,
4060
5775
  nodeId: input.task.id,
5776
+ canonicalRequirements: await resolveFrontendCanonicalRequirements({
5777
+ cwd: input.cwd,
5778
+ sourceBinding: meta.spec.sourceBinding,
5779
+ }),
4061
5780
  });
4062
5781
  writerToolPolicy = {
4063
5782
  requireSdk: true,
@@ -4078,16 +5797,19 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4078
5797
  try {
4079
5798
  const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
4080
5799
  const store = createTypedEventStore();
5800
+ scoutSourceDeclaredPaths = await resolveFrontendScoutSourceDeclaredPaths({
5801
+ cwd: input.cwd,
5802
+ spec: meta.spec,
5803
+ });
4081
5804
  scoutEvidenceTools = await createFrontendScoutEvidenceTools({
5805
+ requirementIds: parseFrontendInputBlock(input.prompt, "scout")?.payload.requirements.map(r => r.id),
5806
+ sourceDigest: (meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined),
4082
5807
  attemptId: `${meta.runId}:${input.task.id}`,
4083
5808
  store,
4084
5809
  runDir: meta.runDir,
4085
5810
  nodeId: input.task.id,
4086
5811
  workspaceRoot: input.cwd,
4087
- sourceDeclaredPaths: await resolveFrontendScoutSourceDeclaredPaths({
4088
- cwd: input.cwd,
4089
- spec: meta.spec,
4090
- }),
5812
+ sourceDeclaredPaths: scoutSourceDeclaredPaths,
4091
5813
  });
4092
5814
  writerToolPolicy = {
4093
5815
  requireSdk: true,
@@ -4120,6 +5842,13 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4120
5842
  cwd: input.cwd,
4121
5843
  sourceBinding: meta.spec.sourceBinding,
4122
5844
  }),
5845
+ declaredUiStateIds: await resolveFrontendDeclaredUiStateIds({
5846
+ runDir: meta.runDir,
5847
+ }),
5848
+ canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
5849
+ runDir: meta.runDir,
5850
+ }),
5851
+ workspaceRoot: input.cwd,
4123
5852
  });
4124
5853
  writerToolPolicy = {
4125
5854
  requireSdk: true,
@@ -4232,8 +5961,11 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4232
5961
  activeTools: toolNames,
4233
5962
  });
4234
5963
  const piSessionOptions = {
5964
+ reserveProviderRequest: input.reserveProviderRequest,
5965
+ frontendExecutionPolicy: meta.spec?.frontendExecutionPolicy,
4235
5966
  attachedFiles: [],
4236
5967
  modelConfig,
5968
+ outputLimitRecovery: { terminalToolNames: ["finalize_plan", "finalize_contract", "approve_design", "request_design_changes", "approve_review", "request_review_changes"] },
4237
5969
  repoRoot: input.cwd,
4238
5970
  step,
4239
5971
  toolNames,
@@ -4259,11 +5991,36 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4259
5991
  }
4260
5992
  : undefined,
4261
5993
  };
4262
- if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
4263
- // Frontend-only split: sequential sessions with independent
4264
- // output budgets (coverage -> UX decisions -> finalize), mirroring the
4265
- // backend-test template's module sharding. Every other template keeps
4266
- // the single-session path below.
5994
+ if (isFrontendScoutEvidenceNode(input.task) &&
5995
+ scoutEvidenceTools &&
5996
+ input.task.complexity !== "LOW") {
5997
+ result = await runFrontendScoutParallelSessions({
5998
+ piStepFn,
5999
+ sessionOptions: piSessionOptions,
6000
+ basePrompt: input.prompt,
6001
+ mainTools: scoutEvidenceTools,
6002
+ runDir: meta.runDir,
6003
+ nodeId: input.task.id,
6004
+ attemptId: `${meta.runId}:${input.task.id}`,
6005
+ workspaceRoot: input.cwd,
6006
+ sourceDeclaredPaths: scoutSourceDeclaredPaths,
6007
+ readBudgetTools: readBudgetTools?.customTools,
6008
+ observation: {
6009
+ artifactPath: path.join(meta.runDir, input.task.id, "session-budget", `attempt-${input.attempt ?? 1}`),
6010
+ runId: meta.runId,
6011
+ nodeId: input.task.id,
6012
+ taskId: meta.spec?.sourceBinding?.taskId,
6013
+ attempt: input.attempt ?? 1,
6014
+ model: input.model,
6015
+ sourceDigest: meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined,
6016
+ },
6017
+ });
6018
+ }
6019
+ else if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
6020
+ // Frontend-only split: independent coverage map sessions feed a single
6021
+ // reducer (UX decisions -> global policy -> finalize), mirroring the
6022
+ // backend-test template's module sharding. Small plans normally have one
6023
+ // coverage batch and retain the compact path below.
4267
6024
  let planRequirementIds = [];
4268
6025
  const planRequirementCosts = new Map();
4269
6026
  const behaviorRequiredRequirementIds = [];
@@ -4281,12 +6038,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4281
6038
  ?.kind === "requirement")
4282
6039
  .map((record) => record.fact?.id)
4283
6040
  .filter((id) => typeof id === "string");
4284
- for (const record of contractFacts) {
4285
- const fact = record.fact;
4286
- if (fact?.kind !== "requirement" || typeof fact.id !== "string")
4287
- continue;
4288
- const evidence = fact.evidence;
4289
- if (evidence?.behavior === "required")
6041
+ for (const fact of resolveFrontendContractRequirements(contractFacts.map((record) => record.fact))) {
6042
+ if (fact.evidence.behavior === "required")
4290
6043
  behaviorRequiredRequirementIds.push(fact.id);
4291
6044
  planRequirementCosts.set(fact.id, estimateFrontendPlanRequirementRecordCalls(fact));
4292
6045
  }
@@ -4294,33 +6047,236 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4294
6047
  catch {
4295
6048
  // Unreadable ledger falls back to a single coverage session.
4296
6049
  }
4297
- result = await runFrontendPlanSegmentedSessions({
6050
+ // Independent requirement-coverage batches are map workers. Each worker
6051
+ // owns an isolated typed-event store; only after all workers settle do we
6052
+ // merge facts into the main Plan ledger and run the single UX/global/
6053
+ // finalize reducer. This avoids revision races while shortening the
6054
+ // longest coverage phase for large plans.
6055
+ const runPlanSessions = (options) => runFrontendPlanSegmentedSessions({
6056
+ observation: options.observation ?? {
6057
+ artifactPath: path.join(meta.runDir, input.task.id, "session-budget", `attempt-${input.attempt ?? 1}`),
6058
+ runId: meta.runId,
6059
+ nodeId: input.task.id,
6060
+ taskId: meta.spec?.sourceBinding?.taskId,
6061
+ attempt: input.attempt ?? 1,
6062
+ model: input.model,
6063
+ sourceDigest: meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined,
6064
+ contractDigest: meta.spec?.taskContractBinding?.canonicalHash,
6065
+ },
4298
6066
  piStepFn,
4299
- sessionOptions: piSessionOptions,
4300
- basePrompt: input.prompt,
6067
+ sessionOptions: options.sessionOptions,
6068
+ basePrompt: options.basePrompt ?? input.prompt,
4301
6069
  attempt: input.attempt ?? 1,
4302
- committedFactCount: () => planLedgerTools.committedFactCount(),
4303
- requirementIds: planRequirementIds,
6070
+ committedFactCount: () => options.ledgerTools.committedFactCount(),
6071
+ requirementIds: options.requirementIds ?? planRequirementIds,
4304
6072
  requirementCosts: planRequirementCosts,
4305
- compactSmallPlan: true,
4306
- committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
4307
- committedFacts: () => planLedgerTools.committedFacts(),
6073
+ ...(options.parallelCoverageOnly !== undefined
6074
+ ? { parallelCoverageOnly: options.parallelCoverageOnly }
6075
+ : {}),
6076
+ ...(options.compactSmallPlan !== undefined
6077
+ ? { compactSmallPlan: options.compactSmallPlan }
6078
+ : {}),
6079
+ committedRequirementIds: () => options.ledgerTools.committedRequirementIds(),
6080
+ committedFacts: () => options.ledgerTools.committedFacts(),
4308
6081
  behaviorRequiredRequirementIds,
4309
- setActiveRequirementScope: (requirementIds) => planLedgerTools.setActiveRequirementScope(requirementIds),
6082
+ setActiveRequirementScope: (requirementIds) => options.ledgerTools.setActiveRequirementScope(requirementIds),
4310
6083
  segmentCustomTools: (toolNames) => toolNames === null
4311
- ? planLedgerTools.customTools
4312
- : planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
6084
+ ? options.ledgerTools.customTools
6085
+ : options.ledgerTools.customTools.filter((tool) => typeof tool === "object" &&
4313
6086
  tool !== null &&
4314
- toolNames.has(tool.name)),
4315
- flushLedger: () => planLedgerTools.flush(),
6087
+ (toolNames.has(tool.name) || tool.name === "read_plan_facts")),
6088
+ flushLedger: () => options.ledgerTools.flush(),
4316
6089
  });
6090
+ if (planRequirementIds.length > 1) {
6091
+ const coverageTools = FRONTEND_PLAN_SEGMENTS.find(segment => segment.id === "coverage").toolNames;
6092
+ const workload = buildFrontendPlanWorkload({
6093
+ basePrompt: input.prompt, requirementIds: planRequirementIds,
6094
+ requirementCosts: planRequirementCosts, sessionOptions: piSessionOptions,
6095
+ allTools: planLedgerTools.customTools,
6096
+ coverageTools: planLedgerTools.customTools.filter(tool => isRecordObject(tool) && coverageTools.has(String(tool.name))),
6097
+ });
6098
+ let coverageBatches;
6099
+ try {
6100
+ coverageBatches = await loadOrCreateFrontendPlanCoverageLayout({
6101
+ runDir: meta.runDir, nodeId: input.task.id, workload, requirementIds: planRequirementIds,
6102
+ binding: {
6103
+ runId: meta.runId, nodeId: input.task.id, sourceBinding: meta.spec.sourceBinding,
6104
+ skeleton: input.task.structuredContractOutput?.skeleton, writeSet: input.task.writeSet,
6105
+ policy: meta.spec.frontendExecutionPolicy, requirements: [...planRequirementCosts],
6106
+ compiledInput: workload.frozenInput, workGroups: workload.workGroups,
6107
+ },
6108
+ });
6109
+ }
6110
+ catch (error) {
6111
+ return { ok: false, stdout: "", durationMs: Date.now() - started, failureCategory: "tool-policy",
6112
+ stderr: `frontend plan coverage layout unavailable: ${error instanceof Error ? error.message : String(error)}` };
6113
+ }
6114
+ if (coverageBatches.length > 1) {
6115
+ const shardResults = await mapWithConcurrency(coverageBatches, piSessionOptions.frontendExecutionPolicy?.maxCoverageConcurrency ?? FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY, async (slice, index) => {
6116
+ let shardTools;
6117
+ let shardResult;
6118
+ try {
6119
+ const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
6120
+ shardTools = await createFrontendPlanLedgerTools({
6121
+ attemptId: `${meta.runId}:${input.task.id}:parallel:${index + 1}`,
6122
+ store: createTypedEventStore(),
6123
+ runDir: meta.runDir,
6124
+ nodeId: `${input.task.id}/parallel/coverage-${index + 1}`,
6125
+ skeleton: input.task.structuredContractOutput?.skeleton,
6126
+ sourceBinding: meta.spec.sourceBinding,
6127
+ requirementIds: slice,
6128
+ writeSetPatterns: input.task.writeSet,
6129
+ canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
6130
+ runDir: meta.runDir,
6131
+ }),
6132
+ componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
6133
+ cwd: input.cwd,
6134
+ sourceBinding: meta.spec.sourceBinding,
6135
+ }),
6136
+ });
6137
+ const shardSessionOptions = {
6138
+ ...piSessionOptions,
6139
+ sessionEventsPath: path.join(meta.runDir, input.task.id, "parallel", `coverage-${index + 1}`, "session-events.jsonl"),
6140
+ };
6141
+ shardResult = await runPlanSessions({
6142
+ ledgerTools: shardTools,
6143
+ sessionOptions: shardSessionOptions,
6144
+ requirementIds: slice,
6145
+ basePrompt: `${input.prompt}\n\nPARALLEL COVERAGE SHARD ${index + 1}: use the frozen canonical behavior verification-target ids listed in the record_plan_verification_target tool description — do not prefix ids with a shard namespace or invent variant ids; identical cross-shard targets are deduped, divergent ones fail the merge.`,
6146
+ parallelCoverageOnly: true,
6147
+ observation: {
6148
+ artifactPath: path.join(meta.runDir, input.task.id, "parallel", `coverage-${index + 1}`, "session-budget", `attempt-${input.attempt ?? 1}`),
6149
+ runId: meta.runId,
6150
+ nodeId: `${input.task.id}/parallel/coverage-${index + 1}`,
6151
+ taskId: meta.spec?.sourceBinding?.taskId,
6152
+ attempt: input.attempt ?? 1,
6153
+ model: input.model,
6154
+ sourceDigest: meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined,
6155
+ contractDigest: meta.spec?.taskContractBinding?.canonicalHash,
6156
+ },
6157
+ });
6158
+ await shardTools.flush();
6159
+ return { index, result: shardResult, tools: shardTools };
6160
+ }
6161
+ catch (error) {
6162
+ const crashMessage = `frontend plan coverage shard ${index + 1} crashed: ${error instanceof Error ? error.message : String(error)}`;
6163
+ return {
6164
+ index,
6165
+ tools: shardTools,
6166
+ result: shardResult
6167
+ ? {
6168
+ ...shardResult,
6169
+ ok: false,
6170
+ failureCategory: shardResult.ok
6171
+ ? "invalid-output"
6172
+ : shardResult.failureCategory,
6173
+ stderr: [shardResult.stderr, crashMessage]
6174
+ .filter(Boolean)
6175
+ .join("\n"),
6176
+ }
6177
+ : {
6178
+ ok: false,
6179
+ assistantText: "",
6180
+ command: [],
6181
+ durationMs: 0,
6182
+ exitCode: null,
6183
+ failureCategory: "tool-policy",
6184
+ modelDisplay: "unknown",
6185
+ parsedEvents: 0,
6186
+ stderr: crashMessage,
6187
+ stdout: "",
6188
+ timedOut: false,
6189
+ attemptedModels: [],
6190
+ fallbackUsed: false,
6191
+ tokensUsed: 0,
6192
+ },
6193
+ };
6194
+ }
6195
+ }, shard => shard.result.failureCategory === "rate-limit");
6196
+ const coverageResult = aggregateParallelPiResults(shardResults.map((shard) => shard.result));
6197
+ let mergeFailure;
6198
+ try {
6199
+ // Flatten every shard's facts FIRST, then pre-reduce: shards
6200
+ // partition requirements but a frozen verification target can
6201
+ // span shards, so same-id targets must merge across shards
6202
+ // before adoption, not per shard.
6203
+ await planLedgerTools.adoptCommittedFacts(reduceParallelCoverageShardRecords(shardResults
6204
+ .filter((item) => item.result.ok && item.tools)
6205
+ .flatMap((shard) => normalizeParallelCoverageShardRecords(shard.tools.committedFacts(), shard.index + 1))));
6206
+ }
6207
+ catch (error) {
6208
+ mergeFailure = error;
6209
+ }
6210
+ const failedShard = shardResults.find((shard) => !shard.result.ok);
6211
+ if (mergeFailure) {
6212
+ result = {
6213
+ ...coverageResult,
6214
+ ok: false,
6215
+ failureCategory: "invalid-output",
6216
+ stderr: [
6217
+ coverageResult.stderr,
6218
+ `frontend plan coverage shard merge failed: ${mergeFailure instanceof Error ? mergeFailure.message : String(mergeFailure)}`,
6219
+ ]
6220
+ .filter(Boolean)
6221
+ .join("\n"),
6222
+ };
6223
+ }
6224
+ else if (failedShard) {
6225
+ result = {
6226
+ ...coverageResult,
6227
+ ok: false,
6228
+ stderr: `${coverageResult.stderr}\nfrontend plan coverage shard ${failedShard.index + 1} failed before reduce`.trim(),
6229
+ };
6230
+ }
6231
+ else {
6232
+ const reducerResult = await runPlanSessions({
6233
+ ledgerTools: planLedgerTools,
6234
+ sessionOptions: piSessionOptions,
6235
+ });
6236
+ result = combineSequentialPiResults(coverageResult, reducerResult);
6237
+ }
6238
+ }
6239
+ else {
6240
+ result = await runPlanSessions({
6241
+ ledgerTools: planLedgerTools,
6242
+ sessionOptions: piSessionOptions,
6243
+ compactSmallPlan: true,
6244
+ });
6245
+ }
6246
+ }
6247
+ else {
6248
+ result = await runPlanSessions({
6249
+ ledgerTools: planLedgerTools,
6250
+ sessionOptions: piSessionOptions,
6251
+ compactSmallPlan: true,
6252
+ });
6253
+ }
6254
+ }
6255
+ else if (contractTools) {
6256
+ result = await runFrontendContractSegmentedSessions({
6257
+ piStepFn, sessionOptions: piSessionOptions, basePrompt: input.prompt, tools: contractTools,
6258
+ observation: { artifactPath: path.join(meta.runDir, input.task.id, "session-budget", `attempt-${input.attempt ?? 1}`), runId: meta.runId, nodeId: input.task.id, taskId: meta.spec?.sourceBinding?.taskId, attempt: input.attempt ?? 1, model: input.model, sourceDigest: meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined },
6259
+ });
6260
+ }
6261
+ else if ((reviewInventory && reviewTerminalTools) || (designInventory && designTerminalTools)) {
6262
+ result = await runFrontendReviewSegmentedSessions({ piStepFn, sessionOptions: piSessionOptions, basePrompt: input.prompt, inventory: (reviewInventory ?? designInventory), tools: (reviewTerminalTools ?? designTerminalTools), customTools: writerToolPolicy.customTools, phase: reviewInventory ? "review" : "design",
6263
+ observation: { artifactPath: path.join(meta.runDir, input.task.id, "session-budget", `attempt-${input.attempt ?? 1}`), runId: meta.runId, nodeId: input.task.id, taskId: meta.spec?.sourceBinding?.taskId, attempt: input.attempt ?? 1, model: input.model, contractDigest: (reviewInventory ?? designInventory).digest } });
6264
+ }
6265
+ else if (scoutEvidenceTools && parseFrontendInputBlock(input.prompt, "scout")) {
6266
+ result = await runFrontendScoutSegmentedSessions({ piStepFn, sessionOptions: piSessionOptions, basePrompt: input.prompt, tools: scoutEvidenceTools, customTools: writerToolPolicy.customTools,
6267
+ observation: { artifactPath: path.join(meta.runDir, input.task.id, "session-budget", `attempt-${input.attempt ?? 1}`), runId: meta.runId, nodeId: input.task.id, taskId: meta.spec?.sourceBinding?.taskId, attempt: input.attempt ?? 1, model: input.model, sourceDigest: meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined } });
4317
6268
  }
4318
6269
  else {
4319
- result = await piStepFn({
6270
+ result = await observeFrontendSession({
6271
+ artifactPath: input.task.id.startsWith("frontend-") ? path.join(meta.runDir, input.task.id, "session-budget", `attempt-${input.attempt ?? 1}.json`) : undefined,
6272
+ phase: input.task.id, prompt: input.prompt, userMessage: piSessionOptions.userMessage, customTools: writerToolPolicy?.customTools,
6273
+ runId: meta.runId, nodeId: input.task.id, taskId: meta.spec?.sourceBinding?.taskId, attempt: input.attempt ?? 1, model: input.model, sourceDigest: (meta.spec?.sourceBinding?.schemaVersion === 2 ? meta.spec.sourceBinding.inputDigest : undefined), contractDigest: meta.spec?.taskContractBinding?.canonicalHash,
6274
+ }, observer => piStepFn({
4320
6275
  ...piSessionOptions,
6276
+ onAttemptObservation: observer,
4321
6277
  prompt: input.prompt,
4322
6278
  ...(writerToolPolicy ? { writerToolPolicy } : {}),
4323
- });
6279
+ }));
4324
6280
  }
4325
6281
  try {
4326
6282
  const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.prompt);
@@ -4376,6 +6332,37 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4376
6332
  : input.task.writerOutcomePolicy
4377
6333
  ? WRITER_OUTCOME_PROTOCOL_LINE
4378
6334
  : input.task.firstProtocolLine);
6335
+ if (input.task.id === "generate-backend-md-plan-pi") {
6336
+ if (mapped.stopReason === "length") {
6337
+ return {
6338
+ ...mapped,
6339
+ ok: false,
6340
+ failureCategory: STRUCTURED_OUTPUT_RETRY_CATEGORY,
6341
+ stderr: [
6342
+ mapped.stderr,
6343
+ "backend-test Markdown plan was truncated before its required protocol could be trusted: stopReason=length; retry with the final artifact first",
6344
+ ]
6345
+ .filter(Boolean)
6346
+ .join("\n"),
6347
+ };
6348
+ }
6349
+ if (mapped.ok) {
6350
+ const protocol = assessBackendTestPlanProtocol(mapped.assistantText || mapped.stdout);
6351
+ if (!protocol.ok) {
6352
+ return {
6353
+ ...mapped,
6354
+ ok: false,
6355
+ failureCategory: "invalid-output",
6356
+ stderr: [
6357
+ mapped.stderr,
6358
+ `backend-test Markdown plan protocol invalid: ${protocol.issues.map((issue) => `${issue.code}: ${issue.detail}`).join("; ")}`,
6359
+ ]
6360
+ .filter(Boolean)
6361
+ .join("\n"),
6362
+ };
6363
+ }
6364
+ }
6365
+ }
4379
6366
  // A node-specific budget is enforced for every frontend reader that opts in,
4380
6367
  // including design/final review. Earlier code only checked contract, scout,
4381
6368
  // and plan nodes, leaving the two largest review sessions unbounded.
@@ -4454,8 +6441,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4454
6441
  try {
4455
6442
  await scoutEvidenceTools?.flush?.();
4456
6443
  }
4457
- catch {
4458
- // best-effort flush
6444
+ catch (error) {
6445
+ return { ...mapped, ok: false, failureCategory: mapped.ok ? "frontend-ledger-invalid" : mapped.failureCategory, stderr: `${mapped.stderr}\nFRONTEND_LEDGER_INTEGRITY_INVALID: ${error instanceof Error ? error.message : String(error)}` };
4459
6446
  }
4460
6447
  if (mapped.ok) {
4461
6448
  const { checkCommittedOriginFacts, readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
@@ -4829,7 +6816,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4829
6816
  runDir: meta.runDir,
4830
6817
  progress,
4831
6818
  attempt: input.attempt ?? 1,
4832
- maxAttempts: input.task.retryPolicy?.maxAttempts ?? 3,
6819
+ maxAttempts: input.task.retryPolicy?.maxAttempts ?? 5,
4833
6820
  });
4834
6821
  if (progress.status !== "PASS") {
4835
6822
  const classified = classifyBackendTestWriterCompletenessFailure(progress);
@@ -4879,7 +6866,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4879
6866
  stderrParts.push(completenessFailure.detail);
4880
6867
  }
4881
6868
  if (writerThinkingExhausted) {
4882
- stderrParts.push(`${WRITER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 write tool calls, 0 attributed diff; recommend a model switch and a new run`);
6869
+ stderrParts.push(`${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before any write tool call; retry from the existing workspace and complete only unfinished targets`);
4883
6870
  }
4884
6871
  if (meta.writeGuardAttribution === "best-effort") {
4885
6872
  stderrParts.push("write guard note: concurrent rank writers use best-effort per-node attribution; keep same-rank writeSet entries disjoint");
@@ -4952,14 +6939,11 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4952
6939
  rawFailureCategory === "context-overflow" &&
4953
6940
  (changeManifestChangedFiles?.length ?? 0) > 0
4954
6941
  ? "partial-success-with-context-overflow"
4955
- : // writer-thinking-exhausted: a length-stopped thinking-only attempt with
4956
- // zero write tool calls and zero attributed diff is terminal and
4957
- // non-retryable; recommend a model switch + fresh run. Only applies on
4958
- // the empty-output base category so provider/transport failures keep
4959
- // their original category, and only when the completeness gate did not
4960
- // upgrade to incomplete-write-set above.
6942
+ : // output-limit: a length-stopped attempt is capacity truncation, not
6943
+ // empty output. Preserve the raw category for diagnostics and let the
6944
+ // node retry from committed facts/current workspace state.
4961
6945
  writerThinkingExhausted
4962
- ? WRITER_THINKING_EXHAUSTED_CATEGORY
6946
+ ? OUTPUT_LIMIT_RETRY_CATEGORY
4963
6947
  : writerCleanTimeout
4964
6948
  ? WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY
4965
6949
  : // writer-budget-exhausted: the provider session consumed an