@tea-agent/loop-agent 0.42.0-next.14 → 0.42.0-next.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/CHANGELOG.md +19 -3
  2. package/dist/application/task-lifecycle/advance.js +9 -4
  3. package/dist/build-stamp.json +3 -3
  4. package/dist/commands/task-source-prepare.js +3 -1
  5. package/dist/executors/dag-pi-executor.js +1437 -154
  6. package/dist/executors/shell-executor.js +107 -35
  7. package/dist/shared/dag-failure-category.js +6 -0
  8. package/dist/task/contract/apply.js +36 -2
  9. package/dist/task/source-prepare/parse-intent.js +7 -0
  10. package/dist/worker/console/chat/pi-runtime.js +45 -2
  11. package/dist/worker/console/chat/resource-loader.js +4 -1
  12. package/dist/worker/console/chat/routes.js +8 -0
  13. package/dist/worker/console/chat/subagents/agent-tool.js +62 -0
  14. package/dist/worker/console/chat/subagents/explore-agent.js +169 -0
  15. package/dist/worker/console/chat/subagents/index.js +5 -0
  16. package/dist/worker/console/chat/subagents/orchestrator.js +273 -0
  17. package/dist/worker/console/chat/subagents/tool-policy.js +53 -0
  18. package/dist/worker/console/chat/subagents/types.js +13 -0
  19. package/dist/worker/console/chat/tool-preview.js +10 -0
  20. package/dist/worker/console/chat/tools.js +11 -1
  21. package/dist/worker/console/interview/tools.js +1 -0
  22. package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-BGMbAdAG.js → abnfDiagram-N423BO3Z-CgXb0EVO.js} +1 -1
  23. package/dist/worker/console/static/assets/{arc-CJu1LND8.js → arc-DN59MZqN.js} +1 -1
  24. package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-BmJh9gMQ.js → architectureDiagram-T3A2C74G-BUk3sWpn.js} +1 -1
  25. package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-BWjaX3ZF.js → blockDiagram-VBNYF7ZC-IH-cPBFE.js} +1 -1
  26. package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-CLEOAxLB.js → c4Diagram-5PPSVZJV-BJWf1nGo.js} +1 -1
  27. package/dist/worker/console/static/assets/channel-i1DjpDIw.js +1 -0
  28. package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-CydnyQUX.js → chunk-2GRJ4B5K-BCPb5H0y.js} +1 -1
  29. package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-tjJ1IoZ1.js → chunk-2Q5K7J3B-Cfs3jeW2.js} +1 -1
  30. package/dist/worker/console/static/assets/{chunk-5RXB4S5H-DR358Phd.js → chunk-5RXB4S5H-BhQI1B_T.js} +1 -1
  31. package/dist/worker/console/static/assets/{chunk-5VM5RSS4-B-LuWFZD.js → chunk-5VM5RSS4-Cc_m4gyV.js} +1 -1
  32. package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-FG53fA6v.js → chunk-6Q2QTUOP-JPqDS3IV.js} +1 -1
  33. package/dist/worker/console/static/assets/{chunk-GF5L2VYU-Bkt-w8SZ.js → chunk-GF5L2VYU-AeGX6EAg.js} +1 -1
  34. package/dist/worker/console/static/assets/{chunk-JWPE2WC7-WxHo0z2I.js → chunk-JWPE2WC7-pEyTjekV.js} +1 -1
  35. package/dist/worker/console/static/assets/{chunk-KBJHAD2P-CqHrBJhp.js → chunk-KBJHAD2P-Bi5UopKd.js} +1 -1
  36. package/dist/worker/console/static/assets/{chunk-RYQCIY6F-D4MVae_f.js → chunk-RYQCIY6F-FkGFJxgQ.js} +1 -1
  37. package/dist/worker/console/static/assets/{chunk-XXDRQBXY-CamBh0Ll.js → chunk-XXDRQBXY-D-7LGrV2.js} +1 -1
  38. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-DgDRdE1V.js +1 -0
  39. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-DgDRdE1V.js +1 -0
  40. package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-C3qheCuW.js → cose-bilkent-JH36ORCC-CCEDNDaX.js} +1 -1
  41. package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-DQlWfkS7.js → cynefin-VYW2F7L2-DDd-t7fm.js} +1 -1
  42. package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-BAxa9rFb.js → cynefinDiagram-MW4NZA55-LWe4Ikzr.js} +1 -1
  43. package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-YM09nLJw.js → dagre-VZM6K2ZE-BBCZScc-.js} +1 -1
  44. package/dist/worker/console/static/assets/{diagram-7IWD3JNH-I4W9EJj2.js → diagram-7IWD3JNH-B6liRVig.js} +1 -1
  45. package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-DH8eJ8Ff.js → diagram-B4RE2ZJO-DO_hIrpW.js} +1 -1
  46. package/dist/worker/console/static/assets/{diagram-LBJQPF4R-D0D0ntDW.js → diagram-LBJQPF4R-dmCy93uR.js} +1 -1
  47. package/dist/worker/console/static/assets/{diagram-Q27KOJAE-BIISKq1s.js → diagram-Q27KOJAE-D6mFTIxP.js} +1 -1
  48. package/dist/worker/console/static/assets/{diagram-UB23O5K3--xyAWj0w.js → diagram-UB23O5K3-DCteVUfY.js} +1 -1
  49. package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-B3sYXLJ4.js → ebnfDiagram-BXEA7PRR-DTv530VK.js} +1 -1
  50. package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-zRSYWhJP.js → erDiagram-JOGREHBK-DCXDMxCr.js} +1 -1
  51. package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-BUYlie03.js → flowDiagram-UKHOOZJN-BwEBLOJh.js} +1 -1
  52. package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-DP4kzzrM.js → ganttDiagram-PKOTCBZU-Ck-Vymjm.js} +1 -1
  53. package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-CAPoUxTF.js → gitGraphDiagram-DS77QQ5N-Hqs6X_3L.js} +1 -1
  54. package/dist/worker/console/static/assets/{index-CWhTyxvc.js → index-BmMi-Bve.js} +71 -71
  55. package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-BXPfjekV.js → infoDiagram-6WML65LV-Cjc6M9eg.js} +1 -1
  56. package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-aOI3pV6t.js → ishikawaDiagram-WSZJBQD7-CPchYZMl.js} +1 -1
  57. package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-DAPZuvi4.js → journeyDiagram-NVQOT4AX-BABdJNNC.js} +1 -1
  58. package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-atwDzL3C.js → kanban-definition-27J2QSJJ-1oXhbM4j.js} +1 -1
  59. package/dist/worker/console/static/assets/{linear-D5E09qji.js → linear-PTmQ9LkV.js} +1 -1
  60. package/dist/worker/console/static/assets/{mermaid.core-CE4idY-s.js → mermaid.core-B6Lxil0V.js} +5 -5
  61. package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-CbZxtPca.js → mindmap-definition-FAOFIHXS-BUqjNGEw.js} +1 -1
  62. package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-De-FYEte.js → pegDiagram-VL7TDLO6-8D1N-orJ.js} +1 -1
  63. package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-Bs0USobz.js → pieDiagram-7S7Q4E2Y-BlPMN9d9.js} +1 -1
  64. package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-Bq3ZrGCJ.js → quadrantDiagram-CIZ2JOQS-BIL5YDes.js} +1 -1
  65. package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-Bk8ojyIj.js → railroadDiagram-AXF67PYL-Dqmil3ie.js} +1 -1
  66. package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-CGhoyws1.js → requirementDiagram-LRYGKXZP-C_-Dp4mH.js} +1 -1
  67. package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-DmVSQs3b.js → sankeyDiagram-W5VNT64P-C0s5-bQS.js} +1 -1
  68. package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-C80h1u0Q.js → sequenceDiagram-SI44F4Z6-DiVCHPfi.js} +1 -1
  69. package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-CcbksQVZ.js → sizeCapture-X5ZJPWSS-CxQ9WVvx.js} +1 -1
  70. package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-CFTTTQKj.js → stateDiagram-OKZ733FA-D9mFHA-Y.js} +1 -1
  71. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-BqXQxfFC.js +1 -0
  72. package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-Cw7Jr7Tg.js → swimlanes-SLNWSIFB-DDnC5l-M.js} +2 -2
  73. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-CXwWNRJl.js +8 -0
  74. package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-CVcdTVa5.js → timeline-definition-Z64GVDOM-C8QySPC-.js} +1 -1
  75. package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-BpLbmWp9.js → vennDiagram-T6HMQDX7-BeYdDiz6.js} +1 -1
  76. package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-6PXfbwj-.js → wardleyDiagram-T6FBY63Y-D8hjFWLH.js} +1 -1
  77. package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-Bjkcz4ih.js → xychartDiagram-ELKLHX3M-ClqG_JGK.js} +1 -1
  78. package/dist/worker/console/static/index.html +1 -1
  79. package/dist/worker/console/static-src/operator-chat/tools-catalog.js +21 -2
  80. package/dist/workflows/dag/dag-retry-schema.js +3 -0
  81. package/dist/workflows/dag/frontend-contract-facts.js +130 -0
  82. package/dist/workflows/dag/frontend-design-policy.js +100 -15
  83. package/dist/workflows/dag/frontend-implementation-contract.js +56 -8
  84. package/dist/workflows/dag/frontend-plan-render.js +13 -2
  85. package/dist/workflows/dag/frontend-risk.js +2 -0
  86. package/dist/workflows/dag/frontend-shadow-dual-write.js +17 -1
  87. package/dist/workflows/dag/frontend-shape.js +16 -6
  88. package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
  89. package/dist/workflows/dag/frontend-typed-event-store.js +3 -0
  90. package/dist/workflows/dag/frontend-verification-trace.js +32 -0
  91. package/dist/workflows/dag/init-hybrid.js +16 -13
  92. package/dist/workflows/dag/node-execution.js +178 -88
  93. package/dist/workflows/dag/rerun-feedback.js +1 -0
  94. package/dist/workflows/dag/rerun-plan.js +90 -4
  95. package/dist/workflows/dag/rerun-run.js +7 -0
  96. package/dist/workflows/dag/retry-policy.js +16 -10
  97. package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
  98. package/dist/workflows/dag/runner.js +67 -7
  99. package/dist/workflows/dag/structured-output-repair.js +4 -1
  100. package/dist/workflows/dag/validate.js +10 -8
  101. package/dist/workflows/dag/workspace-checkpoint.js +66 -0
  102. package/docs/operations/README.md +1 -0
  103. package/docs/templates/agent-dag.schema.json +2 -2
  104. package/docs/templates/frontend-implementation-contract.schema.json +34 -2
  105. package/package.json +2 -3
  106. package/skills/frontend-plan/SKILL.md +14 -1
  107. package/skills/frontend-plan/references/decision-contract.md +84 -10
  108. package/skills/frontend-plan/references/design-decisions.md +32 -0
  109. package/dist/worker/console/static/assets/channel-DOsaUk1k.js +0 -1
  110. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-D4-D49vY.js +0 -1
  111. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-D4-D49vY.js +0 -1
  112. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-C4YCesmi.js +0 -1
  113. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-lN2U_DLl.js +0 -8
@@ -12,11 +12,14 @@ import { createPiReadBudgetCustomTools, } from "./pi-read-budget-policy.js";
12
12
  import { cleanupPlaywrightCliDefaultSession, createPlaywrightCliTool, PI_COMMAND_CAPABILITY_REGISTRY, resolveCaseIdFromWriteSet, resolveEvidenceDirFromWriteSet, } from "./pi-playwright-cli-tool.js";
13
13
  import { dagCommandPolicyAllows, resolveDagCommandPolicy, } from "../workflows/dag/types.js";
14
14
  import { parseLedgerJson } from "../task/source-prepare/ledger.js";
15
+ import { splitDocumentIntoFragments, sha256Text, } from "../task/source-prepare/fragment-inventory.js";
15
16
  import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation.js";
16
17
  import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
17
18
  import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
18
19
  import { pathMatchesPattern } from "../shared/git-progress.js";
19
- import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
20
+ import { readTypedEventStoreFromJsonl, typedEventPayloadSha256, } from "../workflows/dag/frontend-typed-event-store.js";
21
+ import { collectCanonicalStateFlowNames, frontendEvidenceExpectationSchema, resolveFrontendContractRequirements, validateFrontendRequiredDeliverables, } from "../workflows/dag/frontend-contract-facts.js";
22
+ import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
20
23
  import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestCollectionRepairOutcomeRecoveryCandidate, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
21
24
  import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
22
25
  import { assessBackendTestPlanProtocol } from "../workflows/dag/backend-test-plan-protocol.js";
@@ -24,23 +27,221 @@ import { frontendTestLayoutFromSpec } from "../workflows/dag/frontend-test-layou
24
27
  import { redactSecrets, truncateUtf8Preview } from "../shared/preview.js";
25
28
  import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.js";
26
29
  /**
27
- * Writer classification for a length-stopped thinking-only attempt: the model
28
- * exhausted its output budget thinking but never issued a write/edit tool call
29
- * and produced zero attributed diff. This is a terminal, non-retryable
30
- * diagnosis (the same prompt + model will hit the same budget wall); the
31
- * recommendation is a model switch plus a fresh run. It must NOT mask a
32
- * recoverable partial-write-set (incomplete-write-set) upgrade.
30
+ * Legacy diagnostic label retained for artifact compatibility. New executions
31
+ * classify this signal as output-limit so the node can retry incrementally.
33
32
  */
34
33
  export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
35
34
  /**
36
- * Planner classification mirroring writer-thinking-exhausted: a read-only
37
- * planning session stopped on length, observed thinking, and committed zero
38
- * typed facts with no assistant text. By the time this survives the segmented
39
- * ladder the scope has already been degraded, so the durable fix is a
40
- * thinking-capped or non-thinking model for the tier — not another replay of
41
- * the same full-scope prompt.
35
+ * Legacy diagnostic label retained for artifact compatibility. New executions
36
+ * classify this signal as output-limit and preserve committed typed facts.
42
37
  */
43
38
  export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
39
+ /**
40
+ * Parallel coverage shards namespace their verification target ids with
41
+ * `VT-SHARD-<shard>-` so the reducer can detect cross-shard conflicts. Once
42
+ * every shard's facts are known the namespace must be restored: the canonical
43
+ * contract has to carry the exact ids the task source froze (a surviving
44
+ * `VT-SHARD-N-…` id is a contract-requirement gap that design review
45
+ * rejects). Stripping happens per shard before adoption — after the strip,
46
+ * the existing identity-conflict check catches genuine cross-shard
47
+ * duplicate base ids and fails the merge closed.
48
+ */
49
+ /**
50
+ * Deterministic pre-reduce for parallel coverage shard facts. Shards partition
51
+ * requirements, but a frozen verification target can legitimately be derived
52
+ * by several shards (one VT covers multiple ACs across shard boundaries), so
53
+ * same-id verification targets are MERGED: requirementIds and uiStates union,
54
+ * while divergent file/commandId is a real conflict. Everything else passes
55
+ * through unchanged.
56
+ */
57
+ export function reduceParallelCoverageShardRecords(records) {
58
+ // Resolve replacements inside each source shard before comparing shards.
59
+ // A correction is an event-log operation, not a second independent target;
60
+ // retaining both records makes a valid same-shard correction look like a
61
+ // cross-shard conflict during the later reduction.
62
+ const byShard = new Map();
63
+ for (const record of records) {
64
+ const shard = byShard.get(record.attemptId) ?? [];
65
+ shard.push(record);
66
+ byShard.set(record.attemptId, shard);
67
+ }
68
+ const effectiveRecords = [];
69
+ for (const shardRecords of byShard.values()) {
70
+ const effective = [];
71
+ for (const record of shardRecords) {
72
+ const fact = record.fact;
73
+ const entry = fact?.entry;
74
+ const kind = typeof fact?.kind === "string" ? fact.kind : "";
75
+ const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
76
+ typeof entry?.id === "string"
77
+ ? `${kind}:${entry.id}`
78
+ : undefined;
79
+ if (!identity) {
80
+ effective.push(record);
81
+ continue;
82
+ }
83
+ const replaces = typeof fact?.replaces === "string" ? `${kind}:${fact.replaces}` : undefined;
84
+ if (replaces) {
85
+ for (let index = effective.length - 1; index >= 0; index -= 1) {
86
+ const prior = effective[index];
87
+ const priorFact = prior.fact;
88
+ const priorEntry = priorFact?.entry;
89
+ const priorIdentity = typeof priorFact?.kind === "string" &&
90
+ typeof priorEntry?.id === "string"
91
+ ? `${priorFact.kind}:${priorEntry.id}`
92
+ : undefined;
93
+ if (priorIdentity === replaces)
94
+ effective.splice(index, 1);
95
+ }
96
+ }
97
+ // Keep the source record immutable. The replacement is represented by
98
+ // the latest event and its own payload hash.
99
+ effective.push(record);
100
+ }
101
+ effectiveRecords.push(...effective);
102
+ }
103
+ const verificationTargets = new Map();
104
+ const merged = [];
105
+ for (const record of effectiveRecords) {
106
+ const fact = record.fact;
107
+ if (!fact || typeof fact !== "object")
108
+ continue;
109
+ if (fact.kind !== "plan-verification-target") {
110
+ merged.push(record);
111
+ continue;
112
+ }
113
+ const entry = fact.entry;
114
+ if (!entry || typeof entry.id !== "string")
115
+ continue;
116
+ const existing = verificationTargets.get(entry.id);
117
+ if (!existing) {
118
+ verificationTargets.set(entry.id, {
119
+ record, entry: { ...entry },
120
+ sources: [`${record.attemptId}:${record.eventId}:${record.payloadSha256}`],
121
+ });
122
+ continue;
123
+ }
124
+ // Never mutate the entry held by the source record. The reducer creates
125
+ // a new merged entry and recomputes the payload hash for that derived
126
+ // fact, leaving source-event replay integrity intact.
127
+ const baseEntry = { ...existing.entry };
128
+ for (const field of ["commandId", "commandLabel", "file", "scope"]) {
129
+ if (entry[field] !== baseEntry[field]) {
130
+ throw new Error(`frontend plan coverage shard reduce conflict: verification target ${entry.id} has divergent ${field} "${String(entry[field])}" vs "${String(baseEntry[field])}"`);
131
+ }
132
+ }
133
+ const unionSorted = (a, b) => {
134
+ const left = Array.isArray(a) ? a : [];
135
+ const right = Array.isArray(b) ? b : [];
136
+ return [...new Set([...left, ...right])].sort();
137
+ };
138
+ baseEntry.requirementIds = unionSorted(baseEntry.requirementIds, entry.requirementIds);
139
+ baseEntry.uiStates = unionSorted(baseEntry.uiStates, entry.uiStates);
140
+ const mergedFact = { ...fact, entry: { ...baseEntry } };
141
+ existing.entry = baseEntry;
142
+ existing.sources.push(`${record.attemptId}:${record.eventId}:${record.payloadSha256}`);
143
+ existing.record = {
144
+ ...existing.record,
145
+ fact: mergedFact,
146
+ payloadSha256: typedEventPayloadSha256(mergedFact),
147
+ };
148
+ }
149
+ merged.push(...[...verificationTargets.values()].map((item) => {
150
+ // The reducer owns every VT revision, including a one-shard partial
151
+ // result. Content-addressed revisions replay idempotently, while an
152
+ // expanded/replaced source set appends an explicit replacement rather
153
+ // than masquerading as a changed source event or deleting audit facts.
154
+ const fact = {
155
+ ...item.record.fact,
156
+ replaces: item.entry.id,
157
+ entry: {
158
+ ...item.entry,
159
+ requirementIds: [...new Set(planFactStringList(item.entry.requirementIds))].sort(),
160
+ uiStates: [...new Set(planFactStringList(item.entry.uiStates))].sort(),
161
+ },
162
+ };
163
+ const payloadSha256 = typedEventPayloadSha256(fact);
164
+ const revisionId = createHash("sha256")
165
+ .update(JSON.stringify({ sources: [...new Set(item.sources)].sort(), payloadSha256 }))
166
+ .digest("hex");
167
+ return {
168
+ ...item.record,
169
+ attemptId: "frontend-plan-coverage-reducer",
170
+ eventId: `coverage-reduced:${revisionId}`,
171
+ requestId: `coverage-reduced:${revisionId}`,
172
+ fact, payloadSha256,
173
+ };
174
+ }));
175
+ return merged;
176
+ }
177
+ export function normalizeParallelCoverageShardRecords(records, shardNumber) {
178
+ const prefix = `VT-SHARD-${shardNumber}-`;
179
+ // Models sometimes drop the `VT-` stem when applying the shard namespace
180
+ // (observed: frozen `VT-X` became `VT-SHARD-2-X`), so restoring the
181
+ // canonical id requires re-adding the stem after the strip.
182
+ const stripId = (id) => {
183
+ if (typeof id !== "string" || !id.startsWith(prefix))
184
+ return id;
185
+ const base = id.slice(prefix.length);
186
+ return base.startsWith("VT-") ? base : `VT-${base}`;
187
+ };
188
+ const stripIdList = (ids) => Array.isArray(ids) ? ids.map((id) => stripId(id)) : ids;
189
+ return records.map((record) => {
190
+ const fact = record.fact;
191
+ if (!fact || typeof fact !== "object")
192
+ return record;
193
+ const kind = fact.kind;
194
+ const entry = fact.entry;
195
+ if (!entry || typeof entry !== "object")
196
+ return record;
197
+ let rewritten;
198
+ const replaces = kind === "plan-verification-target" ? stripId(fact.replaces) : fact.replaces;
199
+ if (kind === "plan-verification-target") {
200
+ const id = stripId(entry.id);
201
+ if (id !== entry.id || replaces !== fact.replaces)
202
+ rewritten = { ...entry, id };
203
+ }
204
+ else if (kind === "plan-requirement") {
205
+ const verificationTargetIds = stripIdList(entry.verificationTargetIds);
206
+ if (verificationTargetIds !== entry.verificationTargetIds) {
207
+ rewritten = { ...entry, verificationTargetIds };
208
+ }
209
+ }
210
+ else if (kind === "state-flow") {
211
+ const rewriteBoundTargets = (item) => {
212
+ if (!item || typeof item !== "object")
213
+ return item;
214
+ return {
215
+ ...item,
216
+ verificationTargetIds: stripIdList(item.verificationTargetIds),
217
+ };
218
+ };
219
+ const uiStates = Array.isArray(entry.uiStates)
220
+ ? entry.uiStates.map(rewriteBoundTargets)
221
+ : entry.uiStates;
222
+ const interactions = Array.isArray(entry.interactions)
223
+ ? entry.interactions.map(rewriteBoundTargets)
224
+ : entry.interactions;
225
+ if (uiStates !== entry.uiStates || interactions !== entry.interactions) {
226
+ rewritten = { ...entry, uiStates, interactions };
227
+ }
228
+ }
229
+ return rewritten
230
+ ? {
231
+ ...record,
232
+ fact: { ...fact, ...(replaces !== undefined ? { replaces } : {}), entry: rewritten },
233
+ // Keep the integrity hash consistent with the rewritten
234
+ // payload, or the adoption replay check reports the
235
+ // normalized record as a tampered source event.
236
+ payloadSha256: typedEventPayloadSha256({
237
+ ...fact,
238
+ ...(replaces !== undefined ? { replaces } : {}),
239
+ entry: rewritten,
240
+ }),
241
+ }
242
+ : record;
243
+ });
244
+ }
44
245
  export function isPlannerThinkingExhausted(result, committedAnyFacts) {
45
246
  if (result.ok)
46
247
  return false;
@@ -251,6 +452,8 @@ export const FRONTEND_CONTRACT_RECORD_TOOL_NAMES = [
251
452
  "record_handoff_intent",
252
453
  "record_open_question",
253
454
  "record_split_proposal",
455
+ "record_ui_state",
456
+ "record_required_deliverables",
254
457
  ];
255
458
  export const FRONTEND_CONTRACT_TERMINAL_TOOL_NAMES = [
256
459
  "finalize_contract",
@@ -264,6 +467,7 @@ export const FRONTEND_SCOUT_EVIDENCE_TOOL_NAMES = [
264
467
  export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
265
468
  "record_route_selection",
266
469
  "record_component_choice",
470
+ "record_state_registry",
267
471
  "record_state_flow",
268
472
  "record_data_flow",
269
473
  "record_mock_api",
@@ -907,8 +1111,111 @@ async function resolveFrontendPlanNewComponentSourceReferences(input) {
907
1111
  }
908
1112
  }
909
1113
  /**
910
- * A+B: `frontend-plan-pi` now records its decision ledger through seven
911
- * incremental `record_*` tools (origin=plan) and closes with exactly one
1114
+ * Frozen canonical requirements keyed by id, resolved from the source-fidelity
1115
+ * ledger before the contract node starts. record_requirement commits these
1116
+ * runtime-owned values so the model can never rewrite authoritative requirement
1117
+ * text or stringify the fragment bindings.
1118
+ */
1119
+ async function resolveFrontendCanonicalRequirements(input) {
1120
+ const binding = input.sourceBinding;
1121
+ if (!binding || binding.schemaVersion !== 2 || !binding.ledgerPath) {
1122
+ return new Map();
1123
+ }
1124
+ const absolutePath = path.resolve(input.cwd, binding.ledgerPath);
1125
+ const workspaceRoot = path.resolve(input.cwd);
1126
+ if (absolutePath !== workspaceRoot &&
1127
+ !absolutePath.startsWith(`${workspaceRoot}${path.sep}`)) {
1128
+ throw new Error("canonical requirement ledger escapes workspace");
1129
+ }
1130
+ try {
1131
+ const raw = await readFile(absolutePath, "utf8");
1132
+ if (sha256Text(raw) !== binding.ledgerSha256) {
1133
+ throw new Error("source ledger is stale");
1134
+ }
1135
+ const ledger = parseLedgerJson(raw);
1136
+ const sourceFragments = new Map();
1137
+ const boundFragments = new Map(ledger.fragments.map((fragment) => [fragment.id, fragment]));
1138
+ for (const sourcePath of new Set(ledger.fragments.map((fragment) => fragment.path))) {
1139
+ const sourceAbsolute = path.resolve(workspaceRoot, sourcePath);
1140
+ if (!sourceAbsolute.startsWith(`${workspaceRoot}${path.sep}`)) {
1141
+ throw new Error("source fragment escapes workspace");
1142
+ }
1143
+ const content = await readFile(sourceAbsolute, "utf8");
1144
+ for (const fragment of splitDocumentIntoFragments({ path: sourcePath, content })) {
1145
+ const frozen = boundFragments.get(fragment.id);
1146
+ if (frozen && frozen.sha256 === sha256Text(fragment.text)) {
1147
+ sourceFragments.set(fragment.id, fragment.text);
1148
+ }
1149
+ }
1150
+ }
1151
+ return new Map(ledger.canonicalRequirements.map((requirement) => [
1152
+ requirement.id,
1153
+ {
1154
+ text: requirement.text,
1155
+ sourceFragmentIds: [...(requirement.sourceFragmentIds ?? [])],
1156
+ sourceFragments,
1157
+ },
1158
+ ]));
1159
+ }
1160
+ catch (error) {
1161
+ throw new Error(`canonical requirement sources unavailable: ${error instanceof Error ? error.message : String(error)}`);
1162
+ }
1163
+ }
1164
+ /**
1165
+ * Authoritative UI state ids declared by the contract node
1166
+ * (ui-state-declaration facts). Empty when the source declares no UI-state
1167
+ * table, in which case the plan's state vocabulary is registry-only.
1168
+ */
1169
+ async function resolveFrontendDeclaredUiStateIds(input) {
1170
+ try {
1171
+ const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
1172
+ return records.flatMap((record) => {
1173
+ const fact = record.fact;
1174
+ return record.phase === "committed" &&
1175
+ fact.kind === "ui-state-declaration" &&
1176
+ typeof fact.id === "string"
1177
+ ? [fact.id]
1178
+ : [];
1179
+ });
1180
+ }
1181
+ catch {
1182
+ // No contract facts (or no declared states): registry-only vocabulary.
1183
+ return [];
1184
+ }
1185
+ }
1186
+ /**
1187
+ * Resolve the frozen canonical behavior verification-target ids the PRD
1188
+ * declares. The requirement text is the same authority the design review reads
1189
+ * when it rejects `VT-SHARD-N-…` / variant ids as a contract-requirement gap,
1190
+ * so extracting `VT-…` tokens from the frozen contract requirement facts makes
1191
+ * that authority deterministic instead of prose-only. An empty result (no PRD
1192
+ * declared any behavior target id) disables the canonical check and keeps the
1193
+ * historical free-form path.
1194
+ */
1195
+ export async function resolveFrontendCanonicalVerificationTargetIds(input) {
1196
+ try {
1197
+ const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
1198
+ const ids = new Set();
1199
+ for (const record of records) {
1200
+ const fact = record.fact;
1201
+ if (record.phase !== "committed" || fact.kind !== "requirement") {
1202
+ continue;
1203
+ }
1204
+ const text = typeof fact.text === "string" ? fact.text : "";
1205
+ for (const match of text.matchAll(/\bVT-[A-Z0-9][A-Z0-9_-]*\b/g)) {
1206
+ ids.add(match[0]);
1207
+ }
1208
+ }
1209
+ return [...ids].sort();
1210
+ }
1211
+ catch {
1212
+ // No contract facts (or unreadable): fall back to no canonical set.
1213
+ return [];
1214
+ }
1215
+ }
1216
+ /**
1217
+ * A+B: `frontend-plan-pi` records its decision ledger through incremental
1218
+ * `record_*` tools (origin=plan) and closes with exactly one
912
1219
  * `finalize_plan` terminal. A later attempt may explicitly adopt a quarantined
913
1220
  * fact via `adopt_staged_fact`. Flush writes `plan-typed-facts.jsonl` for the
914
1221
  * node validator / compile authority.
@@ -1042,6 +1349,12 @@ export async function createFrontendPlanLedgerTools(input) {
1042
1349
  line: Type.Optional(Type.Number({})),
1043
1350
  }, { additionalProperties: false })),
1044
1351
  rationale: Type.String({}),
1352
+ covers: Type.Optional(Type.Array(Type.String({}), {
1353
+ description: "UI state and/or interaction names this single component choice covers (one choice may cover many ids).",
1354
+ })),
1355
+ evidencePath: Type.Optional(Type.String({
1356
+ description: "REQUIRED for decision=reuse-existing: repo-relative path whose existing file is the reuse evidence. Greenfield paths must use decision=new.",
1357
+ })),
1045
1358
  }, { additionalProperties: false });
1046
1359
  const stringList = (value) => Array.isArray(value)
1047
1360
  ? value.filter((item) => typeof item === "string")
@@ -1120,7 +1433,7 @@ export async function createFrontendPlanLedgerTools(input) {
1120
1433
  const recordComponentChoiceTool = defineTool({
1121
1434
  name: "record_component_choice",
1122
1435
  label: "record_component_choice",
1123
- description: "Record ONE component choice (origin=plan component-choice fact). Declare every UI purpose's component selection. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo (e.g. reusing ActiveRunBadge's styling convention). Omit rationale for reuse-existing; it is optional. Call up to 5 component choices per assistant message (batching reduces API round trips and rate-limit risk); never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\"}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"}",
1436
+ description: "Record ONE component choice (origin=plan component-choice fact). One choice may cover MULTIPLE UI states/interactions via covers: [\"<state-or-interaction name>\", ...]; do not emit one row per interaction. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo and REQUIRES evidencePath: a repo-relative path to the existing file that proves the reuse — the runtime verifies the file exists (fresh evidence); a path with no existing file is greenfield and must use decision=new instead. Call up to 5 component choices per assistant message; never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\", \"covers\": [\"<other interaction names this component also serves>\"]}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"} or {\"choice\": {\"purpose\": \"FocusQueuePanel\", \"component\": \"FocusQueuePanel\", \"decision\": \"reuse-existing\", \"evidencePath\": \"src/journal.ts\"}}",
1124
1437
  promptSnippet: "Record 1-5 component choices (up to 5 per message).",
1125
1438
  parameters: Type.Object({
1126
1439
  choice: uiComponentChoiceSchema,
@@ -1171,6 +1484,46 @@ export async function createFrontendPlanLedgerTools(input) {
1171
1484
  ...(citation.line !== undefined ? { line: citation.line } : {}),
1172
1485
  };
1173
1486
  }
1487
+ if (choice.decision === "reuse-existing") {
1488
+ // reuse-existing must cite a file that actually exists NOW (fresh
1489
+ // existence evidence, same semantics as Scout pathEvidence.fresh).
1490
+ // A path with no file on disk is greenfield: decision=new is the
1491
+ // only honest choice (dogfood run dag-1788504923861-0f7b17a9
1492
+ // claimed 18 reuse-existing conventions in a not-yet-written
1493
+ // src/planner.ts and every gate let it through).
1494
+ const evidencePath = typeof choice.evidencePath === "string"
1495
+ ? choice.evidencePath.trim()
1496
+ : "";
1497
+ if (!evidencePath) {
1498
+ return planToolReceipt({
1499
+ ok: false,
1500
+ kind: "component-choice",
1501
+ error: "decision=reuse-existing requires evidencePath naming the existing repo file that proves the reuse; if the file does not exist yet, use decision=new",
1502
+ });
1503
+ }
1504
+ if (input.workspaceRoot) {
1505
+ const absolute = path.resolve(input.workspaceRoot, evidencePath);
1506
+ const workspaceRoot = path.resolve(input.workspaceRoot);
1507
+ const contained = absolute === workspaceRoot ||
1508
+ absolute.startsWith(`${workspaceRoot}${path.sep}`);
1509
+ let exists = false;
1510
+ if (contained) {
1511
+ try {
1512
+ exists = (await stat(absolute)).isFile();
1513
+ }
1514
+ catch {
1515
+ exists = false;
1516
+ }
1517
+ }
1518
+ if (!exists) {
1519
+ return planToolReceipt({
1520
+ ok: false,
1521
+ kind: "component-choice",
1522
+ error: `decision=reuse-existing evidencePath "${evidencePath}" has no fresh existence evidence (file not found in the workspace); reuse requires an existing file — use decision=new for greenfield paths`,
1523
+ });
1524
+ }
1525
+ }
1526
+ }
1174
1527
  const components = typeof choice.component === "string" ? [choice.component] : [];
1175
1528
  const result = await adoptPlanFact("component-choice", `${attemptId}:record_component_choice:${randomUUID()}`, {
1176
1529
  kind: "component-choice",
@@ -1194,10 +1547,86 @@ export async function createFrontendPlanLedgerTools(input) {
1194
1547
  return planToolReceipt(echo);
1195
1548
  },
1196
1549
  });
1550
+ // Global UX vocabulary: one compact registry committed BEFORE any state
1551
+ // flow details. Coverage may slice by AC; UX must not — the registry is
1552
+ // the anti-duplication anchor that keeps every later slice on the same
1553
+ // named concepts instead of re-inventing them per AC chunk.
1554
+ const recordStateRegistryTool = defineTool({
1555
+ name: "record_state_registry",
1556
+ label: "record_state_registry",
1557
+ description: "Commit the GLOBAL UX vocabulary (origin=plan state-registry fact) BEFORE any record_state_flow call: uiStateNames (use the contract's declared authoritative state ids when provided) and interactionNames (stable behavior-domain kebab-case names, e.g. planner-task-edit / focus-queue-move — one name per behavior domain, never one per AC). Empty arrays explicitly mean the request has no UI state or interaction vocabulary. Re-record the full vocabulary to correct it; the latest commit wins, but it may not remove names still referenced by committed state-flow facts. Example: {\"uiStateNames\": [\"planner-empty\"], \"interactionNames\": [\"planner-task-create\", \"focus-queue-move\"]}",
1558
+ promptSnippet: "Record the global UI-state/interaction vocabulary once, before any state flow.",
1559
+ parameters: Type.Object({
1560
+ uiStateNames: stringArray,
1561
+ interactionNames: stringArray,
1562
+ }, { additionalProperties: false }),
1563
+ async execute(_toolCallId, params) {
1564
+ const uiStateNames = [...new Set(stringList(params?.uiStateNames).map((name) => name.trim()))];
1565
+ const interactionNames = [
1566
+ ...new Set(stringList(params?.interactionNames).map((name) => name.trim())),
1567
+ ];
1568
+ if (uiStateNames.some((name) => name.length === 0) || interactionNames.some((name) => name.length === 0)) {
1569
+ return planToolReceipt({
1570
+ ok: false,
1571
+ kind: "state-registry",
1572
+ error: "record_state_registry names must be non-empty strings",
1573
+ });
1574
+ }
1575
+ const invalidInteractionNames = interactionNames.filter((name) => !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(name));
1576
+ if (invalidInteractionNames.length > 0) {
1577
+ return planToolReceipt({
1578
+ ok: false,
1579
+ kind: "state-registry",
1580
+ error: `record_state_registry interactionNames must be stable kebab-case behavior-domain names: ${invalidInteractionNames.join(", ")}`,
1581
+ });
1582
+ }
1583
+ if (input.declaredUiStateIds &&
1584
+ input.declaredUiStateIds.length > 0) {
1585
+ const declared = new Set(input.declaredUiStateIds);
1586
+ const undeclared = uiStateNames.filter((name) => !declared.has(name));
1587
+ if (undeclared.length > 0) {
1588
+ return planToolReceipt({
1589
+ ok: false,
1590
+ kind: "state-registry",
1591
+ error: `record_state_registry uiStateNames are not declared by the contract's authoritative UI-state table: ${undeclared.join(", ")}; use the declared ids (${input.declaredUiStateIds.join(", ")})`,
1592
+ });
1593
+ }
1594
+ const missing = input.declaredUiStateIds.filter((name) => !uiStateNames.includes(name));
1595
+ if (missing.length > 0) {
1596
+ return planToolReceipt({
1597
+ ok: false,
1598
+ kind: "state-registry",
1599
+ error: `record_state_registry must include every state from the contract's authoritative UI-state table; missing: ${missing.join(", ")}`,
1600
+ });
1601
+ }
1602
+ }
1603
+ // A correction is deterministic only when the new last-wins registry
1604
+ // still contains every live name already committed by state-flow facts.
1605
+ // This permits adding a missed concept, while preventing a registry edit
1606
+ // from retroactively orphaning earlier slices.
1607
+ const liveNames = collectCanonicalStateFlowNames(readCommittedEvents(store, attemptId));
1608
+ const orphanedUiStates = [...liveNames.uiStateNames].filter((name) => !uiStateNames.includes(name));
1609
+ const orphanedInteractions = [...liveNames.interactionNames].filter((name) => !interactionNames.includes(name));
1610
+ if (orphanedUiStates.length > 0 || orphanedInteractions.length > 0) {
1611
+ return planToolReceipt({
1612
+ ok: false,
1613
+ kind: "state-registry",
1614
+ error: `record_state_registry cannot remove names still referenced by committed state-flow facts (uiStates: ${orphanedUiStates.join(", ") || "none"}; interactions: ${orphanedInteractions.join(", ") || "none"}); first correct/remove those state-flow entries, then re-record the full registry`,
1615
+ });
1616
+ }
1617
+ const result = await adoptPlanFact("state-registry", `${attemptId}:record_state_registry:${randomUUID()}`, {
1618
+ kind: "state-registry",
1619
+ origin: "plan",
1620
+ uiStateNames,
1621
+ interactionNames,
1622
+ });
1623
+ return planToolReceipt(result);
1624
+ },
1625
+ });
1197
1626
  const recordStateFlowTool = defineTool({
1198
1627
  name: "record_state_flow",
1199
1628
  label: "record_state_flow",
1200
- description: "Record UI states and interactions as an origin=plan state-flow fact. To correct stale named entries from an earlier UX batch, include removeUiStateNames and/or removeInteractionNames. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
1629
+ description: "Record UI states and interactions as an origin=plan state-flow fact. REQUIRES a committed record_state_registry vocabulary first, and every name here must be in that registry; uiState names must also be contract-declared authoritative ids when the contract declares them. To correct stale named entries from an earlier UX batch, include removeUiStateNames and/or removeInteractionNames. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
1201
1630
  promptSnippet: "Record the plan state-flow fact.",
1202
1631
  parameters: Type.Object({
1203
1632
  uiStates: Type.Array(uiStateSchema),
@@ -1301,9 +1730,63 @@ export async function createFrontendPlanLedgerTools(input) {
1301
1730
  }
1302
1731
  interactions.push({ ...interaction, name: resolvedName });
1303
1732
  }
1733
+ // UX vocabulary gate: every recorded state/interaction name must be
1734
+ // declared in the committed global registry first. This keeps UX out
1735
+ // of AC-number slicing — the model commits one compact vocabulary
1736
+ // (record_state_registry), then fills behavior-domain details, and a
1737
+ // later slice cannot silently rename an earlier concept (dogfood
1738
+ // dag-1788504923861-0f7b17a9: 5 renames of "prioritize" + 7 of
1739
+ // "focus queue" across AC chunks).
1304
1740
  const states = uiStates
1305
1741
  .map((state) => (typeof state?.name === "string" ? state.name : ""))
1306
1742
  .filter(Boolean);
1743
+ const committedRegistry = readCommittedEvents(store, attemptId)
1744
+ .map((event) => event.fact)
1745
+ .filter((fact) => isRecordObject(fact) && fact.kind === "state-registry")
1746
+ .at(-1);
1747
+ if (!committedRegistry) {
1748
+ return planToolReceipt({
1749
+ ok: false,
1750
+ kind: "state-flow",
1751
+ error: "record_state_flow requires a committed UX registry first: call record_state_registry with the full uiStateNames/interactionNames vocabulary, then record state flows against it",
1752
+ });
1753
+ }
1754
+ const registryUiStateNames = new Set(stringList(committedRegistry.uiStateNames));
1755
+ const registryInteractionNames = new Set(stringList(committedRegistry.interactionNames));
1756
+ const unknownStates = states.filter((name) => !registryUiStateNames.has(name));
1757
+ if (unknownStates.length > 0) {
1758
+ return planToolReceipt({
1759
+ ok: false,
1760
+ kind: "state-flow",
1761
+ error: `record_state_flow uiState names not in the committed registry: ${unknownStates.join(", ")}; re-record record_state_registry with the complete vocabulary first (registered: ${[...registryUiStateNames].join(", ") || "(none)"})`,
1762
+ });
1763
+ }
1764
+ const unknownInteractions = interactions
1765
+ .map((interaction) => typeof interaction?.name === "string" ? interaction.name : "")
1766
+ .filter((name) => name && !registryInteractionNames.has(name));
1767
+ if (unknownInteractions.length > 0) {
1768
+ return planToolReceipt({
1769
+ ok: false,
1770
+ kind: "state-flow",
1771
+ error: `record_state_flow interaction names not in the committed registry: ${unknownInteractions.join(", ")}; re-record record_state_registry with the complete vocabulary first (registered: ${[...registryInteractionNames].join(", ") || "(none)"})`,
1772
+ });
1773
+ }
1774
+ // Authoritative UI states: when the contract node declared the
1775
+ // source's UI-state table, planner states must bind those ids — no
1776
+ // invented variants (planner-empty/create-invalid/no-results/
1777
+ // editing/focus-full/storage-unavailable, not "planner-list").
1778
+ if (input.declaredUiStateIds &&
1779
+ input.declaredUiStateIds.length > 0) {
1780
+ const declared = new Set(input.declaredUiStateIds);
1781
+ const undeclaredStates = states.filter((name) => !declared.has(name));
1782
+ if (undeclaredStates.length > 0) {
1783
+ return planToolReceipt({
1784
+ ok: false,
1785
+ kind: "state-flow",
1786
+ error: `record_state_flow uiState names are not declared by the contract's authoritative UI-state table: ${undeclaredStates.join(", ")}; use the declared ids (${input.declaredUiStateIds.join(", ")}), or record a design deviation if the source table is genuinely incomplete`,
1787
+ });
1788
+ }
1789
+ }
1307
1790
  const result = await adoptPlanFact("state-flow", `${attemptId}:record_state_flow:${randomUUID()}`, {
1308
1791
  kind: "state-flow",
1309
1792
  origin: "plan",
@@ -1462,6 +1945,15 @@ export async function createFrontendPlanLedgerTools(input) {
1462
1945
  });
1463
1946
  }
1464
1947
  }
1948
+ if (id &&
1949
+ activeRequirementScope.length > 0 &&
1950
+ !activeRequirementScope.includes(id)) {
1951
+ return planToolReceipt({
1952
+ ok: false,
1953
+ kind: "plan-requirement",
1954
+ error: `record_plan_requirement id "${id}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
1955
+ });
1956
+ }
1465
1957
  const result = await adoptPlanFact("plan-requirement", `${attemptId}:record_plan_requirement:${randomUUID()}`, {
1466
1958
  kind: "plan-requirement",
1467
1959
  origin: "plan",
@@ -1474,7 +1966,11 @@ export async function createFrontendPlanLedgerTools(input) {
1474
1966
  const recordPlanVerificationTargetTool = defineTool({
1475
1967
  name: "record_plan_verification_target",
1476
1968
  label: "record_plan_verification_target",
1477
- description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}",
1969
+ description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}" +
1970
+ (input.canonicalVerificationTargetIds &&
1971
+ input.canonicalVerificationTargetIds.length > 0
1972
+ ? ` Frozen canonical behavior target ids (use exactly for behavior targets): ${input.canonicalVerificationTargetIds.join(", ")}.`
1973
+ : ""),
1478
1974
  promptSnippet: "Commit 1-4 plan verification target entries (up to 4 per message).",
1479
1975
  parameters: Type.Object({
1480
1976
  entry: verificationTargetSchema,
@@ -1525,6 +2021,24 @@ export async function createFrontendPlanLedgerTools(input) {
1525
2021
  error: `record_plan_verification_target verification-target-phase-mismatch: behavior command "${directoryEntry.label}" (${directoryEntry.commandId}) must bind a test file (__tests__/, tests?/, e2e/, cypress/, *.test.*, *.spec.*, *.cy.*); received file "${rawEntry.file}"`,
1526
2022
  });
1527
2023
  }
2024
+ // Canonical-identity check for behavior targets: the PRD freezes
2025
+ // the exact behavior verification-target ids (e.g.
2026
+ // VT-SMOKE-COUNTER-BEHAVIOR). A committed non-canonical id is
2027
+ // immutable and design review rejects it as a
2028
+ // contract-requirement gap, so reject invented ids here.
2029
+ if (directoryEntry.mode === "behavior" &&
2030
+ input.canonicalVerificationTargetIds &&
2031
+ input.canonicalVerificationTargetIds.length > 0) {
2032
+ const canonicalTargetId = typeof rawEntry.id === "string" ? rawEntry.id.trim() : "";
2033
+ if (canonicalTargetId &&
2034
+ !input.canonicalVerificationTargetIds.includes(canonicalTargetId)) {
2035
+ return planToolReceipt({
2036
+ ok: false,
2037
+ kind: "plan-verification-target",
2038
+ error: `record_plan_verification_target id "${canonicalTargetId}" is not a frozen canonical behavior verification target; canonical ids are: ${input.canonicalVerificationTargetIds.join(", ")}`,
2039
+ });
2040
+ }
2041
+ }
1528
2042
  }
1529
2043
  // Duplicate-id rejection: committed typed facts are immutable, so
1530
2044
  // re-recording the same VT id would deadlock the compile by default.
@@ -1594,6 +2108,15 @@ export async function createFrontendPlanLedgerTools(input) {
1594
2108
  : [];
1595
2109
  }));
1596
2110
  const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
2111
+ const outOfScopeRequirementIds = stringList(rawEntry.requirementIds).filter((id) => activeRequirementScope.length > 0 &&
2112
+ !activeRequirementScope.includes(id));
2113
+ if (outOfScopeRequirementIds.length > 0) {
2114
+ return planToolReceipt({
2115
+ ok: false,
2116
+ kind: "plan-verification-target",
2117
+ error: `record_plan_verification_target references requirements outside this session's scope [${activeRequirementScope.join(", ")}]: ${outOfScopeRequirementIds.join(", ")}`,
2118
+ });
2119
+ }
1597
2120
  if (unknownRequirementIds.length > 0) {
1598
2121
  return planToolReceipt({
1599
2122
  ok: false,
@@ -1640,6 +2163,18 @@ export async function createFrontendPlanLedgerTools(input) {
1640
2163
  error: "record_plan_evidence_gap requires a non-empty description describing the gap",
1641
2164
  });
1642
2165
  }
2166
+ const requirementId = typeof entry.requirementId === "string"
2167
+ ? entry.requirementId
2168
+ : undefined;
2169
+ if (requirementId &&
2170
+ activeRequirementScope.length > 0 &&
2171
+ !activeRequirementScope.includes(requirementId)) {
2172
+ return planToolReceipt({
2173
+ ok: false,
2174
+ kind: "plan-evidence-gap",
2175
+ error: `record_plan_evidence_gap requirementId "${requirementId}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
2176
+ });
2177
+ }
1643
2178
  const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
1644
2179
  return planToolReceipt(result);
1645
2180
  },
@@ -1672,6 +2207,29 @@ export async function createFrontendPlanLedgerTools(input) {
1672
2207
  ? { realIntegrationGap: params.realIntegrationGap }
1673
2208
  : {}),
1674
2209
  };
2210
+ const latestRegistry = [...committed]
2211
+ .reverse()
2212
+ .map((record) => record.fact)
2213
+ .find((fact) => isRecordObject(fact) && fact.kind === "state-registry");
2214
+ if (latestRegistry) {
2215
+ const registryUiStates = new Set(stringList(latestRegistry.uiStateNames));
2216
+ const registryInteractions = new Set(stringList(latestRegistry.interactionNames));
2217
+ const live = collectCanonicalStateFlowNames(committed);
2218
+ const missingUiStates = [...registryUiStates].filter((name) => !live.uiStateNames.has(name));
2219
+ const missingInteractions = [...registryInteractions].filter((name) => !live.interactionNames.has(name));
2220
+ const undeclaredUiStates = [...live.uiStateNames].filter((name) => !registryUiStates.has(name));
2221
+ const undeclaredInteractions = [...live.interactionNames].filter((name) => !registryInteractions.has(name));
2222
+ if (missingUiStates.length > 0 ||
2223
+ missingInteractions.length > 0 ||
2224
+ undeclaredUiStates.length > 0 ||
2225
+ undeclaredInteractions.length > 0) {
2226
+ return planToolReceipt({
2227
+ ok: false,
2228
+ kind: "finalize_plan",
2229
+ error: `finalize_plan UX registry mismatch: every registered name must have one live state-flow entry and every live entry must be registered (missing uiStates: ${missingUiStates.join(", ") || "none"}; missing interactions: ${missingInteractions.join(", ") || "none"}; undeclared uiStates: ${undeclaredUiStates.join(", ") || "none"}; undeclared interactions: ${undeclaredInteractions.join(", ") || "none"})`,
2230
+ });
2231
+ }
2232
+ }
1675
2233
  // Front-load the node's compile + policy gates into the finalize
1676
2234
  // receipt (same pipeline the design-policy shell and the node
1677
2235
  // self-check run: merge the patch onto the runtime skeleton,
@@ -1698,9 +2256,10 @@ export async function createFrontendPlanLedgerTools(input) {
1698
2256
  catch (error) {
1699
2257
  if (error instanceof PlanPolicyPrecheckFailure) {
1700
2258
  // Template the fix: every uncovered interaction / state
1701
- // maps to a ready-to-submit record_component_choice
1702
- // call. One reuse-existing choice covers all
1703
- // behavioural interactions.
2259
+ // maps to a record_component_choice skeleton. Reuse is
2260
+ // never implied: the operator/model must fill an existing
2261
+ // evidencePath or switch the choice to decision=new with
2262
+ // the required frozen source citation.
1704
2263
  const suggestions = error.findings
1705
2264
  .filter((finding) => finding.code === "ui-design-coverage-missing" &&
1706
2265
  finding.path)
@@ -1711,6 +2270,7 @@ export async function createFrontendPlanLedgerTools(input) {
1711
2270
  purpose: finding.path,
1712
2271
  component: "<name the existing or new component>",
1713
2272
  decision: "reuse-existing",
2273
+ evidencePath: "<existing repo file that proves this reuse>",
1714
2274
  },
1715
2275
  },
1716
2276
  }));
@@ -1796,6 +2356,7 @@ export async function createFrontendPlanLedgerTools(input) {
1796
2356
  customTools: [
1797
2357
  recordRouteSelectionTool,
1798
2358
  recordComponentChoiceTool,
2359
+ recordStateRegistryTool,
1799
2360
  recordStateFlowTool,
1800
2361
  recordDataFlowTool,
1801
2362
  recordMockApiTool,
@@ -1807,6 +2368,66 @@ export async function createFrontendPlanLedgerTools(input) {
1807
2368
  adoptStagedFactTool,
1808
2369
  finalizePlanTool,
1809
2370
  ],
2371
+ adoptCommittedFacts: async (records) => {
2372
+ for (const record of records) {
2373
+ if (record.phase !== "committed")
2374
+ continue;
2375
+ const fact = record.fact;
2376
+ if (!fact || typeof fact !== "object" || Array.isArray(fact))
2377
+ continue;
2378
+ const kind = typeof fact.kind === "string"
2379
+ ? fact.kind
2380
+ : "plan-fact";
2381
+ const entry = fact.entry;
2382
+ const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
2383
+ typeof entry?.id === "string"
2384
+ ? `${kind}:${entry.id}`
2385
+ : undefined;
2386
+ const sourceShardPrefix = `${attemptId}:parallel-merge:${record.attemptId}:`;
2387
+ const requestId = `${sourceShardPrefix}${record.eventId}`;
2388
+ const committed = readCommittedEvents(store, attemptId);
2389
+ const replay = committed.find((candidate) => candidate.requestId === requestId);
2390
+ if (replay) {
2391
+ if (replay.payloadSha256 !== record.payloadSha256) {
2392
+ throw new Error(`frontend plan shard fact merge replay conflict: source event ${record.eventId} changed payload`);
2393
+ }
2394
+ continue;
2395
+ }
2396
+ const explicitReplacement = typeof fact.replaces === "string" &&
2397
+ fact.replaces === entry?.id;
2398
+ const conflicting = committed.find((candidate) => {
2399
+ const candidateFact = candidate.fact;
2400
+ const candidateEntry = candidateFact.entry;
2401
+ const sameSourceShard = candidate.requestId.startsWith(sourceShardPrefix);
2402
+ return (candidateFact.kind === kind &&
2403
+ typeof candidateEntry?.id === "string" &&
2404
+ `${kind}:${candidateEntry.id}` === identity &&
2405
+ !(sameSourceShard && explicitReplacement));
2406
+ });
2407
+ if (conflicting) {
2408
+ // Two shards may honestly emit the same fact (e.g. both
2409
+ // derive the same frozen verification target). Identical
2410
+ // payloads are duplicates to skip; divergent payloads are
2411
+ // a real conflict the ladder must resolve.
2412
+ if (conflicting.payloadSha256 === record.payloadSha256) {
2413
+ continue;
2414
+ }
2415
+ const fromSameSourceShard = conflicting.requestId.startsWith(sourceShardPrefix);
2416
+ if (fromSameSourceShard) {
2417
+ throw new Error(`frontend plan shard fact merge conflict: ${identity} was already committed by another shard with a different payload`);
2418
+ }
2419
+ // Divergent payload from a different source attempt means a
2420
+ // ladder retry re-derived the coverage phase: the fresh
2421
+ // derivation supersedes the stored record.
2422
+ const storeWithRecords = store;
2423
+ storeWithRecords.records = storeWithRecords.records.filter((candidate) => candidate.eventId !== conflicting.eventId);
2424
+ }
2425
+ const result = await adoptPlanFact(kind, requestId, fact);
2426
+ if (!result.ok) {
2427
+ throw new Error(`frontend plan shard fact merge failed: ${result.error}`);
2428
+ }
2429
+ }
2430
+ },
1810
2431
  setActiveRequirementScope: (requirementIds) => {
1811
2432
  activeRequirementScope = [
1812
2433
  ...new Set(requirementIds.filter((id) => id.trim().length > 0)),
@@ -1845,6 +2466,10 @@ function contractFactSourceSpan(fact) {
1845
2466
  if (Array.isArray(fact.sourceRefs) && fact.sourceRefs.length > 0) {
1846
2467
  return fact.sourceRefs;
1847
2468
  }
2469
+ if (Array.isArray(fact.sourceFragmentIds) &&
2470
+ fact.sourceFragmentIds.some((value) => typeof value === "string" && value.trim().length > 0)) {
2471
+ return fact.sourceFragmentIds;
2472
+ }
1848
2473
  if (typeof fact.source === "string" && fact.source.trim().length > 0) {
1849
2474
  return fact.source;
1850
2475
  }
@@ -1894,6 +2519,7 @@ export function buildFrontendTaskContractVNext(records) {
1894
2519
  requirements,
1895
2520
  constraints: byKind("constraint"),
1896
2521
  evidenceExpectations: byKind("evidence-expectation"),
2522
+ deliverableDeclarations: byKind("required-deliverables").at(-1)?.items ?? [],
1897
2523
  handoffIntents: byKind("handoff-intent"),
1898
2524
  openQuestions: byKind("open-question"),
1899
2525
  splitProposals: byKind("split-proposal"),
@@ -1901,7 +2527,7 @@ export function buildFrontendTaskContractVNext(records) {
1901
2527
  };
1902
2528
  }
1903
2529
  /**
1904
- * A+B: `frontend-contract-pi` incremental contract tools. Six `record_*` tools
2530
+ * A+B: `frontend-contract-pi` incremental contract tools. The `record_*` tools
1905
2531
  * commit origin=contract facts and `finalize_contract` commits the terminal
1906
2532
  * disposition (ready | ready-with-assumptions | blocked + blockingOwner).
1907
2533
  */
@@ -1953,9 +2579,7 @@ export async function createFrontendContractTools(input) {
1953
2579
  }
1954
2580
  }
1955
2581
  const recordKinds = {
1956
- record_requirement: "requirement",
1957
2582
  record_constraint: "constraint",
1958
- record_evidence_expectation: "evidence-expectation",
1959
2583
  record_handoff_intent: "handoff-intent",
1960
2584
  record_open_question: "open-question",
1961
2585
  record_split_proposal: "split-proposal",
@@ -1975,6 +2599,141 @@ export async function createFrontendContractTools(input) {
1975
2599
  return receipt(result);
1976
2600
  },
1977
2601
  }));
2602
+ // record_requirement is runtime-owned by design: the requirement TEXT and
2603
+ // sourceFragmentIds come from the frozen source-fidelity ledger, never from
2604
+ // model-authored prose. The model only names the canonical id it confirms.
2605
+ // This closes the free-shape hole (additionalProperties:true accepted
2606
+ // `statement` rewrites and JSON-stringified `sourceFragmentIds` arrays,
2607
+ // which then reached the planner as empty text + dead fragment bindings).
2608
+ const recordRequirementTool = defineTool({
2609
+ name: "record_requirement",
2610
+ label: "record_requirement",
2611
+ description: 'Confirm one canonical ledger requirement (origin=contract requirement fact). Pass ONLY the canonical id listed in the <frontend_contract_input> inventory, e.g. {"id": "AC-001"} — the runtime commits the authoritative text and sourceFragmentIds from the frozen ledger. Never pass text/statement/sourceFragmentIds yourself: free-form rewrites and JSON-stringified fragment arrays are rejected. IMPORTANT: batch up to 5 record_* calls per message, starting from the FIRST message.',
2612
+ promptSnippet: "Confirm 1-5 canonical requirements by id (up to 5 per message).",
2613
+ parameters: Type.Object({ id: Type.String({ description: "Canonical ledger requirement id (e.g. AC-001)" }) }, { additionalProperties: false }),
2614
+ async execute(_toolCallId, params) {
2615
+ const id = typeof params?.id === "string" ? params.id.trim() : "";
2616
+ if (!id) {
2617
+ return receipt({
2618
+ ok: false,
2619
+ kind: "requirement",
2620
+ error: "record_requirement requires the canonical requirement id",
2621
+ });
2622
+ }
2623
+ const canonical = input.canonicalRequirements?.get(id);
2624
+ if (!canonical) {
2625
+ const known = [...(input.canonicalRequirements?.keys() ?? [])];
2626
+ return receipt({
2627
+ ok: false,
2628
+ kind: "requirement",
2629
+ error: `record_requirement id "${id}" is not a canonical ledger requirement; canonical ids are: ${known.join(", ") || "(none)"}`,
2630
+ });
2631
+ }
2632
+ const result = await adoptContractFact("requirement", {
2633
+ kind: "requirement",
2634
+ origin: "contract",
2635
+ disposition: "explicit",
2636
+ id,
2637
+ text: canonical.text,
2638
+ sourceFragmentIds: canonical.sourceFragmentIds,
2639
+ });
2640
+ return receipt(result);
2641
+ },
2642
+ });
2643
+ // Authoritative UI state declarations: the contract node extracts the
2644
+ // PRD/reference UI-state table into structured facts so the planner binds
2645
+ // uiStates to declared ids instead of inventing names (dogfood
2646
+ // dag-1788504923861-0f7b17a9: 10 invented list-visibility variants, 0 of
2647
+ // the 7 PRD states).
2648
+ const recordUiStateTool = defineTool({
2649
+ name: "record_ui_state",
2650
+ label: "record_ui_state",
2651
+ description: 'Declare ONE authoritative UI state extracted from the task source\'s UI-state table (origin=contract ui-state-declaration fact). Call once per declared state, exactly as the source names it: {"id": "<state id from the source table>", "trigger": "<when this state applies>", "observableOutcome": "<what the user can observe>"}. The planner must bind these ids later; do not rename or invent states.',
2652
+ promptSnippet: "Declare 1-5 authoritative UI states (up to 5 per message).",
2653
+ parameters: Type.Object({
2654
+ id: Type.String({ description: "UI state id exactly as the source table declares it" }),
2655
+ trigger: Type.String({ description: "When this state applies" }),
2656
+ observableOutcome: Type.String({ description: "Observable result for the user" }),
2657
+ }, { additionalProperties: false }),
2658
+ async execute(_toolCallId, params) {
2659
+ const id = typeof params?.id === "string" ? params.id.trim() : "";
2660
+ const trigger = typeof params?.trigger === "string" ? params.trigger.trim() : "";
2661
+ const observableOutcome = typeof params?.observableOutcome === "string"
2662
+ ? params.observableOutcome.trim()
2663
+ : "";
2664
+ if (!id || !trigger || !observableOutcome) {
2665
+ return receipt({
2666
+ ok: false,
2667
+ kind: "ui-state-declaration",
2668
+ error: "record_ui_state requires non-empty id, trigger, and observableOutcome",
2669
+ });
2670
+ }
2671
+ const result = await adoptContractFact("ui-state-declaration", {
2672
+ kind: "ui-state-declaration",
2673
+ origin: "contract",
2674
+ id,
2675
+ trigger,
2676
+ observableOutcome,
2677
+ });
2678
+ return receipt(result);
2679
+ },
2680
+ });
2681
+ const evidenceStatus = Type.Enum({
2682
+ required: "required", optional: "optional", "not-applicable": "not-applicable",
2683
+ });
2684
+ const recordEvidenceExpectationTool = defineTool({
2685
+ name: "record_evidence_expectation",
2686
+ label: "record_evidence_expectation",
2687
+ description: 'Record evidence requirements as {requirementId:"<canonical id>",evidence:{static:"required|optional|not-applicable",behavior:"required|optional|not-applicable",mock:"required|optional|not-applicable","real-integration":"required|optional|not-applicable"}}. Declare at least one lane; later calls replace only the lanes they name.',
2688
+ parameters: Type.Object({
2689
+ requirementId: Type.String(),
2690
+ evidence: Type.Object({
2691
+ static: Type.Optional(evidenceStatus),
2692
+ behavior: Type.Optional(evidenceStatus),
2693
+ mock: Type.Optional(evidenceStatus),
2694
+ "real-integration": Type.Optional(evidenceStatus),
2695
+ }, { additionalProperties: false }),
2696
+ }, { additionalProperties: false }),
2697
+ async execute(_toolCallId, params) {
2698
+ try {
2699
+ const fact = frontendEvidenceExpectationSchema.parse(params);
2700
+ if (!input.canonicalRequirements?.has(fact.requirementId)) {
2701
+ throw new Error(`unknown canonical requirement ${fact.requirementId}`);
2702
+ }
2703
+ return receipt(await adoptContractFact("evidence-expectation", {
2704
+ kind: "evidence-expectation", origin: "contract", ...fact,
2705
+ }));
2706
+ }
2707
+ catch (error) {
2708
+ return receipt({
2709
+ ok: false, kind: "evidence-expectation",
2710
+ error: error instanceof Error ? error.message : String(error),
2711
+ });
2712
+ }
2713
+ },
2714
+ });
2715
+ const recordRequiredDeliverablesTool = defineTool({
2716
+ name: "record_required_deliverables",
2717
+ label: "record_required_deliverables",
2718
+ description: 'Declare the complete source-required file deliverables as {items:[{path,requirementId,sourceFragmentId}]}. Interpret obligations from the original source, including lists/tables: permissions (allowedPaths/only allowed to modify), prohibitions, examples and read-only references are NOT delivery obligations. Each path must appear exactly in its frozen requirement-bound source fragment. Submit {items:[]} explicitly if no files are mandatory. A correction replaces the whole inventory. Required before finalize_contract ready.',
2719
+ parameters: Type.Object({ items: Type.Array(Type.Object({
2720
+ path: Type.String(), requirementId: Type.String(), sourceFragmentId: Type.String(),
2721
+ }, { additionalProperties: false })) }, { additionalProperties: false }),
2722
+ async execute(_toolCallId, params) {
2723
+ try {
2724
+ const declaration = validateFrontendRequiredDeliverables(params, input.canonicalRequirements ?? new Map());
2725
+ return receipt(await adoptContractFact("required-deliverables", {
2726
+ kind: "required-deliverables", origin: "contract", ...declaration,
2727
+ }));
2728
+ }
2729
+ catch (error) {
2730
+ return receipt({
2731
+ ok: false, kind: "required-deliverables",
2732
+ error: error instanceof Error ? error.message : String(error),
2733
+ });
2734
+ }
2735
+ },
2736
+ });
1978
2737
  // OpenSpec selection committed as individual typed facts (one path per
1979
2738
  // call) so a large candidate set never exceeds a single model output
1980
2739
  // budget: each tool call carries exactly one {path, disposition,
@@ -2045,6 +2804,13 @@ export async function createFrontendContractTools(input) {
2045
2804
  error: "blocked disposition requires blockingOwner (blocked-human | blocked-external)",
2046
2805
  });
2047
2806
  }
2807
+ if (disposition !== "blocked" && input.canonicalRequirements?.size &&
2808
+ !readCommittedEvents(store, attemptId).some((record) => record.fact.kind === "required-deliverables")) {
2809
+ return receipt({
2810
+ ok: false, kind: "finalize_contract",
2811
+ error: "call record_required_deliverables with the complete source-bound inventory (or items:[] when none) before finalizing",
2812
+ });
2813
+ }
2048
2814
  const blockedOwner = mapContractBlockedOwner({
2049
2815
  disposition: disposition ?? "",
2050
2816
  blockingOwner,
@@ -2068,6 +2834,10 @@ export async function createFrontendContractTools(input) {
2068
2834
  return {
2069
2835
  customTools: [
2070
2836
  ...recordTools,
2837
+ recordRequirementTool,
2838
+ recordEvidenceExpectationTool,
2839
+ recordUiStateTool,
2840
+ recordRequiredDeliverablesTool,
2071
2841
  recordOpenspecSelectionTool,
2072
2842
  finalizeContractTool,
2073
2843
  ],
@@ -2287,6 +3057,22 @@ export async function createFrontendScoutEvidenceTools(input) {
2287
3057
  });
2288
3058
  return {
2289
3059
  customTools: [recordTargetSurfaceTool, recordDesignEvidenceTool],
3060
+ committedFacts: () => readCommittedEvents(store, attemptId),
3061
+ adoptCommittedFacts: async (records) => {
3062
+ for (const record of records) {
3063
+ if (record.phase !== "committed")
3064
+ continue;
3065
+ const fact = record.fact;
3066
+ if (!fact || typeof fact !== "object" || Array.isArray(fact))
3067
+ continue;
3068
+ const result = await adoptScoutFact(typeof fact.kind === "string"
3069
+ ? fact.kind
3070
+ : "scout-fact", fact);
3071
+ if (!result.ok) {
3072
+ throw new Error(`frontend scout shard fact merge failed: ${result.error}`);
3073
+ }
3074
+ }
3075
+ },
2290
3076
  flush: async () => {
2291
3077
  const committed = readCommittedEvents(store, attemptId);
2292
3078
  await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "scout-typed-facts.jsonl"), committed);
@@ -2679,19 +3465,30 @@ const FRONTEND_PLAN_SEGMENTS = [
2679
3465
  instruction: [
2680
3466
  "PLAN PHASE — requirement coverage only.",
2681
3467
  "Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target facts (verification target bound to requirement ids and files). Group related requirements under one non-static behavior target when one observable test behavior proves them together; do not mechanically create one target per requirement. A non-static target id is the stable machine trace token; never submit prose as a symbol. Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
3468
+ "A coverage session is complete only when EVERY requirement assigned to this session (the full inventory, or the exact COVERAGE BATCH / shard list when present) has committed coverage facts: a record_plan_requirement entry plus verification targets, or a committed evidence gap. Keep committing in batches of up to 4 record_* calls per assistant message until then; do not write a concluding summary while any assigned requirement is still uncommitted — an early stop strands the remainder into a MISSING-FACT repair session and doubles the sessions needed.",
2682
3469
  "If a requirement genuinely cannot have a verification target, record a non-empty record_plan_evidence_gap. Do not call finalize_plan; it is not available in this phase.",
2683
3470
  ].join(" "),
2684
3471
  },
3472
+ {
3473
+ id: "ux-registry",
3474
+ toolNames: new Set(["record_state_registry", "adopt_staged_fact"]),
3475
+ instruction: [
3476
+ "PLAN PHASE — global UX vocabulary.",
3477
+ "Review ALL frozen requirements together and call record_state_registry exactly once with the complete UI-state and interaction vocabulary. UI states use declaredUiStates ids when present. Interaction names are stable kebab-case behavior domains; merge requirements that describe the same behavior instead of renaming it per AC slice. Empty arrays explicitly declare that no UX vocabulary applies. Do not record component choices or state-flow details in this phase.",
3478
+ "Do not call finalize_plan; it is not available in this phase.",
3479
+ ].join(" "),
3480
+ },
2685
3481
  {
2686
3482
  id: "ux-local",
2687
3483
  toolNames: new Set([
2688
3484
  "record_component_choice",
2689
3485
  "record_state_flow",
3486
+ "record_plan_verification_target",
2690
3487
  "adopt_staged_fact",
2691
3488
  ]),
2692
3489
  instruction: [
2693
- "PLAN PHASE — requirement-local UX decisions.",
2694
- "Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). For ONLY this requirement slice, record component choices and UI state flow. Keep these facts for this slice together. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
3490
+ "PLAN PHASE — global UX decisions.",
3491
+ "Requirements and verification targets are already committed in the ledger. Review the complete requirement set and the committed global UX registry together, then record each component choice, UI state and interaction. Bind each applicable state to its verificationTargetIds; the runtime derives the reverse VT.uiStates relation. If a VT requires correction, record_plan_verification_target with replace:true is available after declaring its states; preserve its requirement coverage. Multiple requirements describing one behavior share one registry name and state-flow entry. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
2695
3492
  "Do not call finalize_plan; it is not available in this phase.",
2696
3493
  ].join(" "),
2697
3494
  },
@@ -2749,40 +3546,6 @@ function planFactStringList(value) {
2749
3546
  function planFactScopeIntersects(fact, requirementIds) {
2750
3547
  return planFactStringList(fact.scopeRequirementIds).some((id) => requirementIds.has(id));
2751
3548
  }
2752
- /** Apply state-flow replacement/removal semantics exactly as the plan compiler does. */
2753
- function collectCanonicalStateFlowNames(committedFacts) {
2754
- const uiStateNames = new Set();
2755
- const interactionNames = new Set();
2756
- for (const value of committedFacts) {
2757
- const fact = committedFactFromPlanRecord(value);
2758
- if (!fact || fact.origin !== "plan" || fact.kind !== "state-flow")
2759
- continue;
2760
- for (const name of planFactStringList(fact.removeUiStateNames)) {
2761
- uiStateNames.delete(name);
2762
- }
2763
- for (const name of planFactStringList(fact.removeInteractionNames)) {
2764
- interactionNames.delete(name);
2765
- }
2766
- for (const state of Array.isArray(fact.uiStates) ? fact.uiStates : []) {
2767
- if (!state || typeof state !== "object" || Array.isArray(state))
2768
- continue;
2769
- const name = state.name;
2770
- if (typeof name === "string" && name.trim())
2771
- uiStateNames.add(name);
2772
- }
2773
- for (const interaction of Array.isArray(fact.interactions)
2774
- ? fact.interactions
2775
- : []) {
2776
- if (!interaction || typeof interaction !== "object" || Array.isArray(interaction)) {
2777
- continue;
2778
- }
2779
- const name = interaction.name;
2780
- if (typeof name === "string" && name.trim())
2781
- interactionNames.add(name);
2782
- }
2783
- }
2784
- return { uiStateNames, interactionNames };
2785
- }
2786
3549
  /** Compute the authoritative coverage queue from the committed plan ledger. */
2787
3550
  export function collectFrontendPlanMissingFacts(input) {
2788
3551
  const requirements = new Map();
@@ -2864,6 +3627,17 @@ export function collectFrontendPlanPhaseMissingFacts(input) {
2864
3627
  const facts = input.committedFacts
2865
3628
  .map(committedFactFromPlanRecord)
2866
3629
  .filter((fact) => Boolean(fact && fact.origin === "plan"));
3630
+ if (input.phase === "ux-registry") {
3631
+ return facts.some((fact) => fact.kind === "state-registry")
3632
+ ? []
3633
+ : [
3634
+ {
3635
+ kind: "state-registry",
3636
+ requirementIds: [...input.requirementIds],
3637
+ reason: "global UX vocabulary phase has no committed state-registry fact",
3638
+ },
3639
+ ];
3640
+ }
2867
3641
  if (input.phase === "ux-local") {
2868
3642
  const needsUx = input.requirementIds.some((id) => input.behaviorRequiredRequirementIds?.includes(id));
2869
3643
  if (!needsUx)
@@ -2958,13 +3732,28 @@ export function batchFrontendPlanRequirements(input) {
2958
3732
  return batches;
2959
3733
  }
2960
3734
  const FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS = 10;
2961
- const FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS = 6;
2962
- // A large requirement set creates both coverage and UX-local shards. Keep a
2963
- // safety bound, but do not let the old 32-session ceiling skip finalize for a
2964
- // legitimate large plan.
3735
+ const FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY = 4;
3736
+ // A large requirement set creates parallel coverage shards; UX remains one
3737
+ // global decision session so behavior names and component/state facts are not
3738
+ // reinvented per AC batch. Keep a safety bound for adaptive coverage retries
3739
+ // without letting the old 32-session ceiling skip finalize.
2965
3740
  const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
2966
3741
  const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
2967
3742
  const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
3743
+ async function mapWithConcurrency(items, limit, worker) {
3744
+ const results = new Array(items.length);
3745
+ let nextIndex = 0;
3746
+ const workerCount = Math.min(Math.max(1, limit), items.length);
3747
+ await Promise.all(Array.from({ length: workerCount }, async () => {
3748
+ while (true) {
3749
+ const index = nextIndex++;
3750
+ if (index >= items.length)
3751
+ return;
3752
+ results[index] = await worker(items[index], index);
3753
+ }
3754
+ }));
3755
+ return results;
3756
+ }
2968
3757
  function compactPromptString(value, maxChars) {
2969
3758
  if (typeof value !== "string" || value.trim().length === 0)
2970
3759
  return undefined;
@@ -3147,8 +3936,33 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3147
3936
  }];
3148
3937
  })
3149
3938
  : [];
3939
+ const compactDeclaredUiStates = Array.isArray(payload.declaredUiStates)
3940
+ ? payload.declaredUiStates.flatMap((value) => {
3941
+ if (!value || typeof value !== "object" || Array.isArray(value))
3942
+ return [];
3943
+ const state = value;
3944
+ const id = compactPromptString(state.id, 80);
3945
+ if (!id)
3946
+ return [];
3947
+ return [{
3948
+ id,
3949
+ ...(compactPromptString(state.trigger, 180)
3950
+ ? { trigger: compactPromptString(state.trigger, 180) }
3951
+ : {}),
3952
+ ...(compactPromptString(state.observableOutcome, 240)
3953
+ ? { observableOutcome: compactPromptString(state.observableOutcome, 240) }
3954
+ : {}),
3955
+ }];
3956
+ })
3957
+ : [];
3958
+ const committedUx = payload.committedUx &&
3959
+ typeof payload.committedUx === "object" &&
3960
+ !Array.isArray(payload.committedUx)
3961
+ ? payload.committedUx
3962
+ : undefined;
3150
3963
  const compactPayload = {
3151
3964
  requirements: compactRequirements,
3965
+ requiredDeliverables: Array.isArray(payload.requiredDeliverables) ? payload.requiredDeliverables : [],
3152
3966
  ...(compactTargetSurface.length > 0
3153
3967
  ? { targetSurface: compactTargetSurface }
3154
3968
  : {}),
@@ -3158,6 +3972,17 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
3158
3972
  ...(compactDesignEvidence.length > 0
3159
3973
  ? { designEvidence: compactDesignEvidence }
3160
3974
  : {}),
3975
+ ...(compactDeclaredUiStates.length > 0
3976
+ ? { declaredUiStates: compactDeclaredUiStates }
3977
+ : {}),
3978
+ ...(committedUx
3979
+ ? {
3980
+ committedUx: {
3981
+ uiStateNames: compactPromptStringArray(committedUx.uiStateNames, 40, 100),
3982
+ interactionNames: compactPromptStringArray(committedUx.interactionNames, 60, 120),
3983
+ },
3984
+ }
3985
+ : {}),
3161
3986
  };
3162
3987
  const compactBlock = [
3163
3988
  "<frontend_plan_input>",
@@ -3254,6 +4079,8 @@ function compactFrontendPlanLedgerContext(input) {
3254
4079
  purpose: compactPromptString(item.purpose, 120),
3255
4080
  component: compactPromptString(item.component, 120),
3256
4081
  decision: compactPromptString(item.decision, 40),
4082
+ covers: compactPromptStringArray(item.covers, 40, 120),
4083
+ evidencePath: compactPromptString(item.evidencePath, 180),
3257
4084
  }];
3258
4085
  }).slice(0, 24)
3259
4086
  : [];
@@ -3261,6 +4088,14 @@ function compactFrontendPlanLedgerContext(input) {
3261
4088
  compactFacts.push({ kind: fact.kind, uiComponentChoices: choices });
3262
4089
  continue;
3263
4090
  }
4091
+ if (fact.kind === "state-registry") {
4092
+ compactFacts.push({
4093
+ kind: fact.kind,
4094
+ uiStateNames: compactPromptStringArray(fact.uiStateNames, 40, 100),
4095
+ interactionNames: compactPromptStringArray(fact.interactionNames, 60, 120),
4096
+ });
4097
+ continue;
4098
+ }
3264
4099
  if (fact.kind === "state-flow") {
3265
4100
  compactFacts.push({
3266
4101
  kind: fact.kind,
@@ -3323,6 +4158,7 @@ function compactFrontendPlanLedgerContext(input) {
3323
4158
  export async function runFrontendPlanSegmentedSessions(input) {
3324
4159
  const queue = [];
3325
4160
  const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
4161
+ const uxRegistrySegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-registry");
3326
4162
  const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
3327
4163
  const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
3328
4164
  const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
@@ -3342,7 +4178,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3342
4178
  ? ["plan-requirement", "plan-verification-target", "state-flow", "data-flow", "mock-api"]
3343
4179
  : segment.id === "global-dependency-deviation"
3344
4180
  ? ["dependency", "design-deviation"]
3345
- : ["plan-requirement", "plan-verification-target", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
4181
+ : ["plan-requirement", "plan-verification-target", "state-registry", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
3346
4182
  const ledger = input.committedFacts
3347
4183
  ? compactFrontendPlanLedgerContext({
3348
4184
  committedFacts: input.committedFacts(),
@@ -3365,7 +4201,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3365
4201
  const buildCoveragePrompt = (slice, missing = []) => [
3366
4202
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
3367
4203
  coverageSegment.instruction,
3368
- `COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
4204
+ `COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them. Finish the whole list before concluding: commit every listed requirement's coverage facts (verification targets or an evidence gap), batching up to 4 record_* calls per message; an early stop re-queues the remainder as a MISSING-FACT repair session.`,
3369
4205
  ...(missing.length > 0
3370
4206
  ? [
3371
4207
  "MISSING-FACT QUEUE: the previous session did not establish complete coverage. Repair ONLY these items, then re-check the slice:",
@@ -3378,14 +4214,20 @@ export async function runFrontendPlanSegmentedSessions(input) {
3378
4214
  ? compactFrontendPlanLedgerContext({
3379
4215
  committedFacts: input.committedFacts(),
3380
4216
  requirementIds: slice,
3381
- kinds: ["plan-requirement", "plan-verification-target", "component-choice", "state-flow"],
4217
+ kinds: [
4218
+ "plan-requirement",
4219
+ "plan-verification-target",
4220
+ "state-registry",
4221
+ "component-choice",
4222
+ "state-flow",
4223
+ ],
3382
4224
  scopedKinds: ["component-choice", "state-flow"],
3383
4225
  })
3384
4226
  : "";
3385
4227
  return [
3386
4228
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
3387
4229
  uxSegment.instruction,
3388
- `UX LOCAL BATCH: process ONLY these requirements: ${slice.join(", ")}.`,
4230
+ `GLOBAL UX SCOPE: process the complete requirement set together: ${slice.join(", ")}. Do not split or rename one behavior by requirement id.`,
3389
4231
  ledger,
3390
4232
  ...(missing.length > 0
3391
4233
  ? [
@@ -3395,6 +4237,37 @@ export async function runFrontendPlanSegmentedSessions(input) {
3395
4237
  : []),
3396
4238
  ].filter(Boolean).join("\n\n");
3397
4239
  };
4240
+ const buildUxRegistryPrompt = (missing = []) => {
4241
+ const ledger = input.committedFacts
4242
+ ? compactFrontendPlanLedgerContext({
4243
+ committedFacts: input.committedFacts(),
4244
+ requirementIds: allRequirementIds,
4245
+ kinds: [
4246
+ "plan-requirement",
4247
+ "plan-verification-target",
4248
+ "state-registry",
4249
+ ],
4250
+ })
4251
+ : "";
4252
+ const compact = allRequirementIds.length > 0
4253
+ ? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds, {
4254
+ includeVerificationTargets: false,
4255
+ includeDesignEvidence: false,
4256
+ includeChecklist: false,
4257
+ })
4258
+ : input.basePrompt;
4259
+ return [
4260
+ compact,
4261
+ uxRegistrySegment.instruction,
4262
+ ledger,
4263
+ ...(missing.length > 0
4264
+ ? [
4265
+ "MISSING-FACT QUEUE: commit the global UX registry, then re-check this phase:",
4266
+ ...missing.map((item) => `- ${item.reason}`),
4267
+ ]
4268
+ : []),
4269
+ ].filter(Boolean).join("\n\n");
4270
+ };
3398
4271
  const buildCompactLocalPrompt = (missing = []) => {
3399
4272
  const ledger = input.committedFacts
3400
4273
  ? compactFrontendPlanLedgerContext({
@@ -3403,6 +4276,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3403
4276
  kinds: [
3404
4277
  "plan-requirement",
3405
4278
  "plan-verification-target",
4279
+ "state-registry",
3406
4280
  "component-choice",
3407
4281
  "state-flow",
3408
4282
  ],
@@ -3411,7 +4285,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3411
4285
  return [
3412
4286
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds),
3413
4287
  "PLAN PHASE — compact local planning for a small frontend request.",
3414
- "For every listed requirement, record plan-requirement and verification-target facts, then record the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
4288
+ "Review all listed requirements together and record_state_registry first with one global UX vocabulary. Then record every plan-requirement and verification-target fact, followed by the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
3415
4289
  "TOOL-FIRST: your first assistant actions must be record_* tool calls, at most 2-3 facts per message. Do not draft the whole analysis before recording; if a fact is uncertain, record it with an evidence gap instead of reasoning longer.",
3416
4290
  ledger,
3417
4291
  ...(missing.length > 0
@@ -3427,8 +4301,8 @@ export async function runFrontendPlanSegmentedSessions(input) {
3427
4301
  const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
3428
4302
  ? {
3429
4303
  ...r,
3430
- failureCategory: PLANNER_THINKING_EXHAUSTED_CATEGORY,
3431
- stderr: `${r.stderr}\n${PLANNER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 typed facts committed; the batch ladder degraded the scope without converging — set thinking=off for this tier or switch to a non-thinking model`.trim(),
4304
+ failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
4305
+ stderr: `${r.stderr}\n${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before the next typed fact; preserve committed facts and retry only the unfinished phase`.trim(),
3432
4306
  }
3433
4307
  : r;
3434
4308
  // An empty list means "ledger unreadable / unknown" and falls back to one
@@ -3443,6 +4317,29 @@ export async function runFrontendPlanSegmentedSessions(input) {
3443
4317
  : [];
3444
4318
  const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
3445
4319
  const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
4320
+ if (input.parallelCoverageOnly === true &&
4321
+ requirementIdsProvided &&
4322
+ coverageWorkIds.length === 0) {
4323
+ // A retry may reopen a shard whose committed facts are already complete.
4324
+ // Treat that shard as an idempotent no-op; returning the normal empty
4325
+ // session failure would make a partially failed map impossible to resume.
4326
+ return {
4327
+ ok: true,
4328
+ assistantText: "",
4329
+ command: [],
4330
+ durationMs: 0,
4331
+ exitCode: 0,
4332
+ failureCategory: "success",
4333
+ modelDisplay: "reused-coverage-facts",
4334
+ parsedEvents: 0,
4335
+ stderr: "",
4336
+ stdout: "",
4337
+ timedOut: false,
4338
+ attemptedModels: [],
4339
+ fallbackUsed: false,
4340
+ tokensUsed: 0,
4341
+ };
4342
+ }
3446
4343
  const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
3447
4344
  const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
3448
4345
  // Small, single-surface requests do not benefit from six isolated Pi
@@ -3462,6 +4359,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3462
4359
  "record_plan_requirement",
3463
4360
  "record_plan_verification_target",
3464
4361
  "record_plan_evidence_gap",
4362
+ "record_state_registry",
3465
4363
  "record_component_choice",
3466
4364
  "record_state_flow",
3467
4365
  "adopt_staged_fact",
@@ -3503,29 +4401,30 @@ export async function runFrontendPlanSegmentedSessions(input) {
3503
4401
  if (useCompactSmallPlan) {
3504
4402
  // Compact mode already queued both sessions above.
3505
4403
  }
3506
- else {
3507
- const uxCosts = new Map((input.requirementIds ?? []).map((id) => [
3508
- id,
3509
- Math.max(2, Math.min(3, input.requirementCosts?.get(id) ?? 2)),
3510
- ]));
3511
- const uxSlices = requirementIdsProvided
3512
- ? batchFrontendPlanRequirements({
3513
- requirementIds: input.requirementIds,
3514
- maxEstimatedRecordCalls: FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS,
3515
- maxRequirements: 3,
3516
- requirementCosts: uxCosts,
3517
- })
3518
- : [[]];
3519
- uxSlices.forEach((slice, batchIndex) => queue.push({
3520
- id: `ux-local-${batchIndex + 1}`,
3521
- toolNames: uxSegment.toolNames,
3522
- ...(slice.length > 0 ? { requirementSlice: slice } : {}),
3523
- prompt: slice.length > 0
3524
- ? buildUxPrompt(slice)
4404
+ else if (!input.parallelCoverageOnly) {
4405
+ if (requirementIdsProvided) {
4406
+ queue.push({
4407
+ id: uxRegistrySegment.id,
4408
+ toolNames: uxRegistrySegment.toolNames,
4409
+ prompt: buildUxRegistryPrompt(),
4410
+ });
4411
+ }
4412
+ const uxSlice = requirementIdsProvided ? [...allRequirementIds] : [];
4413
+ queue.push({
4414
+ id: "ux-local-1",
4415
+ // Compatibility path for an unreadable requirement inventory: there is
4416
+ // no safe all-requirements registry phase, so the unscoped UX session
4417
+ // must establish its registry before recording state flow.
4418
+ toolNames: requirementIdsProvided
4419
+ ? uxSegment.toolNames
4420
+ : new Set([...(uxSegment.toolNames ?? []), "record_state_registry"]),
4421
+ ...(uxSlice.length > 0 ? { requirementSlice: uxSlice } : {}),
4422
+ prompt: uxSlice.length > 0
4423
+ ? buildUxPrompt(uxSlice)
3525
4424
  : buildPhasePrompt(uxSegment),
3526
- }));
4425
+ });
3527
4426
  for (const segment of FRONTEND_PLAN_SEGMENTS) {
3528
- if (["coverage", "ux-local", "finalize"].includes(segment.id))
4427
+ if (["coverage", "ux-registry", "ux-local", "finalize"].includes(segment.id))
3529
4428
  continue;
3530
4429
  queue.push({
3531
4430
  id: segment.id,
@@ -3539,7 +4438,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3539
4438
  prompt: buildPhasePrompt(finalizeSegment),
3540
4439
  });
3541
4440
  }
3542
- let last;
4441
+ let accumulated;
3543
4442
  let index = 0;
3544
4443
  let invocationCount = 0;
3545
4444
  while (index < queue.length) {
@@ -3567,6 +4466,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3567
4466
  else if (session.id === "compact-local") {
3568
4467
  prompt = buildCompactLocalPrompt(session.missingFacts);
3569
4468
  }
4469
+ else if (session.id === "ux-registry") {
4470
+ prompt = buildUxRegistryPrompt(session.missingFacts);
4471
+ }
3570
4472
  else if (session.id.startsWith("ux-local-")) {
3571
4473
  // Build this at execution time: coverage facts are committed by the
3572
4474
  // preceding sessions and must be visible to the UX-local model.
@@ -3587,8 +4489,10 @@ export async function runFrontendPlanSegmentedSessions(input) {
3587
4489
  }
3588
4490
  }
3589
4491
  const committedBefore = input.committedFactCount();
3590
- input.setActiveRequirementScope?.(session.id === "compact-local" || session.id.startsWith("ux-local-")
3591
- ? session.requirementSlice ?? []
4492
+ input.setActiveRequirementScope?.(session.coverageOnly ||
4493
+ session.id === "compact-local" ||
4494
+ session.id.startsWith("ux-local-")
4495
+ ? session.coverageSlice ?? session.requirementSlice ?? []
3592
4496
  : []);
3593
4497
  if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
3594
4498
  break;
@@ -3606,7 +4510,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3606
4510
  }
3607
4511
  : {}),
3608
4512
  });
3609
- last = result;
4513
+ accumulated = accumulated
4514
+ ? combineSequentialPiResults(accumulated, result)
4515
+ : result;
3610
4516
  try {
3611
4517
  await input.flushLedger();
3612
4518
  }
@@ -3626,6 +4532,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3626
4532
  })
3627
4533
  : [];
3628
4534
  const isCompactLocalSession = session.id === "compact-local";
4535
+ const isUxRegistrySession = session.id === "ux-registry";
3629
4536
  const isUxLocalSession = session.id.startsWith("ux-local-");
3630
4537
  const missingPhase = isCompactLocalSession && input.committedFacts
3631
4538
  ? [
@@ -3633,6 +4540,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
3633
4540
  requirementIds: allRequirementIds,
3634
4541
  committedFacts: input.committedFacts(),
3635
4542
  }),
4543
+ ...collectFrontendPlanPhaseMissingFacts({
4544
+ phase: "ux-registry",
4545
+ requirementIds: allRequirementIds,
4546
+ committedFacts: input.committedFacts(),
4547
+ }),
3636
4548
  ...collectFrontendPlanPhaseMissingFacts({
3637
4549
  phase: "ux-local",
3638
4550
  requirementIds: allRequirementIds,
@@ -3640,20 +4552,26 @@ export async function runFrontendPlanSegmentedSessions(input) {
3640
4552
  behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3641
4553
  }),
3642
4554
  ]
3643
- : isUxLocalSession && session.requirementSlice && input.committedFacts
4555
+ : isUxRegistrySession && input.committedFacts
3644
4556
  ? collectFrontendPlanPhaseMissingFacts({
3645
- phase: "ux-local",
3646
- requirementIds: session.requirementSlice,
4557
+ phase: "ux-registry",
4558
+ requirementIds: allRequirementIds,
3647
4559
  committedFacts: input.committedFacts(),
3648
- behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3649
4560
  })
3650
- : session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
4561
+ : isUxLocalSession && session.requirementSlice && input.committedFacts
3651
4562
  ? collectFrontendPlanPhaseMissingFacts({
3652
- phase: "global-mock-data",
3653
- requirementIds: allRequirementIds,
4563
+ phase: "ux-local",
4564
+ requirementIds: session.requirementSlice,
3654
4565
  committedFacts: input.committedFacts(),
4566
+ behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3655
4567
  })
3656
- : [];
4568
+ : session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
4569
+ ? collectFrontendPlanPhaseMissingFacts({
4570
+ phase: "global-mock-data",
4571
+ requirementIds: allRequirementIds,
4572
+ committedFacts: input.committedFacts(),
4573
+ })
4574
+ : [];
3657
4575
  const missingPhaseFacts = [...missingCoverage, ...missingPhase];
3658
4576
  // Frozen verification commands must operate on files the planner has
3659
4577
  // committed verification targets for; otherwise the admission writeSet
@@ -3681,11 +4599,13 @@ export async function runFrontendPlanSegmentedSessions(input) {
3681
4599
  ? buildCoveragePrompt(session.coverageSlice ?? [], missing)
3682
4600
  : isCompactLocalSession
3683
4601
  ? buildCompactLocalPrompt(missing)
3684
- : session.id === "finalize" && useCompactSmallPlan
3685
- ? buildCompactFinalizePrompt(missing)
3686
- : isUxLocalSession
3687
- ? buildUxPrompt(session.requirementSlice ?? [], missing)
3688
- : buildPhasePrompt(globalMockDataSegment, missing);
4602
+ : isUxRegistrySession
4603
+ ? buildUxRegistryPrompt(missing)
4604
+ : session.id === "finalize" && useCompactSmallPlan
4605
+ ? buildCompactFinalizePrompt(missing)
4606
+ : isUxLocalSession
4607
+ ? buildUxPrompt(session.requirementSlice ?? [], missing)
4608
+ : buildPhasePrompt(globalMockDataSegment, missing);
3689
4609
  if (result.ok) {
3690
4610
  if (missingPhaseFacts.length === 0) {
3691
4611
  index += 1;
@@ -3701,9 +4621,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3701
4621
  continue;
3702
4622
  }
3703
4623
  return {
3704
- ...result,
4624
+ ...accumulated,
3705
4625
  ok: false,
3706
- stderr: `${result.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
4626
+ stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
3707
4627
  failureCategory: "invalid-output",
3708
4628
  };
3709
4629
  }
@@ -3720,9 +4640,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3720
4640
  }
3721
4641
  if (missingPhaseFacts.length > 0) {
3722
4642
  return {
3723
- ...result,
4643
+ ...accumulated,
3724
4644
  ok: false,
3725
- stderr: `frontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`,
4645
+ stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
3726
4646
  failureCategory: "invalid-output",
3727
4647
  };
3728
4648
  }
@@ -3800,7 +4720,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3800
4720
  });
3801
4721
  continue;
3802
4722
  }
3803
- if ((session.coverageOnly || isUxLocalSession || session.id === "global-mock-data") &&
4723
+ if ((session.coverageOnly || isUxRegistrySession || isUxLocalSession || session.id === "global-mock-data") &&
3804
4724
  missingPhaseFacts.length > 0 &&
3805
4725
  !result.stderr.trim() &&
3806
4726
  !result.timedOut &&
@@ -3813,11 +4733,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
3813
4733
  };
3814
4734
  continue;
3815
4735
  }
3816
- return mapPlannerExhaustion(result, committedAfter > committedBefore);
4736
+ return mapPlannerExhaustion(accumulated, committedAfter > committedBefore);
3817
4737
  }
3818
4738
  if (index < queue.length) {
3819
4739
  return {
3820
- ...(last ?? {
4740
+ ...(accumulated ?? {
3821
4741
  ok: false,
3822
4742
  stdout: "",
3823
4743
  stderr: "",
@@ -3834,11 +4754,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
3834
4754
  tokensUsed: 0,
3835
4755
  }),
3836
4756
  ok: false,
3837
- stderr: `${last?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
4757
+ stderr: `${accumulated?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
3838
4758
  failureCategory: "invalid-output",
3839
4759
  };
3840
4760
  }
3841
- return mapPlannerExhaustion(last ?? {
4761
+ return mapPlannerExhaustion(accumulated ?? {
3842
4762
  ok: false,
3843
4763
  stdout: "",
3844
4764
  stderr: "frontend plan segmentation produced no session",
@@ -3846,6 +4766,184 @@ export async function runFrontendPlanSegmentedSessions(input) {
3846
4766
  durationMs: 0,
3847
4767
  }, false);
3848
4768
  }
4769
+ /**
4770
+ * Run the two independent Scout evidence surfaces concurrently while keeping
4771
+ * their typed-event stores isolated. The main Scout store is the only store
4772
+ * visible to Plan; shard facts are merged in completion-order-independent
4773
+ * order after both sessions settle. This gives discovery real parallelism
4774
+ * without allowing sibling models to race a shared revision counter.
4775
+ */
4776
+ function aggregateParallelPiResults(results) {
4777
+ const first = results[0];
4778
+ const failed = results.find((result) => !result.ok);
4779
+ const representative = failed ?? first;
4780
+ return {
4781
+ ...representative,
4782
+ ok: failed === undefined,
4783
+ failureCategory: failed?.failureCategory ?? "success",
4784
+ durationMs: Math.max(...results.map((result) => result.durationMs), 0),
4785
+ exitCode: failed ? failed.exitCode : 0,
4786
+ stderr: results.map((result) => result.stderr).filter(Boolean).join("\n"),
4787
+ tokensUsed: results.reduce((total, result) => total + result.tokensUsed, 0),
4788
+ parsedEvents: results.reduce((total, result) => total + result.parsedEvents, 0),
4789
+ attemptedModels: [
4790
+ ...new Set(results.flatMap((result) => result.attemptedModels)),
4791
+ ],
4792
+ fallbackUsed: results.some((result) => result.fallbackUsed),
4793
+ timedOut: results.some((result) => result.timedOut),
4794
+ };
4795
+ }
4796
+ function combineSequentialPiResults(first, second) {
4797
+ return {
4798
+ ...second,
4799
+ durationMs: first.durationMs + second.durationMs,
4800
+ stderr: [first.stderr, second.stderr].filter(Boolean).join("\n"),
4801
+ tokensUsed: first.tokensUsed + second.tokensUsed,
4802
+ parsedEvents: first.parsedEvents + second.parsedEvents,
4803
+ attemptedModels: [
4804
+ ...new Set([...first.attemptedModels, ...second.attemptedModels]),
4805
+ ],
4806
+ fallbackUsed: first.fallbackUsed || second.fallbackUsed,
4807
+ timedOut: first.timedOut || second.timedOut,
4808
+ };
4809
+ }
4810
+ async function runFrontendScoutParallelSessions(input) {
4811
+ const [{ createTypedEventStore }] = await Promise.all([
4812
+ import("../workflows/dag/frontend-typed-event-store.js"),
4813
+ ]);
4814
+ const shards = [
4815
+ {
4816
+ id: "surface",
4817
+ toolName: "record_target_surface",
4818
+ instruction: [
4819
+ "PARALLEL SCOUT SHARD — target surface only.",
4820
+ "Inspect routes, entrypoints, implementation ownership, data source, and applicable test paths.",
4821
+ "Call record_target_surface exactly once with the complete runtime-evidenced surface. Do not call record_design_evidence.",
4822
+ ].join(" "),
4823
+ },
4824
+ {
4825
+ id: "design",
4826
+ toolName: "record_design_evidence",
4827
+ instruction: [
4828
+ "PARALLEL SCOUT SHARD — design evidence only.",
4829
+ "Inspect the frontend framework, styling/theme conventions, reusable components, and relevant design/spec files.",
4830
+ "Call record_design_evidence for the evidence you actually read. Do not call record_target_surface.",
4831
+ ].join(" "),
4832
+ },
4833
+ ];
4834
+ const outcomes = await Promise.all(shards.map(async (shard) => {
4835
+ let shardTools;
4836
+ let shardResult;
4837
+ try {
4838
+ const store = createTypedEventStore();
4839
+ const shardNodeId = `${input.nodeId}/parallel/${shard.id}`;
4840
+ shardTools = await createFrontendScoutEvidenceTools({
4841
+ attemptId: `${input.attemptId}:parallel:${shard.id}`,
4842
+ store,
4843
+ runDir: input.runDir,
4844
+ nodeId: shardNodeId,
4845
+ workspaceRoot: input.workspaceRoot,
4846
+ sourceDeclaredPaths: input.sourceDeclaredPaths,
4847
+ });
4848
+ const customTools = shardTools.customTools.filter((tool) => typeof tool === "object" &&
4849
+ tool !== null &&
4850
+ tool.name === shard.toolName);
4851
+ const sessionOptions = {
4852
+ ...input.sessionOptions,
4853
+ sessionEventsPath: path.join(input.runDir, shardNodeId, "session-events.jsonl"),
4854
+ };
4855
+ shardResult = await input.piStepFn({
4856
+ ...sessionOptions,
4857
+ prompt: `${input.basePrompt}\n\n${shard.instruction}`,
4858
+ writerToolPolicy: {
4859
+ requireSdk: true,
4860
+ customTools: [...customTools, ...(input.readBudgetTools ?? [])],
4861
+ },
4862
+ });
4863
+ await shardTools.flush();
4864
+ return { shard, result: shardResult, tools: shardTools };
4865
+ }
4866
+ catch (error) {
4867
+ const crashMessage = `frontend scout parallel shard ${shard.id} crashed: ${error instanceof Error ? error.message : String(error)}`;
4868
+ return {
4869
+ shard,
4870
+ tools: shardTools,
4871
+ result: shardResult
4872
+ ? {
4873
+ ...shardResult,
4874
+ ok: false,
4875
+ failureCategory: shardResult.ok
4876
+ ? "invalid-output"
4877
+ : shardResult.failureCategory,
4878
+ stderr: [shardResult.stderr, crashMessage]
4879
+ .filter(Boolean)
4880
+ .join("\n"),
4881
+ }
4882
+ : {
4883
+ ok: false,
4884
+ assistantText: "",
4885
+ command: [],
4886
+ durationMs: 0,
4887
+ exitCode: null,
4888
+ failureCategory: "tool-policy",
4889
+ modelDisplay: "unknown",
4890
+ parsedEvents: 0,
4891
+ stderr: crashMessage,
4892
+ stdout: "",
4893
+ timedOut: false,
4894
+ attemptedModels: [],
4895
+ fallbackUsed: false,
4896
+ tokensUsed: 0,
4897
+ },
4898
+ };
4899
+ }
4900
+ }));
4901
+ const ordered = [...outcomes].sort((left, right) => left.shard.id.localeCompare(right.shard.id));
4902
+ try {
4903
+ for (const outcome of ordered) {
4904
+ if (outcome.result.ok && outcome.tools) {
4905
+ await input.mainTools.adoptCommittedFacts(outcome.tools.committedFacts());
4906
+ }
4907
+ }
4908
+ }
4909
+ catch (error) {
4910
+ const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4911
+ return {
4912
+ ...aggregate,
4913
+ ok: false,
4914
+ failureCategory: "invalid-output",
4915
+ stderr: [
4916
+ aggregate.stderr,
4917
+ `frontend scout parallel fact merge failed: ${error instanceof Error ? error.message : String(error)}`,
4918
+ ]
4919
+ .filter(Boolean)
4920
+ .join("\n"),
4921
+ };
4922
+ }
4923
+ const failed = ordered.find((outcome) => !outcome.result.ok);
4924
+ if (failed) {
4925
+ const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4926
+ return {
4927
+ ...aggregate,
4928
+ ok: false,
4929
+ stderr: `${aggregate.stderr}\nfrontend scout parallel shard failed: ${failed.shard.id}`.trim(),
4930
+ };
4931
+ }
4932
+ const committedKinds = new Set(ordered.flatMap((outcome) => (outcome.tools?.committedFacts() ?? [])
4933
+ .map((record) => record.fact?.kind)
4934
+ .filter((kind) => typeof kind === "string")));
4935
+ const missing = ["target-surface", "design-evidence"].filter((kind) => !committedKinds.has(kind));
4936
+ if (missing.length > 0) {
4937
+ const fallback = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4938
+ return {
4939
+ ...fallback,
4940
+ ok: false,
4941
+ failureCategory: "invalid-output",
4942
+ stderr: `${fallback.stderr}\nfrontend scout parallel shards committed no ${missing.join(" or ")} fact`.trim(),
4943
+ };
4944
+ }
4945
+ return aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4946
+ }
3849
4947
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
3850
4948
  const started = Date.now();
3851
4949
  const persona = resolveDagPiPersona(input.task);
@@ -3940,6 +5038,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
3940
5038
  let planLedgerTools;
3941
5039
  let contractTools;
3942
5040
  let scoutEvidenceTools;
5041
+ let scoutSourceDeclaredPaths;
3943
5042
  let readBudgetTools;
3944
5043
  const commandPolicy = resolveDagCommandPolicy(input.task.commandPolicy);
3945
5044
  const allowsPlaywrightCli = dagCommandPolicyAllows(input.task.commandPolicy, "playwright-cli");
@@ -4083,6 +5182,10 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4083
5182
  store,
4084
5183
  runDir: meta.runDir,
4085
5184
  nodeId: input.task.id,
5185
+ canonicalRequirements: await resolveFrontendCanonicalRequirements({
5186
+ cwd: input.cwd,
5187
+ sourceBinding: meta.spec.sourceBinding,
5188
+ }),
4086
5189
  });
4087
5190
  writerToolPolicy = {
4088
5191
  requireSdk: true,
@@ -4103,16 +5206,17 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4103
5206
  try {
4104
5207
  const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
4105
5208
  const store = createTypedEventStore();
5209
+ scoutSourceDeclaredPaths = await resolveFrontendScoutSourceDeclaredPaths({
5210
+ cwd: input.cwd,
5211
+ spec: meta.spec,
5212
+ });
4106
5213
  scoutEvidenceTools = await createFrontendScoutEvidenceTools({
4107
5214
  attemptId: `${meta.runId}:${input.task.id}`,
4108
5215
  store,
4109
5216
  runDir: meta.runDir,
4110
5217
  nodeId: input.task.id,
4111
5218
  workspaceRoot: input.cwd,
4112
- sourceDeclaredPaths: await resolveFrontendScoutSourceDeclaredPaths({
4113
- cwd: input.cwd,
4114
- spec: meta.spec,
4115
- }),
5219
+ sourceDeclaredPaths: scoutSourceDeclaredPaths,
4116
5220
  });
4117
5221
  writerToolPolicy = {
4118
5222
  requireSdk: true,
@@ -4145,6 +5249,13 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4145
5249
  cwd: input.cwd,
4146
5250
  sourceBinding: meta.spec.sourceBinding,
4147
5251
  }),
5252
+ declaredUiStateIds: await resolveFrontendDeclaredUiStateIds({
5253
+ runDir: meta.runDir,
5254
+ }),
5255
+ canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
5256
+ runDir: meta.runDir,
5257
+ }),
5258
+ workspaceRoot: input.cwd,
4148
5259
  });
4149
5260
  writerToolPolicy = {
4150
5261
  requireSdk: true,
@@ -4284,11 +5395,27 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4284
5395
  }
4285
5396
  : undefined,
4286
5397
  };
4287
- if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
4288
- // Frontend-only split: sequential sessions with independent
4289
- // output budgets (coverage -> UX decisions -> finalize), mirroring the
4290
- // backend-test template's module sharding. Every other template keeps
4291
- // the single-session path below.
5398
+ if (isFrontendScoutEvidenceNode(input.task) &&
5399
+ scoutEvidenceTools &&
5400
+ input.task.complexity !== "LOW") {
5401
+ result = await runFrontendScoutParallelSessions({
5402
+ piStepFn,
5403
+ sessionOptions: piSessionOptions,
5404
+ basePrompt: input.prompt,
5405
+ mainTools: scoutEvidenceTools,
5406
+ runDir: meta.runDir,
5407
+ nodeId: input.task.id,
5408
+ attemptId: `${meta.runId}:${input.task.id}`,
5409
+ workspaceRoot: input.cwd,
5410
+ sourceDeclaredPaths: scoutSourceDeclaredPaths,
5411
+ readBudgetTools: readBudgetTools?.customTools,
5412
+ });
5413
+ }
5414
+ else if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
5415
+ // Frontend-only split: independent coverage map sessions feed a single
5416
+ // reducer (UX decisions -> global policy -> finalize), mirroring the
5417
+ // backend-test template's module sharding. Small plans normally have one
5418
+ // coverage batch and retain the compact path below.
4292
5419
  let planRequirementIds = [];
4293
5420
  const planRequirementCosts = new Map();
4294
5421
  const behaviorRequiredRequirementIds = [];
@@ -4306,12 +5433,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4306
5433
  ?.kind === "requirement")
4307
5434
  .map((record) => record.fact?.id)
4308
5435
  .filter((id) => typeof id === "string");
4309
- for (const record of contractFacts) {
4310
- const fact = record.fact;
4311
- if (fact?.kind !== "requirement" || typeof fact.id !== "string")
4312
- continue;
4313
- const evidence = fact.evidence;
4314
- if (evidence?.behavior === "required")
5436
+ for (const fact of resolveFrontendContractRequirements(contractFacts.map((record) => record.fact))) {
5437
+ if (fact.evidence.behavior === "required")
4315
5438
  behaviorRequiredRequirementIds.push(fact.id);
4316
5439
  planRequirementCosts.set(fact.id, estimateFrontendPlanRequirementRecordCalls(fact));
4317
5440
  }
@@ -4319,26 +5442,189 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4319
5442
  catch {
4320
5443
  // Unreadable ledger falls back to a single coverage session.
4321
5444
  }
4322
- result = await runFrontendPlanSegmentedSessions({
5445
+ // Independent requirement-coverage batches are map workers. Each worker
5446
+ // owns an isolated typed-event store; only after all workers settle do we
5447
+ // merge facts into the main Plan ledger and run the single UX/global/
5448
+ // finalize reducer. This avoids revision races while shortening the
5449
+ // longest coverage phase for large plans.
5450
+ const runPlanSessions = (options) => runFrontendPlanSegmentedSessions({
4323
5451
  piStepFn,
4324
- sessionOptions: piSessionOptions,
4325
- basePrompt: input.prompt,
5452
+ sessionOptions: options.sessionOptions,
5453
+ basePrompt: options.basePrompt ?? input.prompt,
4326
5454
  attempt: input.attempt ?? 1,
4327
- committedFactCount: () => planLedgerTools.committedFactCount(),
4328
- requirementIds: planRequirementIds,
5455
+ committedFactCount: () => options.ledgerTools.committedFactCount(),
5456
+ requirementIds: options.requirementIds ?? planRequirementIds,
4329
5457
  requirementCosts: planRequirementCosts,
4330
- compactSmallPlan: true,
4331
- committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
4332
- committedFacts: () => planLedgerTools.committedFacts(),
5458
+ ...(options.parallelCoverageOnly !== undefined
5459
+ ? { parallelCoverageOnly: options.parallelCoverageOnly }
5460
+ : {}),
5461
+ ...(options.compactSmallPlan !== undefined
5462
+ ? { compactSmallPlan: options.compactSmallPlan }
5463
+ : {}),
5464
+ committedRequirementIds: () => options.ledgerTools.committedRequirementIds(),
5465
+ committedFacts: () => options.ledgerTools.committedFacts(),
4333
5466
  behaviorRequiredRequirementIds,
4334
- setActiveRequirementScope: (requirementIds) => planLedgerTools.setActiveRequirementScope(requirementIds),
5467
+ setActiveRequirementScope: (requirementIds) => options.ledgerTools.setActiveRequirementScope(requirementIds),
4335
5468
  segmentCustomTools: (toolNames) => toolNames === null
4336
- ? planLedgerTools.customTools
4337
- : planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
5469
+ ? options.ledgerTools.customTools
5470
+ : options.ledgerTools.customTools.filter((tool) => typeof tool === "object" &&
4338
5471
  tool !== null &&
4339
5472
  toolNames.has(tool.name)),
4340
- flushLedger: () => planLedgerTools.flush(),
5473
+ flushLedger: () => options.ledgerTools.flush(),
4341
5474
  });
5475
+ if (planRequirementIds.length > 1) {
5476
+ const estimatedPlanCalls = planRequirementIds.reduce((total, id) => total + Math.max(1, planRequirementCosts.get(id) ?? 2), 0);
5477
+ const compactEligibleBeforeSharding = planRequirementIds.length <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
5478
+ estimatedPlanCalls > 12 &&
5479
+ estimatedPlanCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
5480
+ countFrontendPlanTargetSurfaces(input.prompt) === 1;
5481
+ if (compactEligibleBeforeSharding) {
5482
+ // Decide the small-plan topology before creating coverage shards.
5483
+ // Sharding first would make the compact two-session path unreachable.
5484
+ result = await runPlanSessions({
5485
+ ledgerTools: planLedgerTools,
5486
+ sessionOptions: piSessionOptions,
5487
+ compactSmallPlan: true,
5488
+ });
5489
+ }
5490
+ else {
5491
+ const coverageBatches = batchFrontendPlanRequirements({
5492
+ requirementIds: planRequirementIds,
5493
+ maxEstimatedRecordCalls: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS,
5494
+ maxRequirements: 4,
5495
+ requirementCosts: planRequirementCosts,
5496
+ });
5497
+ if (coverageBatches.length > 1) {
5498
+ const shardResults = await mapWithConcurrency(coverageBatches, FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY, async (slice, index) => {
5499
+ let shardTools;
5500
+ let shardResult;
5501
+ try {
5502
+ const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
5503
+ shardTools = await createFrontendPlanLedgerTools({
5504
+ attemptId: `${meta.runId}:${input.task.id}:parallel:${index + 1}`,
5505
+ store: createTypedEventStore(),
5506
+ runDir: meta.runDir,
5507
+ nodeId: `${input.task.id}/parallel/coverage-${index + 1}`,
5508
+ skeleton: input.task.structuredContractOutput?.skeleton,
5509
+ sourceBinding: meta.spec.sourceBinding,
5510
+ requirementIds: slice,
5511
+ writeSetPatterns: input.task.writeSet,
5512
+ canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
5513
+ runDir: meta.runDir,
5514
+ }),
5515
+ componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
5516
+ cwd: input.cwd,
5517
+ sourceBinding: meta.spec.sourceBinding,
5518
+ }),
5519
+ });
5520
+ const shardSessionOptions = {
5521
+ ...piSessionOptions,
5522
+ sessionEventsPath: path.join(meta.runDir, input.task.id, "parallel", `coverage-${index + 1}`, "session-events.jsonl"),
5523
+ };
5524
+ shardResult = await runPlanSessions({
5525
+ ledgerTools: shardTools,
5526
+ sessionOptions: shardSessionOptions,
5527
+ requirementIds: slice,
5528
+ basePrompt: `${input.prompt}\n\nPARALLEL COVERAGE SHARD ${index + 1}: use the frozen canonical behavior verification-target ids listed in the record_plan_verification_target tool description — do not prefix ids with a shard namespace or invent variant ids; identical cross-shard targets are deduped, divergent ones fail the merge.`,
5529
+ parallelCoverageOnly: true,
5530
+ });
5531
+ await shardTools.flush();
5532
+ return { index, result: shardResult, tools: shardTools };
5533
+ }
5534
+ catch (error) {
5535
+ const crashMessage = `frontend plan coverage shard ${index + 1} crashed: ${error instanceof Error ? error.message : String(error)}`;
5536
+ return {
5537
+ index,
5538
+ tools: shardTools,
5539
+ result: shardResult
5540
+ ? {
5541
+ ...shardResult,
5542
+ ok: false,
5543
+ failureCategory: shardResult.ok
5544
+ ? "invalid-output"
5545
+ : shardResult.failureCategory,
5546
+ stderr: [shardResult.stderr, crashMessage]
5547
+ .filter(Boolean)
5548
+ .join("\n"),
5549
+ }
5550
+ : {
5551
+ ok: false,
5552
+ assistantText: "",
5553
+ command: [],
5554
+ durationMs: 0,
5555
+ exitCode: null,
5556
+ failureCategory: "tool-policy",
5557
+ modelDisplay: "unknown",
5558
+ parsedEvents: 0,
5559
+ stderr: crashMessage,
5560
+ stdout: "",
5561
+ timedOut: false,
5562
+ attemptedModels: [],
5563
+ fallbackUsed: false,
5564
+ tokensUsed: 0,
5565
+ },
5566
+ };
5567
+ }
5568
+ });
5569
+ const coverageResult = aggregateParallelPiResults(shardResults.map((shard) => shard.result));
5570
+ let mergeFailure;
5571
+ try {
5572
+ // Flatten every shard's facts FIRST, then pre-reduce: shards
5573
+ // partition requirements but a frozen verification target can
5574
+ // span shards, so same-id targets must merge across shards
5575
+ // before adoption, not per shard.
5576
+ await planLedgerTools.adoptCommittedFacts(reduceParallelCoverageShardRecords(shardResults
5577
+ .filter((item) => item.result.ok && item.tools)
5578
+ .flatMap((shard) => normalizeParallelCoverageShardRecords(shard.tools.committedFacts(), shard.index + 1))));
5579
+ }
5580
+ catch (error) {
5581
+ mergeFailure = error;
5582
+ }
5583
+ const failedShard = shardResults.find((shard) => !shard.result.ok);
5584
+ if (mergeFailure) {
5585
+ result = {
5586
+ ...coverageResult,
5587
+ ok: false,
5588
+ failureCategory: "invalid-output",
5589
+ stderr: [
5590
+ coverageResult.stderr,
5591
+ `frontend plan coverage shard merge failed: ${mergeFailure instanceof Error ? mergeFailure.message : String(mergeFailure)}`,
5592
+ ]
5593
+ .filter(Boolean)
5594
+ .join("\n"),
5595
+ };
5596
+ }
5597
+ else if (failedShard) {
5598
+ result = {
5599
+ ...coverageResult,
5600
+ ok: false,
5601
+ stderr: `${coverageResult.stderr}\nfrontend plan coverage shard ${failedShard.index + 1} failed before reduce`.trim(),
5602
+ };
5603
+ }
5604
+ else {
5605
+ const reducerResult = await runPlanSessions({
5606
+ ledgerTools: planLedgerTools,
5607
+ sessionOptions: piSessionOptions,
5608
+ });
5609
+ result = combineSequentialPiResults(coverageResult, reducerResult);
5610
+ }
5611
+ }
5612
+ else {
5613
+ result = await runPlanSessions({
5614
+ ledgerTools: planLedgerTools,
5615
+ sessionOptions: piSessionOptions,
5616
+ compactSmallPlan: true,
5617
+ });
5618
+ }
5619
+ }
5620
+ }
5621
+ else {
5622
+ result = await runPlanSessions({
5623
+ ledgerTools: planLedgerTools,
5624
+ sessionOptions: piSessionOptions,
5625
+ compactSmallPlan: true,
5626
+ });
5627
+ }
4342
5628
  }
4343
5629
  else {
4344
5630
  result = await piStepFn({
@@ -4885,7 +6171,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4885
6171
  runDir: meta.runDir,
4886
6172
  progress,
4887
6173
  attempt: input.attempt ?? 1,
4888
- maxAttempts: input.task.retryPolicy?.maxAttempts ?? 3,
6174
+ maxAttempts: input.task.retryPolicy?.maxAttempts ?? 5,
4889
6175
  });
4890
6176
  if (progress.status !== "PASS") {
4891
6177
  const classified = classifyBackendTestWriterCompletenessFailure(progress);
@@ -4935,7 +6221,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4935
6221
  stderrParts.push(completenessFailure.detail);
4936
6222
  }
4937
6223
  if (writerThinkingExhausted) {
4938
- stderrParts.push(`${WRITER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 write tool calls, 0 attributed diff; recommend a model switch and a new run`);
6224
+ stderrParts.push(`${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before any write tool call; retry from the existing workspace and complete only unfinished targets`);
4939
6225
  }
4940
6226
  if (meta.writeGuardAttribution === "best-effort") {
4941
6227
  stderrParts.push("write guard note: concurrent rank writers use best-effort per-node attribution; keep same-rank writeSet entries disjoint");
@@ -5008,14 +6294,11 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
5008
6294
  rawFailureCategory === "context-overflow" &&
5009
6295
  (changeManifestChangedFiles?.length ?? 0) > 0
5010
6296
  ? "partial-success-with-context-overflow"
5011
- : // writer-thinking-exhausted: a length-stopped thinking-only attempt with
5012
- // zero write tool calls and zero attributed diff is terminal and
5013
- // non-retryable; recommend a model switch + fresh run. Only applies on
5014
- // the empty-output base category so provider/transport failures keep
5015
- // their original category, and only when the completeness gate did not
5016
- // upgrade to incomplete-write-set above.
6297
+ : // output-limit: a length-stopped attempt is capacity truncation, not
6298
+ // empty output. Preserve the raw category for diagnostics and let the
6299
+ // node retry from committed facts/current workspace state.
5017
6300
  writerThinkingExhausted
5018
- ? WRITER_THINKING_EXHAUSTED_CATEGORY
6301
+ ? OUTPUT_LIMIT_RETRY_CATEGORY
5019
6302
  : writerCleanTimeout
5020
6303
  ? WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY
5021
6304
  : // writer-budget-exhausted: the provider session consumed an