@tea-agent/loop-agent 0.42.0-next.15 → 0.42.0-next.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/dist/application/task-lifecycle/advance.js +9 -4
  3. package/dist/build-stamp.json +3 -3
  4. package/dist/commands/task-source-prepare.js +3 -1
  5. package/dist/executors/dag-pi-executor.js +825 -65
  6. package/dist/executors/shell-executor.js +103 -35
  7. package/dist/shared/dag-failure-category.js +6 -0
  8. package/dist/task/contract/apply.js +36 -2
  9. package/dist/task/source-prepare/parse-intent.js +7 -0
  10. package/dist/workflows/dag/dag-retry-schema.js +3 -0
  11. package/dist/workflows/dag/frontend-design-policy.js +1 -1
  12. package/dist/workflows/dag/frontend-risk.js +2 -0
  13. package/dist/workflows/dag/frontend-shadow-dual-write.js +16 -1
  14. package/dist/workflows/dag/frontend-shape.js +16 -6
  15. package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
  16. package/dist/workflows/dag/frontend-verification-trace.js +32 -0
  17. package/dist/workflows/dag/init-hybrid.js +9 -10
  18. package/dist/workflows/dag/node-execution.js +124 -72
  19. package/dist/workflows/dag/rerun-feedback.js +1 -0
  20. package/dist/workflows/dag/rerun-plan.js +90 -4
  21. package/dist/workflows/dag/rerun-run.js +7 -0
  22. package/dist/workflows/dag/retry-policy.js +16 -10
  23. package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
  24. package/dist/workflows/dag/runner.js +67 -7
  25. package/dist/workflows/dag/structured-output-repair.js +4 -1
  26. package/dist/workflows/dag/validate.js +10 -8
  27. package/dist/workflows/dag/workspace-checkpoint.js +66 -0
  28. package/docs/templates/agent-dag.schema.json +2 -2
  29. package/package.json +1 -1
  30. package/skills/frontend-plan/SKILL.md +6 -1
  31. package/skills/frontend-plan/references/decision-contract.md +69 -18
  32. package/skills/frontend-plan/references/design-decisions.md +32 -0
@@ -17,9 +17,9 @@ import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation
17
17
  import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
18
18
  import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
19
19
  import { pathMatchesPattern } from "../shared/git-progress.js";
20
- import { readTypedEventStoreFromJsonl } from "../workflows/dag/frontend-typed-event-store.js";
20
+ import { readTypedEventStoreFromJsonl, typedEventPayloadSha256, } from "../workflows/dag/frontend-typed-event-store.js";
21
21
  import { collectCanonicalStateFlowNames, frontendEvidenceExpectationSchema, resolveFrontendContractRequirements, validateFrontendRequiredDeliverables, } from "../workflows/dag/frontend-contract-facts.js";
22
- import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
22
+ import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
23
23
  import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestCollectionRepairOutcomeRecoveryCandidate, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
24
24
  import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
25
25
  import { assessBackendTestPlanProtocol } from "../workflows/dag/backend-test-plan-protocol.js";
@@ -27,23 +27,221 @@ import { frontendTestLayoutFromSpec } from "../workflows/dag/frontend-test-layou
27
27
  import { redactSecrets, truncateUtf8Preview } from "../shared/preview.js";
28
28
  import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.js";
29
29
  /**
30
- * Writer classification for a length-stopped thinking-only attempt: the model
31
- * exhausted its output budget thinking but never issued a write/edit tool call
32
- * and produced zero attributed diff. This is a terminal, non-retryable
33
- * diagnosis (the same prompt + model will hit the same budget wall); the
34
- * recommendation is a model switch plus a fresh run. It must NOT mask a
35
- * recoverable partial-write-set (incomplete-write-set) upgrade.
30
+ * Legacy diagnostic label retained for artifact compatibility. New executions
31
+ * classify this signal as output-limit so the node can retry incrementally.
36
32
  */
37
33
  export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
38
34
  /**
39
- * Planner classification mirroring writer-thinking-exhausted: a read-only
40
- * planning session stopped on length, observed thinking, and committed zero
41
- * typed facts with no assistant text. By the time this survives the segmented
42
- * ladder the scope has already been degraded, so the durable fix is a
43
- * thinking-capped or non-thinking model for the tier — not another replay of
44
- * the same full-scope prompt.
35
+ * Legacy diagnostic label retained for artifact compatibility. New executions
36
+ * classify this signal as output-limit and preserve committed typed facts.
45
37
  */
46
38
  export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
39
+ /**
40
+ * Parallel coverage shards namespace their verification target ids with
41
+ * `VT-SHARD-<shard>-` so the reducer can detect cross-shard conflicts. Once
42
+ * every shard's facts are known the namespace must be restored: the canonical
43
+ * contract has to carry the exact ids the task source froze (a surviving
44
+ * `VT-SHARD-N-…` id is a contract-requirement gap that design review
45
+ * rejects). Stripping happens per shard before adoption — after the strip,
46
+ * the existing identity-conflict check catches genuine cross-shard
47
+ * duplicate base ids and fails the merge closed.
48
+ */
49
+ /**
50
+ * Deterministic pre-reduce for parallel coverage shard facts. Shards partition
51
+ * requirements, but a frozen verification target can legitimately be derived
52
+ * by several shards (one VT covers multiple ACs across shard boundaries), so
53
+ * same-id verification targets are MERGED: requirementIds and uiStates union,
54
+ * while divergent file/commandId is a real conflict. Everything else passes
55
+ * through unchanged.
56
+ */
57
+ export function reduceParallelCoverageShardRecords(records) {
58
+ // Resolve replacements inside each source shard before comparing shards.
59
+ // A correction is an event-log operation, not a second independent target;
60
+ // retaining both records makes a valid same-shard correction look like a
61
+ // cross-shard conflict during the later reduction.
62
+ const byShard = new Map();
63
+ for (const record of records) {
64
+ const shard = byShard.get(record.attemptId) ?? [];
65
+ shard.push(record);
66
+ byShard.set(record.attemptId, shard);
67
+ }
68
+ const effectiveRecords = [];
69
+ for (const shardRecords of byShard.values()) {
70
+ const effective = [];
71
+ for (const record of shardRecords) {
72
+ const fact = record.fact;
73
+ const entry = fact?.entry;
74
+ const kind = typeof fact?.kind === "string" ? fact.kind : "";
75
+ const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
76
+ typeof entry?.id === "string"
77
+ ? `${kind}:${entry.id}`
78
+ : undefined;
79
+ if (!identity) {
80
+ effective.push(record);
81
+ continue;
82
+ }
83
+ const replaces = typeof fact?.replaces === "string" ? `${kind}:${fact.replaces}` : undefined;
84
+ if (replaces) {
85
+ for (let index = effective.length - 1; index >= 0; index -= 1) {
86
+ const prior = effective[index];
87
+ const priorFact = prior.fact;
88
+ const priorEntry = priorFact?.entry;
89
+ const priorIdentity = typeof priorFact?.kind === "string" &&
90
+ typeof priorEntry?.id === "string"
91
+ ? `${priorFact.kind}:${priorEntry.id}`
92
+ : undefined;
93
+ if (priorIdentity === replaces)
94
+ effective.splice(index, 1);
95
+ }
96
+ }
97
+ // Keep the source record immutable. The replacement is represented by
98
+ // the latest event and its own payload hash.
99
+ effective.push(record);
100
+ }
101
+ effectiveRecords.push(...effective);
102
+ }
103
+ const verificationTargets = new Map();
104
+ const merged = [];
105
+ for (const record of effectiveRecords) {
106
+ const fact = record.fact;
107
+ if (!fact || typeof fact !== "object")
108
+ continue;
109
+ if (fact.kind !== "plan-verification-target") {
110
+ merged.push(record);
111
+ continue;
112
+ }
113
+ const entry = fact.entry;
114
+ if (!entry || typeof entry.id !== "string")
115
+ continue;
116
+ const existing = verificationTargets.get(entry.id);
117
+ if (!existing) {
118
+ verificationTargets.set(entry.id, {
119
+ record, entry: { ...entry },
120
+ sources: [`${record.attemptId}:${record.eventId}:${record.payloadSha256}`],
121
+ });
122
+ continue;
123
+ }
124
+ // Never mutate the entry held by the source record. The reducer creates
125
+ // a new merged entry and recomputes the payload hash for that derived
126
+ // fact, leaving source-event replay integrity intact.
127
+ const baseEntry = { ...existing.entry };
128
+ for (const field of ["commandId", "commandLabel", "file", "scope"]) {
129
+ if (entry[field] !== baseEntry[field]) {
130
+ throw new Error(`frontend plan coverage shard reduce conflict: verification target ${entry.id} has divergent ${field} "${String(entry[field])}" vs "${String(baseEntry[field])}"`);
131
+ }
132
+ }
133
+ const unionSorted = (a, b) => {
134
+ const left = Array.isArray(a) ? a : [];
135
+ const right = Array.isArray(b) ? b : [];
136
+ return [...new Set([...left, ...right])].sort();
137
+ };
138
+ baseEntry.requirementIds = unionSorted(baseEntry.requirementIds, entry.requirementIds);
139
+ baseEntry.uiStates = unionSorted(baseEntry.uiStates, entry.uiStates);
140
+ const mergedFact = { ...fact, entry: { ...baseEntry } };
141
+ existing.entry = baseEntry;
142
+ existing.sources.push(`${record.attemptId}:${record.eventId}:${record.payloadSha256}`);
143
+ existing.record = {
144
+ ...existing.record,
145
+ fact: mergedFact,
146
+ payloadSha256: typedEventPayloadSha256(mergedFact),
147
+ };
148
+ }
149
+ merged.push(...[...verificationTargets.values()].map((item) => {
150
+ // The reducer owns every VT revision, including a one-shard partial
151
+ // result. Content-addressed revisions replay idempotently, while an
152
+ // expanded/replaced source set appends an explicit replacement rather
153
+ // than masquerading as a changed source event or deleting audit facts.
154
+ const fact = {
155
+ ...item.record.fact,
156
+ replaces: item.entry.id,
157
+ entry: {
158
+ ...item.entry,
159
+ requirementIds: [...new Set(planFactStringList(item.entry.requirementIds))].sort(),
160
+ uiStates: [...new Set(planFactStringList(item.entry.uiStates))].sort(),
161
+ },
162
+ };
163
+ const payloadSha256 = typedEventPayloadSha256(fact);
164
+ const revisionId = createHash("sha256")
165
+ .update(JSON.stringify({ sources: [...new Set(item.sources)].sort(), payloadSha256 }))
166
+ .digest("hex");
167
+ return {
168
+ ...item.record,
169
+ attemptId: "frontend-plan-coverage-reducer",
170
+ eventId: `coverage-reduced:${revisionId}`,
171
+ requestId: `coverage-reduced:${revisionId}`,
172
+ fact, payloadSha256,
173
+ };
174
+ }));
175
+ return merged;
176
+ }
177
+ export function normalizeParallelCoverageShardRecords(records, shardNumber) {
178
+ const prefix = `VT-SHARD-${shardNumber}-`;
179
+ // Models sometimes drop the `VT-` stem when applying the shard namespace
180
+ // (observed: frozen `VT-X` became `VT-SHARD-2-X`), so restoring the
181
+ // canonical id requires re-adding the stem after the strip.
182
+ const stripId = (id) => {
183
+ if (typeof id !== "string" || !id.startsWith(prefix))
184
+ return id;
185
+ const base = id.slice(prefix.length);
186
+ return base.startsWith("VT-") ? base : `VT-${base}`;
187
+ };
188
+ const stripIdList = (ids) => Array.isArray(ids) ? ids.map((id) => stripId(id)) : ids;
189
+ return records.map((record) => {
190
+ const fact = record.fact;
191
+ if (!fact || typeof fact !== "object")
192
+ return record;
193
+ const kind = fact.kind;
194
+ const entry = fact.entry;
195
+ if (!entry || typeof entry !== "object")
196
+ return record;
197
+ let rewritten;
198
+ const replaces = kind === "plan-verification-target" ? stripId(fact.replaces) : fact.replaces;
199
+ if (kind === "plan-verification-target") {
200
+ const id = stripId(entry.id);
201
+ if (id !== entry.id || replaces !== fact.replaces)
202
+ rewritten = { ...entry, id };
203
+ }
204
+ else if (kind === "plan-requirement") {
205
+ const verificationTargetIds = stripIdList(entry.verificationTargetIds);
206
+ if (verificationTargetIds !== entry.verificationTargetIds) {
207
+ rewritten = { ...entry, verificationTargetIds };
208
+ }
209
+ }
210
+ else if (kind === "state-flow") {
211
+ const rewriteBoundTargets = (item) => {
212
+ if (!item || typeof item !== "object")
213
+ return item;
214
+ return {
215
+ ...item,
216
+ verificationTargetIds: stripIdList(item.verificationTargetIds),
217
+ };
218
+ };
219
+ const uiStates = Array.isArray(entry.uiStates)
220
+ ? entry.uiStates.map(rewriteBoundTargets)
221
+ : entry.uiStates;
222
+ const interactions = Array.isArray(entry.interactions)
223
+ ? entry.interactions.map(rewriteBoundTargets)
224
+ : entry.interactions;
225
+ if (uiStates !== entry.uiStates || interactions !== entry.interactions) {
226
+ rewritten = { ...entry, uiStates, interactions };
227
+ }
228
+ }
229
+ return rewritten
230
+ ? {
231
+ ...record,
232
+ fact: { ...fact, ...(replaces !== undefined ? { replaces } : {}), entry: rewritten },
233
+ // Keep the integrity hash consistent with the rewritten
234
+ // payload, or the adoption replay check reports the
235
+ // normalized record as a tampered source event.
236
+ payloadSha256: typedEventPayloadSha256({
237
+ ...fact,
238
+ ...(replaces !== undefined ? { replaces } : {}),
239
+ entry: rewritten,
240
+ }),
241
+ }
242
+ : record;
243
+ });
244
+ }
47
245
  export function isPlannerThinkingExhausted(result, committedAnyFacts) {
48
246
  if (result.ok)
49
247
  return false;
@@ -985,6 +1183,36 @@ async function resolveFrontendDeclaredUiStateIds(input) {
985
1183
  return [];
986
1184
  }
987
1185
  }
1186
+ /**
1187
+ * Resolve the frozen canonical behavior verification-target ids the PRD
1188
+ * declares. The requirement text is the same authority the design review reads
1189
+ * when it rejects `VT-SHARD-N-…` / variant ids as a contract-requirement gap,
1190
+ * so extracting `VT-…` tokens from the frozen contract requirement facts makes
1191
+ * that authority deterministic instead of prose-only. An empty result (no PRD
1192
+ * declared any behavior target id) disables the canonical check and keeps the
1193
+ * historical free-form path.
1194
+ */
1195
+ export async function resolveFrontendCanonicalVerificationTargetIds(input) {
1196
+ try {
1197
+ const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
1198
+ const ids = new Set();
1199
+ for (const record of records) {
1200
+ const fact = record.fact;
1201
+ if (record.phase !== "committed" || fact.kind !== "requirement") {
1202
+ continue;
1203
+ }
1204
+ const text = typeof fact.text === "string" ? fact.text : "";
1205
+ for (const match of text.matchAll(/\bVT-[A-Z0-9][A-Z0-9_-]*\b/g)) {
1206
+ ids.add(match[0]);
1207
+ }
1208
+ }
1209
+ return [...ids].sort();
1210
+ }
1211
+ catch {
1212
+ // No contract facts (or unreadable): fall back to no canonical set.
1213
+ return [];
1214
+ }
1215
+ }
988
1216
  /**
989
1217
  * A+B: `frontend-plan-pi` records its decision ledger through incremental
990
1218
  * `record_*` tools (origin=plan) and closes with exactly one
@@ -1717,6 +1945,15 @@ export async function createFrontendPlanLedgerTools(input) {
1717
1945
  });
1718
1946
  }
1719
1947
  }
1948
+ if (id &&
1949
+ activeRequirementScope.length > 0 &&
1950
+ !activeRequirementScope.includes(id)) {
1951
+ return planToolReceipt({
1952
+ ok: false,
1953
+ kind: "plan-requirement",
1954
+ error: `record_plan_requirement id "${id}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
1955
+ });
1956
+ }
1720
1957
  const result = await adoptPlanFact("plan-requirement", `${attemptId}:record_plan_requirement:${randomUUID()}`, {
1721
1958
  kind: "plan-requirement",
1722
1959
  origin: "plan",
@@ -1729,7 +1966,11 @@ export async function createFrontendPlanLedgerTools(input) {
1729
1966
  const recordPlanVerificationTargetTool = defineTool({
1730
1967
  name: "record_plan_verification_target",
1731
1968
  label: "record_plan_verification_target",
1732
- description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}",
1969
+ description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}" +
1970
+ (input.canonicalVerificationTargetIds &&
1971
+ input.canonicalVerificationTargetIds.length > 0
1972
+ ? ` Frozen canonical behavior target ids (use exactly for behavior targets): ${input.canonicalVerificationTargetIds.join(", ")}.`
1973
+ : ""),
1733
1974
  promptSnippet: "Commit 1-4 plan verification target entries (up to 4 per message).",
1734
1975
  parameters: Type.Object({
1735
1976
  entry: verificationTargetSchema,
@@ -1780,6 +2021,24 @@ export async function createFrontendPlanLedgerTools(input) {
1780
2021
  error: `record_plan_verification_target verification-target-phase-mismatch: behavior command "${directoryEntry.label}" (${directoryEntry.commandId}) must bind a test file (__tests__/, tests?/, e2e/, cypress/, *.test.*, *.spec.*, *.cy.*); received file "${rawEntry.file}"`,
1781
2022
  });
1782
2023
  }
2024
+ // Canonical-identity check for behavior targets: the PRD freezes
2025
+ // the exact behavior verification-target ids (e.g.
2026
+ // VT-SMOKE-COUNTER-BEHAVIOR). A committed non-canonical id is
2027
+ // immutable and design review rejects it as a
2028
+ // contract-requirement gap, so reject invented ids here.
2029
+ if (directoryEntry.mode === "behavior" &&
2030
+ input.canonicalVerificationTargetIds &&
2031
+ input.canonicalVerificationTargetIds.length > 0) {
2032
+ const canonicalTargetId = typeof rawEntry.id === "string" ? rawEntry.id.trim() : "";
2033
+ if (canonicalTargetId &&
2034
+ !input.canonicalVerificationTargetIds.includes(canonicalTargetId)) {
2035
+ return planToolReceipt({
2036
+ ok: false,
2037
+ kind: "plan-verification-target",
2038
+ error: `record_plan_verification_target id "${canonicalTargetId}" is not a frozen canonical behavior verification target; canonical ids are: ${input.canonicalVerificationTargetIds.join(", ")}`,
2039
+ });
2040
+ }
2041
+ }
1783
2042
  }
1784
2043
  // Duplicate-id rejection: committed typed facts are immutable, so
1785
2044
  // re-recording the same VT id would deadlock the compile by default.
@@ -1849,6 +2108,15 @@ export async function createFrontendPlanLedgerTools(input) {
1849
2108
  : [];
1850
2109
  }));
1851
2110
  const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
2111
+ const outOfScopeRequirementIds = stringList(rawEntry.requirementIds).filter((id) => activeRequirementScope.length > 0 &&
2112
+ !activeRequirementScope.includes(id));
2113
+ if (outOfScopeRequirementIds.length > 0) {
2114
+ return planToolReceipt({
2115
+ ok: false,
2116
+ kind: "plan-verification-target",
2117
+ error: `record_plan_verification_target references requirements outside this session's scope [${activeRequirementScope.join(", ")}]: ${outOfScopeRequirementIds.join(", ")}`,
2118
+ });
2119
+ }
1852
2120
  if (unknownRequirementIds.length > 0) {
1853
2121
  return planToolReceipt({
1854
2122
  ok: false,
@@ -1895,6 +2163,18 @@ export async function createFrontendPlanLedgerTools(input) {
1895
2163
  error: "record_plan_evidence_gap requires a non-empty description describing the gap",
1896
2164
  });
1897
2165
  }
2166
+ const requirementId = typeof entry.requirementId === "string"
2167
+ ? entry.requirementId
2168
+ : undefined;
2169
+ if (requirementId &&
2170
+ activeRequirementScope.length > 0 &&
2171
+ !activeRequirementScope.includes(requirementId)) {
2172
+ return planToolReceipt({
2173
+ ok: false,
2174
+ kind: "plan-evidence-gap",
2175
+ error: `record_plan_evidence_gap requirementId "${requirementId}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
2176
+ });
2177
+ }
1898
2178
  const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
1899
2179
  return planToolReceipt(result);
1900
2180
  },
@@ -2088,6 +2368,66 @@ export async function createFrontendPlanLedgerTools(input) {
2088
2368
  adoptStagedFactTool,
2089
2369
  finalizePlanTool,
2090
2370
  ],
2371
+ adoptCommittedFacts: async (records) => {
2372
+ for (const record of records) {
2373
+ if (record.phase !== "committed")
2374
+ continue;
2375
+ const fact = record.fact;
2376
+ if (!fact || typeof fact !== "object" || Array.isArray(fact))
2377
+ continue;
2378
+ const kind = typeof fact.kind === "string"
2379
+ ? fact.kind
2380
+ : "plan-fact";
2381
+ const entry = fact.entry;
2382
+ const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
2383
+ typeof entry?.id === "string"
2384
+ ? `${kind}:${entry.id}`
2385
+ : undefined;
2386
+ const sourceShardPrefix = `${attemptId}:parallel-merge:${record.attemptId}:`;
2387
+ const requestId = `${sourceShardPrefix}${record.eventId}`;
2388
+ const committed = readCommittedEvents(store, attemptId);
2389
+ const replay = committed.find((candidate) => candidate.requestId === requestId);
2390
+ if (replay) {
2391
+ if (replay.payloadSha256 !== record.payloadSha256) {
2392
+ throw new Error(`frontend plan shard fact merge replay conflict: source event ${record.eventId} changed payload`);
2393
+ }
2394
+ continue;
2395
+ }
2396
+ const explicitReplacement = typeof fact.replaces === "string" &&
2397
+ fact.replaces === entry?.id;
2398
+ const conflicting = committed.find((candidate) => {
2399
+ const candidateFact = candidate.fact;
2400
+ const candidateEntry = candidateFact.entry;
2401
+ const sameSourceShard = candidate.requestId.startsWith(sourceShardPrefix);
2402
+ return (candidateFact.kind === kind &&
2403
+ typeof candidateEntry?.id === "string" &&
2404
+ `${kind}:${candidateEntry.id}` === identity &&
2405
+ !(sameSourceShard && explicitReplacement));
2406
+ });
2407
+ if (conflicting) {
2408
+ // Two shards may honestly emit the same fact (e.g. both
2409
+ // derive the same frozen verification target). Identical
2410
+ // payloads are duplicates to skip; divergent payloads are
2411
+ // a real conflict the ladder must resolve.
2412
+ if (conflicting.payloadSha256 === record.payloadSha256) {
2413
+ continue;
2414
+ }
2415
+ const fromSameSourceShard = conflicting.requestId.startsWith(sourceShardPrefix);
2416
+ if (fromSameSourceShard) {
2417
+ throw new Error(`frontend plan shard fact merge conflict: ${identity} was already committed by another shard with a different payload`);
2418
+ }
2419
+ // Divergent payload from a different source attempt means a
2420
+ // ladder retry re-derived the coverage phase: the fresh
2421
+ // derivation supersedes the stored record.
2422
+ const storeWithRecords = store;
2423
+ storeWithRecords.records = storeWithRecords.records.filter((candidate) => candidate.eventId !== conflicting.eventId);
2424
+ }
2425
+ const result = await adoptPlanFact(kind, requestId, fact);
2426
+ if (!result.ok) {
2427
+ throw new Error(`frontend plan shard fact merge failed: ${result.error}`);
2428
+ }
2429
+ }
2430
+ },
2091
2431
  setActiveRequirementScope: (requirementIds) => {
2092
2432
  activeRequirementScope = [
2093
2433
  ...new Set(requirementIds.filter((id) => id.trim().length > 0)),
@@ -2717,6 +3057,22 @@ export async function createFrontendScoutEvidenceTools(input) {
2717
3057
  });
2718
3058
  return {
2719
3059
  customTools: [recordTargetSurfaceTool, recordDesignEvidenceTool],
3060
+ committedFacts: () => readCommittedEvents(store, attemptId),
3061
+ adoptCommittedFacts: async (records) => {
3062
+ for (const record of records) {
3063
+ if (record.phase !== "committed")
3064
+ continue;
3065
+ const fact = record.fact;
3066
+ if (!fact || typeof fact !== "object" || Array.isArray(fact))
3067
+ continue;
3068
+ const result = await adoptScoutFact(typeof fact.kind === "string"
3069
+ ? fact.kind
3070
+ : "scout-fact", fact);
3071
+ if (!result.ok) {
3072
+ throw new Error(`frontend scout shard fact merge failed: ${result.error}`);
3073
+ }
3074
+ }
3075
+ },
2720
3076
  flush: async () => {
2721
3077
  const committed = readCommittedEvents(store, attemptId);
2722
3078
  await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "scout-typed-facts.jsonl"), committed);
@@ -3109,6 +3465,7 @@ const FRONTEND_PLAN_SEGMENTS = [
3109
3465
  instruction: [
3110
3466
  "PLAN PHASE — requirement coverage only.",
3111
3467
  "Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target facts (verification target bound to requirement ids and files). Group related requirements under one non-static behavior target when one observable test behavior proves them together; do not mechanically create one target per requirement. A non-static target id is the stable machine trace token; never submit prose as a symbol. Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
3468
+ "A coverage session is complete only when EVERY requirement assigned to this session (the full inventory, or the exact COVERAGE BATCH / shard list when present) has committed coverage facts: a record_plan_requirement entry plus verification targets, or a committed evidence gap. Keep committing in batches of up to 4 record_* calls per assistant message until then; do not write a concluding summary while any assigned requirement is still uncommitted — an early stop strands the remainder into a MISSING-FACT repair session and doubles the sessions needed.",
3112
3469
  "If a requirement genuinely cannot have a verification target, record a non-empty record_plan_evidence_gap. Do not call finalize_plan; it is not available in this phase.",
3113
3470
  ].join(" "),
3114
3471
  },
@@ -3126,11 +3483,12 @@ const FRONTEND_PLAN_SEGMENTS = [
3126
3483
  toolNames: new Set([
3127
3484
  "record_component_choice",
3128
3485
  "record_state_flow",
3486
+ "record_plan_verification_target",
3129
3487
  "adopt_staged_fact",
3130
3488
  ]),
3131
3489
  instruction: [
3132
3490
  "PLAN PHASE — global UX decisions.",
3133
- "Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Review the complete requirement set and the committed global UX registry together, then record each component choice, UI state and interaction exactly once. Multiple requirements that describe one behavior share one registry name and one state-flow entry; never repeat or rename it per AC. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
3491
+ "Requirements and verification targets are already committed in the ledger. Review the complete requirement set and the committed global UX registry together, then record each component choice, UI state and interaction. Bind each applicable state to its verificationTargetIds; the runtime derives the reverse VT.uiStates relation. If a VT requires correction, record_plan_verification_target with replace:true is available after declaring its states; preserve its requirement coverage. Multiple requirements describing one behavior share one registry name and state-flow entry. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
3134
3492
  "Do not call finalize_plan; it is not available in this phase.",
3135
3493
  ].join(" "),
3136
3494
  },
@@ -3374,12 +3732,28 @@ export function batchFrontendPlanRequirements(input) {
3374
3732
  return batches;
3375
3733
  }
3376
3734
  const FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS = 10;
3377
- // A large requirement set creates coverage shards; UX remains one global
3378
- // decision session so behavior names and component/state facts are not
3379
- // reinvented per AC batch. Keep a safety bound for adaptive coverage retries.
3735
+ const FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY = 4;
3736
+ // A large requirement set creates parallel coverage shards; UX remains one
3737
+ // global decision session so behavior names and component/state facts are not
3738
+ // reinvented per AC batch. Keep a safety bound for adaptive coverage retries
3739
+ // without letting the old 32-session ceiling skip finalize.
3380
3740
  const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
3381
3741
  const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
3382
3742
  const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
3743
+ async function mapWithConcurrency(items, limit, worker) {
3744
+ const results = new Array(items.length);
3745
+ let nextIndex = 0;
3746
+ const workerCount = Math.min(Math.max(1, limit), items.length);
3747
+ await Promise.all(Array.from({ length: workerCount }, async () => {
3748
+ while (true) {
3749
+ const index = nextIndex++;
3750
+ if (index >= items.length)
3751
+ return;
3752
+ results[index] = await worker(items[index], index);
3753
+ }
3754
+ }));
3755
+ return results;
3756
+ }
3383
3757
  function compactPromptString(value, maxChars) {
3384
3758
  if (typeof value !== "string" || value.trim().length === 0)
3385
3759
  return undefined;
@@ -3827,7 +4201,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3827
4201
  const buildCoveragePrompt = (slice, missing = []) => [
3828
4202
  compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
3829
4203
  coverageSegment.instruction,
3830
- `COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
4204
+ `COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them. Finish the whole list before concluding: commit every listed requirement's coverage facts (verification targets or an evidence gap), batching up to 4 record_* calls per message; an early stop re-queues the remainder as a MISSING-FACT repair session.`,
3831
4205
  ...(missing.length > 0
3832
4206
  ? [
3833
4207
  "MISSING-FACT QUEUE: the previous session did not establish complete coverage. Repair ONLY these items, then re-check the slice:",
@@ -3927,8 +4301,8 @@ export async function runFrontendPlanSegmentedSessions(input) {
3927
4301
  const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
3928
4302
  ? {
3929
4303
  ...r,
3930
- failureCategory: PLANNER_THINKING_EXHAUSTED_CATEGORY,
3931
- stderr: `${r.stderr}\n${PLANNER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 typed facts committed; the batch ladder degraded the scope without converging — set thinking=off for this tier or switch to a non-thinking model`.trim(),
4304
+ failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
4305
+ stderr: `${r.stderr}\n${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before the next typed fact; preserve committed facts and retry only the unfinished phase`.trim(),
3932
4306
  }
3933
4307
  : r;
3934
4308
  // An empty list means "ledger unreadable / unknown" and falls back to one
@@ -3943,6 +4317,29 @@ export async function runFrontendPlanSegmentedSessions(input) {
3943
4317
  : [];
3944
4318
  const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
3945
4319
  const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
4320
+ if (input.parallelCoverageOnly === true &&
4321
+ requirementIdsProvided &&
4322
+ coverageWorkIds.length === 0) {
4323
+ // A retry may reopen a shard whose committed facts are already complete.
4324
+ // Treat that shard as an idempotent no-op; returning the normal empty
4325
+ // session failure would make a partially failed map impossible to resume.
4326
+ return {
4327
+ ok: true,
4328
+ assistantText: "",
4329
+ command: [],
4330
+ durationMs: 0,
4331
+ exitCode: 0,
4332
+ failureCategory: "success",
4333
+ modelDisplay: "reused-coverage-facts",
4334
+ parsedEvents: 0,
4335
+ stderr: "",
4336
+ stdout: "",
4337
+ timedOut: false,
4338
+ attemptedModels: [],
4339
+ fallbackUsed: false,
4340
+ tokensUsed: 0,
4341
+ };
4342
+ }
3946
4343
  const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
3947
4344
  const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
3948
4345
  // Small, single-surface requests do not benefit from six isolated Pi
@@ -4004,7 +4401,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
4004
4401
  if (useCompactSmallPlan) {
4005
4402
  // Compact mode already queued both sessions above.
4006
4403
  }
4007
- else {
4404
+ else if (!input.parallelCoverageOnly) {
4008
4405
  if (requirementIdsProvided) {
4009
4406
  queue.push({
4010
4407
  id: uxRegistrySegment.id,
@@ -4041,7 +4438,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
4041
4438
  prompt: buildPhasePrompt(finalizeSegment),
4042
4439
  });
4043
4440
  }
4044
- let last;
4441
+ let accumulated;
4045
4442
  let index = 0;
4046
4443
  let invocationCount = 0;
4047
4444
  while (index < queue.length) {
@@ -4092,8 +4489,10 @@ export async function runFrontendPlanSegmentedSessions(input) {
4092
4489
  }
4093
4490
  }
4094
4491
  const committedBefore = input.committedFactCount();
4095
- input.setActiveRequirementScope?.(session.id === "compact-local" || session.id.startsWith("ux-local-")
4096
- ? session.requirementSlice ?? []
4492
+ input.setActiveRequirementScope?.(session.coverageOnly ||
4493
+ session.id === "compact-local" ||
4494
+ session.id.startsWith("ux-local-")
4495
+ ? session.coverageSlice ?? session.requirementSlice ?? []
4097
4496
  : []);
4098
4497
  if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
4099
4498
  break;
@@ -4111,7 +4510,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
4111
4510
  }
4112
4511
  : {}),
4113
4512
  });
4114
- last = result;
4513
+ accumulated = accumulated
4514
+ ? combineSequentialPiResults(accumulated, result)
4515
+ : result;
4115
4516
  try {
4116
4517
  await input.flushLedger();
4117
4518
  }
@@ -4220,9 +4621,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
4220
4621
  continue;
4221
4622
  }
4222
4623
  return {
4223
- ...result,
4624
+ ...accumulated,
4224
4625
  ok: false,
4225
- stderr: `${result.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
4626
+ stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
4226
4627
  failureCategory: "invalid-output",
4227
4628
  };
4228
4629
  }
@@ -4239,9 +4640,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
4239
4640
  }
4240
4641
  if (missingPhaseFacts.length > 0) {
4241
4642
  return {
4242
- ...result,
4643
+ ...accumulated,
4243
4644
  ok: false,
4244
- stderr: `frontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`,
4645
+ stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
4245
4646
  failureCategory: "invalid-output",
4246
4647
  };
4247
4648
  }
@@ -4332,11 +4733,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
4332
4733
  };
4333
4734
  continue;
4334
4735
  }
4335
- return mapPlannerExhaustion(result, committedAfter > committedBefore);
4736
+ return mapPlannerExhaustion(accumulated, committedAfter > committedBefore);
4336
4737
  }
4337
4738
  if (index < queue.length) {
4338
4739
  return {
4339
- ...(last ?? {
4740
+ ...(accumulated ?? {
4340
4741
  ok: false,
4341
4742
  stdout: "",
4342
4743
  stderr: "",
@@ -4353,11 +4754,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
4353
4754
  tokensUsed: 0,
4354
4755
  }),
4355
4756
  ok: false,
4356
- stderr: `${last?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
4757
+ stderr: `${accumulated?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
4357
4758
  failureCategory: "invalid-output",
4358
4759
  };
4359
4760
  }
4360
- return mapPlannerExhaustion(last ?? {
4761
+ return mapPlannerExhaustion(accumulated ?? {
4361
4762
  ok: false,
4362
4763
  stdout: "",
4363
4764
  stderr: "frontend plan segmentation produced no session",
@@ -4365,6 +4766,184 @@ export async function runFrontendPlanSegmentedSessions(input) {
4365
4766
  durationMs: 0,
4366
4767
  }, false);
4367
4768
  }
4769
+ /**
4770
+ * Run the two independent Scout evidence surfaces concurrently while keeping
4771
+ * their typed-event stores isolated. The main Scout store is the only store
4772
+ * visible to Plan; shard facts are merged in completion-order-independent
4773
+ * order after both sessions settle. This gives discovery real parallelism
4774
+ * without allowing sibling models to race a shared revision counter.
4775
+ */
4776
+ function aggregateParallelPiResults(results) {
4777
+ const first = results[0];
4778
+ const failed = results.find((result) => !result.ok);
4779
+ const representative = failed ?? first;
4780
+ return {
4781
+ ...representative,
4782
+ ok: failed === undefined,
4783
+ failureCategory: failed?.failureCategory ?? "success",
4784
+ durationMs: Math.max(...results.map((result) => result.durationMs), 0),
4785
+ exitCode: failed ? failed.exitCode : 0,
4786
+ stderr: results.map((result) => result.stderr).filter(Boolean).join("\n"),
4787
+ tokensUsed: results.reduce((total, result) => total + result.tokensUsed, 0),
4788
+ parsedEvents: results.reduce((total, result) => total + result.parsedEvents, 0),
4789
+ attemptedModels: [
4790
+ ...new Set(results.flatMap((result) => result.attemptedModels)),
4791
+ ],
4792
+ fallbackUsed: results.some((result) => result.fallbackUsed),
4793
+ timedOut: results.some((result) => result.timedOut),
4794
+ };
4795
+ }
4796
+ function combineSequentialPiResults(first, second) {
4797
+ return {
4798
+ ...second,
4799
+ durationMs: first.durationMs + second.durationMs,
4800
+ stderr: [first.stderr, second.stderr].filter(Boolean).join("\n"),
4801
+ tokensUsed: first.tokensUsed + second.tokensUsed,
4802
+ parsedEvents: first.parsedEvents + second.parsedEvents,
4803
+ attemptedModels: [
4804
+ ...new Set([...first.attemptedModels, ...second.attemptedModels]),
4805
+ ],
4806
+ fallbackUsed: first.fallbackUsed || second.fallbackUsed,
4807
+ timedOut: first.timedOut || second.timedOut,
4808
+ };
4809
+ }
4810
+ async function runFrontendScoutParallelSessions(input) {
4811
+ const [{ createTypedEventStore }] = await Promise.all([
4812
+ import("../workflows/dag/frontend-typed-event-store.js"),
4813
+ ]);
4814
+ const shards = [
4815
+ {
4816
+ id: "surface",
4817
+ toolName: "record_target_surface",
4818
+ instruction: [
4819
+ "PARALLEL SCOUT SHARD — target surface only.",
4820
+ "Inspect routes, entrypoints, implementation ownership, data source, and applicable test paths.",
4821
+ "Call record_target_surface exactly once with the complete runtime-evidenced surface. Do not call record_design_evidence.",
4822
+ ].join(" "),
4823
+ },
4824
+ {
4825
+ id: "design",
4826
+ toolName: "record_design_evidence",
4827
+ instruction: [
4828
+ "PARALLEL SCOUT SHARD — design evidence only.",
4829
+ "Inspect the frontend framework, styling/theme conventions, reusable components, and relevant design/spec files.",
4830
+ "Call record_design_evidence for the evidence you actually read. Do not call record_target_surface.",
4831
+ ].join(" "),
4832
+ },
4833
+ ];
4834
+ const outcomes = await Promise.all(shards.map(async (shard) => {
4835
+ let shardTools;
4836
+ let shardResult;
4837
+ try {
4838
+ const store = createTypedEventStore();
4839
+ const shardNodeId = `${input.nodeId}/parallel/${shard.id}`;
4840
+ shardTools = await createFrontendScoutEvidenceTools({
4841
+ attemptId: `${input.attemptId}:parallel:${shard.id}`,
4842
+ store,
4843
+ runDir: input.runDir,
4844
+ nodeId: shardNodeId,
4845
+ workspaceRoot: input.workspaceRoot,
4846
+ sourceDeclaredPaths: input.sourceDeclaredPaths,
4847
+ });
4848
+ const customTools = shardTools.customTools.filter((tool) => typeof tool === "object" &&
4849
+ tool !== null &&
4850
+ tool.name === shard.toolName);
4851
+ const sessionOptions = {
4852
+ ...input.sessionOptions,
4853
+ sessionEventsPath: path.join(input.runDir, shardNodeId, "session-events.jsonl"),
4854
+ };
4855
+ shardResult = await input.piStepFn({
4856
+ ...sessionOptions,
4857
+ prompt: `${input.basePrompt}\n\n${shard.instruction}`,
4858
+ writerToolPolicy: {
4859
+ requireSdk: true,
4860
+ customTools: [...customTools, ...(input.readBudgetTools ?? [])],
4861
+ },
4862
+ });
4863
+ await shardTools.flush();
4864
+ return { shard, result: shardResult, tools: shardTools };
4865
+ }
4866
+ catch (error) {
4867
+ const crashMessage = `frontend scout parallel shard ${shard.id} crashed: ${error instanceof Error ? error.message : String(error)}`;
4868
+ return {
4869
+ shard,
4870
+ tools: shardTools,
4871
+ result: shardResult
4872
+ ? {
4873
+ ...shardResult,
4874
+ ok: false,
4875
+ failureCategory: shardResult.ok
4876
+ ? "invalid-output"
4877
+ : shardResult.failureCategory,
4878
+ stderr: [shardResult.stderr, crashMessage]
4879
+ .filter(Boolean)
4880
+ .join("\n"),
4881
+ }
4882
+ : {
4883
+ ok: false,
4884
+ assistantText: "",
4885
+ command: [],
4886
+ durationMs: 0,
4887
+ exitCode: null,
4888
+ failureCategory: "tool-policy",
4889
+ modelDisplay: "unknown",
4890
+ parsedEvents: 0,
4891
+ stderr: crashMessage,
4892
+ stdout: "",
4893
+ timedOut: false,
4894
+ attemptedModels: [],
4895
+ fallbackUsed: false,
4896
+ tokensUsed: 0,
4897
+ },
4898
+ };
4899
+ }
4900
+ }));
4901
+ const ordered = [...outcomes].sort((left, right) => left.shard.id.localeCompare(right.shard.id));
4902
+ try {
4903
+ for (const outcome of ordered) {
4904
+ if (outcome.result.ok && outcome.tools) {
4905
+ await input.mainTools.adoptCommittedFacts(outcome.tools.committedFacts());
4906
+ }
4907
+ }
4908
+ }
4909
+ catch (error) {
4910
+ const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4911
+ return {
4912
+ ...aggregate,
4913
+ ok: false,
4914
+ failureCategory: "invalid-output",
4915
+ stderr: [
4916
+ aggregate.stderr,
4917
+ `frontend scout parallel fact merge failed: ${error instanceof Error ? error.message : String(error)}`,
4918
+ ]
4919
+ .filter(Boolean)
4920
+ .join("\n"),
4921
+ };
4922
+ }
4923
+ const failed = ordered.find((outcome) => !outcome.result.ok);
4924
+ if (failed) {
4925
+ const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4926
+ return {
4927
+ ...aggregate,
4928
+ ok: false,
4929
+ stderr: `${aggregate.stderr}\nfrontend scout parallel shard failed: ${failed.shard.id}`.trim(),
4930
+ };
4931
+ }
4932
+ const committedKinds = new Set(ordered.flatMap((outcome) => (outcome.tools?.committedFacts() ?? [])
4933
+ .map((record) => record.fact?.kind)
4934
+ .filter((kind) => typeof kind === "string")));
4935
+ const missing = ["target-surface", "design-evidence"].filter((kind) => !committedKinds.has(kind));
4936
+ if (missing.length > 0) {
4937
+ const fallback = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4938
+ return {
4939
+ ...fallback,
4940
+ ok: false,
4941
+ failureCategory: "invalid-output",
4942
+ stderr: `${fallback.stderr}\nfrontend scout parallel shards committed no ${missing.join(" or ")} fact`.trim(),
4943
+ };
4944
+ }
4945
+ return aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
4946
+ }
4368
4947
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
4369
4948
  const started = Date.now();
4370
4949
  const persona = resolveDagPiPersona(input.task);
@@ -4459,6 +5038,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4459
5038
  let planLedgerTools;
4460
5039
  let contractTools;
4461
5040
  let scoutEvidenceTools;
5041
+ let scoutSourceDeclaredPaths;
4462
5042
  let readBudgetTools;
4463
5043
  const commandPolicy = resolveDagCommandPolicy(input.task.commandPolicy);
4464
5044
  const allowsPlaywrightCli = dagCommandPolicyAllows(input.task.commandPolicy, "playwright-cli");
@@ -4626,16 +5206,17 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4626
5206
  try {
4627
5207
  const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
4628
5208
  const store = createTypedEventStore();
5209
+ scoutSourceDeclaredPaths = await resolveFrontendScoutSourceDeclaredPaths({
5210
+ cwd: input.cwd,
5211
+ spec: meta.spec,
5212
+ });
4629
5213
  scoutEvidenceTools = await createFrontendScoutEvidenceTools({
4630
5214
  attemptId: `${meta.runId}:${input.task.id}`,
4631
5215
  store,
4632
5216
  runDir: meta.runDir,
4633
5217
  nodeId: input.task.id,
4634
5218
  workspaceRoot: input.cwd,
4635
- sourceDeclaredPaths: await resolveFrontendScoutSourceDeclaredPaths({
4636
- cwd: input.cwd,
4637
- spec: meta.spec,
4638
- }),
5219
+ sourceDeclaredPaths: scoutSourceDeclaredPaths,
4639
5220
  });
4640
5221
  writerToolPolicy = {
4641
5222
  requireSdk: true,
@@ -4671,6 +5252,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4671
5252
  declaredUiStateIds: await resolveFrontendDeclaredUiStateIds({
4672
5253
  runDir: meta.runDir,
4673
5254
  }),
5255
+ canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
5256
+ runDir: meta.runDir,
5257
+ }),
4674
5258
  workspaceRoot: input.cwd,
4675
5259
  });
4676
5260
  writerToolPolicy = {
@@ -4811,11 +5395,27 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4811
5395
  }
4812
5396
  : undefined,
4813
5397
  };
4814
- if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
4815
- // Frontend-only split: sequential sessions with independent
4816
- // output budgets (coverage -> UX decisions -> finalize), mirroring the
4817
- // backend-test template's module sharding. Every other template keeps
4818
- // the single-session path below.
5398
+ if (isFrontendScoutEvidenceNode(input.task) &&
5399
+ scoutEvidenceTools &&
5400
+ input.task.complexity !== "LOW") {
5401
+ result = await runFrontendScoutParallelSessions({
5402
+ piStepFn,
5403
+ sessionOptions: piSessionOptions,
5404
+ basePrompt: input.prompt,
5405
+ mainTools: scoutEvidenceTools,
5406
+ runDir: meta.runDir,
5407
+ nodeId: input.task.id,
5408
+ attemptId: `${meta.runId}:${input.task.id}`,
5409
+ workspaceRoot: input.cwd,
5410
+ sourceDeclaredPaths: scoutSourceDeclaredPaths,
5411
+ readBudgetTools: readBudgetTools?.customTools,
5412
+ });
5413
+ }
5414
+ else if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
5415
+ // Frontend-only split: independent coverage map sessions feed a single
5416
+ // reducer (UX decisions -> global policy -> finalize), mirroring the
5417
+ // backend-test template's module sharding. Small plans normally have one
5418
+ // coverage batch and retain the compact path below.
4819
5419
  let planRequirementIds = [];
4820
5420
  const planRequirementCosts = new Map();
4821
5421
  const behaviorRequiredRequirementIds = [];
@@ -4842,26 +5442,189 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4842
5442
  catch {
4843
5443
  // Unreadable ledger falls back to a single coverage session.
4844
5444
  }
4845
- result = await runFrontendPlanSegmentedSessions({
5445
+ // Independent requirement-coverage batches are map workers. Each worker
5446
+ // owns an isolated typed-event store; only after all workers settle do we
5447
+ // merge facts into the main Plan ledger and run the single UX/global/
5448
+ // finalize reducer. This avoids revision races while shortening the
5449
+ // longest coverage phase for large plans.
5450
+ const runPlanSessions = (options) => runFrontendPlanSegmentedSessions({
4846
5451
  piStepFn,
4847
- sessionOptions: piSessionOptions,
4848
- basePrompt: input.prompt,
5452
+ sessionOptions: options.sessionOptions,
5453
+ basePrompt: options.basePrompt ?? input.prompt,
4849
5454
  attempt: input.attempt ?? 1,
4850
- committedFactCount: () => planLedgerTools.committedFactCount(),
4851
- requirementIds: planRequirementIds,
5455
+ committedFactCount: () => options.ledgerTools.committedFactCount(),
5456
+ requirementIds: options.requirementIds ?? planRequirementIds,
4852
5457
  requirementCosts: planRequirementCosts,
4853
- compactSmallPlan: true,
4854
- committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
4855
- committedFacts: () => planLedgerTools.committedFacts(),
5458
+ ...(options.parallelCoverageOnly !== undefined
5459
+ ? { parallelCoverageOnly: options.parallelCoverageOnly }
5460
+ : {}),
5461
+ ...(options.compactSmallPlan !== undefined
5462
+ ? { compactSmallPlan: options.compactSmallPlan }
5463
+ : {}),
5464
+ committedRequirementIds: () => options.ledgerTools.committedRequirementIds(),
5465
+ committedFacts: () => options.ledgerTools.committedFacts(),
4856
5466
  behaviorRequiredRequirementIds,
4857
- setActiveRequirementScope: (requirementIds) => planLedgerTools.setActiveRequirementScope(requirementIds),
5467
+ setActiveRequirementScope: (requirementIds) => options.ledgerTools.setActiveRequirementScope(requirementIds),
4858
5468
  segmentCustomTools: (toolNames) => toolNames === null
4859
- ? planLedgerTools.customTools
4860
- : planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
5469
+ ? options.ledgerTools.customTools
5470
+ : options.ledgerTools.customTools.filter((tool) => typeof tool === "object" &&
4861
5471
  tool !== null &&
4862
5472
  toolNames.has(tool.name)),
4863
- flushLedger: () => planLedgerTools.flush(),
5473
+ flushLedger: () => options.ledgerTools.flush(),
4864
5474
  });
5475
+ if (planRequirementIds.length > 1) {
5476
+ const estimatedPlanCalls = planRequirementIds.reduce((total, id) => total + Math.max(1, planRequirementCosts.get(id) ?? 2), 0);
5477
+ const compactEligibleBeforeSharding = planRequirementIds.length <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
5478
+ estimatedPlanCalls > 12 &&
5479
+ estimatedPlanCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
5480
+ countFrontendPlanTargetSurfaces(input.prompt) === 1;
5481
+ if (compactEligibleBeforeSharding) {
5482
+ // Decide the small-plan topology before creating coverage shards.
5483
+ // Sharding first would make the compact two-session path unreachable.
5484
+ result = await runPlanSessions({
5485
+ ledgerTools: planLedgerTools,
5486
+ sessionOptions: piSessionOptions,
5487
+ compactSmallPlan: true,
5488
+ });
5489
+ }
5490
+ else {
5491
+ const coverageBatches = batchFrontendPlanRequirements({
5492
+ requirementIds: planRequirementIds,
5493
+ maxEstimatedRecordCalls: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS,
5494
+ maxRequirements: 4,
5495
+ requirementCosts: planRequirementCosts,
5496
+ });
5497
+ if (coverageBatches.length > 1) {
5498
+ const shardResults = await mapWithConcurrency(coverageBatches, FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY, async (slice, index) => {
5499
+ let shardTools;
5500
+ let shardResult;
5501
+ try {
5502
+ const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
5503
+ shardTools = await createFrontendPlanLedgerTools({
5504
+ attemptId: `${meta.runId}:${input.task.id}:parallel:${index + 1}`,
5505
+ store: createTypedEventStore(),
5506
+ runDir: meta.runDir,
5507
+ nodeId: `${input.task.id}/parallel/coverage-${index + 1}`,
5508
+ skeleton: input.task.structuredContractOutput?.skeleton,
5509
+ sourceBinding: meta.spec.sourceBinding,
5510
+ requirementIds: slice,
5511
+ writeSetPatterns: input.task.writeSet,
5512
+ canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
5513
+ runDir: meta.runDir,
5514
+ }),
5515
+ componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
5516
+ cwd: input.cwd,
5517
+ sourceBinding: meta.spec.sourceBinding,
5518
+ }),
5519
+ });
5520
+ const shardSessionOptions = {
5521
+ ...piSessionOptions,
5522
+ sessionEventsPath: path.join(meta.runDir, input.task.id, "parallel", `coverage-${index + 1}`, "session-events.jsonl"),
5523
+ };
5524
+ shardResult = await runPlanSessions({
5525
+ ledgerTools: shardTools,
5526
+ sessionOptions: shardSessionOptions,
5527
+ requirementIds: slice,
5528
+ basePrompt: `${input.prompt}\n\nPARALLEL COVERAGE SHARD ${index + 1}: use the frozen canonical behavior verification-target ids listed in the record_plan_verification_target tool description — do not prefix ids with a shard namespace or invent variant ids; identical cross-shard targets are deduped, divergent ones fail the merge.`,
5529
+ parallelCoverageOnly: true,
5530
+ });
5531
+ await shardTools.flush();
5532
+ return { index, result: shardResult, tools: shardTools };
5533
+ }
5534
+ catch (error) {
5535
+ const crashMessage = `frontend plan coverage shard ${index + 1} crashed: ${error instanceof Error ? error.message : String(error)}`;
5536
+ return {
5537
+ index,
5538
+ tools: shardTools,
5539
+ result: shardResult
5540
+ ? {
5541
+ ...shardResult,
5542
+ ok: false,
5543
+ failureCategory: shardResult.ok
5544
+ ? "invalid-output"
5545
+ : shardResult.failureCategory,
5546
+ stderr: [shardResult.stderr, crashMessage]
5547
+ .filter(Boolean)
5548
+ .join("\n"),
5549
+ }
5550
+ : {
5551
+ ok: false,
5552
+ assistantText: "",
5553
+ command: [],
5554
+ durationMs: 0,
5555
+ exitCode: null,
5556
+ failureCategory: "tool-policy",
5557
+ modelDisplay: "unknown",
5558
+ parsedEvents: 0,
5559
+ stderr: crashMessage,
5560
+ stdout: "",
5561
+ timedOut: false,
5562
+ attemptedModels: [],
5563
+ fallbackUsed: false,
5564
+ tokensUsed: 0,
5565
+ },
5566
+ };
5567
+ }
5568
+ });
5569
+ const coverageResult = aggregateParallelPiResults(shardResults.map((shard) => shard.result));
5570
+ let mergeFailure;
5571
+ try {
5572
+ // Flatten every shard's facts FIRST, then pre-reduce: shards
5573
+ // partition requirements but a frozen verification target can
5574
+ // span shards, so same-id targets must merge across shards
5575
+ // before adoption, not per shard.
5576
+ await planLedgerTools.adoptCommittedFacts(reduceParallelCoverageShardRecords(shardResults
5577
+ .filter((item) => item.result.ok && item.tools)
5578
+ .flatMap((shard) => normalizeParallelCoverageShardRecords(shard.tools.committedFacts(), shard.index + 1))));
5579
+ }
5580
+ catch (error) {
5581
+ mergeFailure = error;
5582
+ }
5583
+ const failedShard = shardResults.find((shard) => !shard.result.ok);
5584
+ if (mergeFailure) {
5585
+ result = {
5586
+ ...coverageResult,
5587
+ ok: false,
5588
+ failureCategory: "invalid-output",
5589
+ stderr: [
5590
+ coverageResult.stderr,
5591
+ `frontend plan coverage shard merge failed: ${mergeFailure instanceof Error ? mergeFailure.message : String(mergeFailure)}`,
5592
+ ]
5593
+ .filter(Boolean)
5594
+ .join("\n"),
5595
+ };
5596
+ }
5597
+ else if (failedShard) {
5598
+ result = {
5599
+ ...coverageResult,
5600
+ ok: false,
5601
+ stderr: `${coverageResult.stderr}\nfrontend plan coverage shard ${failedShard.index + 1} failed before reduce`.trim(),
5602
+ };
5603
+ }
5604
+ else {
5605
+ const reducerResult = await runPlanSessions({
5606
+ ledgerTools: planLedgerTools,
5607
+ sessionOptions: piSessionOptions,
5608
+ });
5609
+ result = combineSequentialPiResults(coverageResult, reducerResult);
5610
+ }
5611
+ }
5612
+ else {
5613
+ result = await runPlanSessions({
5614
+ ledgerTools: planLedgerTools,
5615
+ sessionOptions: piSessionOptions,
5616
+ compactSmallPlan: true,
5617
+ });
5618
+ }
5619
+ }
5620
+ }
5621
+ else {
5622
+ result = await runPlanSessions({
5623
+ ledgerTools: planLedgerTools,
5624
+ sessionOptions: piSessionOptions,
5625
+ compactSmallPlan: true,
5626
+ });
5627
+ }
4865
5628
  }
4866
5629
  else {
4867
5630
  result = await piStepFn({
@@ -5408,7 +6171,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
5408
6171
  runDir: meta.runDir,
5409
6172
  progress,
5410
6173
  attempt: input.attempt ?? 1,
5411
- maxAttempts: input.task.retryPolicy?.maxAttempts ?? 3,
6174
+ maxAttempts: input.task.retryPolicy?.maxAttempts ?? 5,
5412
6175
  });
5413
6176
  if (progress.status !== "PASS") {
5414
6177
  const classified = classifyBackendTestWriterCompletenessFailure(progress);
@@ -5458,7 +6221,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
5458
6221
  stderrParts.push(completenessFailure.detail);
5459
6222
  }
5460
6223
  if (writerThinkingExhausted) {
5461
- stderrParts.push(`${WRITER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 write tool calls, 0 attributed diff; recommend a model switch and a new run`);
6224
+ stderrParts.push(`${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before any write tool call; retry from the existing workspace and complete only unfinished targets`);
5462
6225
  }
5463
6226
  if (meta.writeGuardAttribution === "best-effort") {
5464
6227
  stderrParts.push("write guard note: concurrent rank writers use best-effort per-node attribution; keep same-rank writeSet entries disjoint");
@@ -5531,14 +6294,11 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
5531
6294
  rawFailureCategory === "context-overflow" &&
5532
6295
  (changeManifestChangedFiles?.length ?? 0) > 0
5533
6296
  ? "partial-success-with-context-overflow"
5534
- : // writer-thinking-exhausted: a length-stopped thinking-only attempt with
5535
- // zero write tool calls and zero attributed diff is terminal and
5536
- // non-retryable; recommend a model switch + fresh run. Only applies on
5537
- // the empty-output base category so provider/transport failures keep
5538
- // their original category, and only when the completeness gate did not
5539
- // upgrade to incomplete-write-set above.
6297
+ : // output-limit: a length-stopped attempt is capacity truncation, not
6298
+ // empty output. Preserve the raw category for diagnostics and let the
6299
+ // node retry from committed facts/current workspace state.
5540
6300
  writerThinkingExhausted
5541
- ? WRITER_THINKING_EXHAUSTED_CATEGORY
6301
+ ? OUTPUT_LIMIT_RETRY_CATEGORY
5542
6302
  : writerCleanTimeout
5543
6303
  ? WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY
5544
6304
  : // writer-budget-exhausted: the provider session consumed an