@tea-agent/loop-agent 0.42.0-next.15 → 0.42.0-next.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/application/task-lifecycle/advance.js +9 -4
- package/dist/build-stamp.json +3 -3
- package/dist/commands/task-source-prepare.js +3 -1
- package/dist/executors/dag-pi-executor.js +825 -65
- package/dist/executors/shell-executor.js +103 -35
- package/dist/shared/dag-failure-category.js +6 -0
- package/dist/task/contract/apply.js +36 -2
- package/dist/task/source-prepare/parse-intent.js +7 -0
- package/dist/workflows/dag/dag-retry-schema.js +3 -0
- package/dist/workflows/dag/frontend-design-policy.js +1 -1
- package/dist/workflows/dag/frontend-risk.js +2 -0
- package/dist/workflows/dag/frontend-shadow-dual-write.js +16 -1
- package/dist/workflows/dag/frontend-shape.js +16 -6
- package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
- package/dist/workflows/dag/frontend-verification-trace.js +32 -0
- package/dist/workflows/dag/init-hybrid.js +9 -10
- package/dist/workflows/dag/node-execution.js +124 -72
- package/dist/workflows/dag/rerun-feedback.js +1 -0
- package/dist/workflows/dag/rerun-plan.js +90 -4
- package/dist/workflows/dag/rerun-run.js +7 -0
- package/dist/workflows/dag/retry-policy.js +16 -10
- package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
- package/dist/workflows/dag/runner.js +67 -7
- package/dist/workflows/dag/structured-output-repair.js +4 -1
- package/dist/workflows/dag/validate.js +10 -8
- package/dist/workflows/dag/workspace-checkpoint.js +66 -0
- package/docs/templates/agent-dag.schema.json +2 -2
- package/package.json +1 -1
- package/skills/frontend-plan/SKILL.md +6 -1
- package/skills/frontend-plan/references/decision-contract.md +69 -18
- package/skills/frontend-plan/references/design-decisions.md +32 -0
|
@@ -17,9 +17,9 @@ import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation
|
|
|
17
17
|
import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
|
|
18
18
|
import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
|
|
19
19
|
import { pathMatchesPattern } from "../shared/git-progress.js";
|
|
20
|
-
import { readTypedEventStoreFromJsonl } from "../workflows/dag/frontend-typed-event-store.js";
|
|
20
|
+
import { readTypedEventStoreFromJsonl, typedEventPayloadSha256, } from "../workflows/dag/frontend-typed-event-store.js";
|
|
21
21
|
import { collectCanonicalStateFlowNames, frontendEvidenceExpectationSchema, resolveFrontendContractRequirements, validateFrontendRequiredDeliverables, } from "../workflows/dag/frontend-contract-facts.js";
|
|
22
|
-
import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
|
|
22
|
+
import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
|
|
23
23
|
import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestCollectionRepairOutcomeRecoveryCandidate, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
|
|
24
24
|
import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
|
|
25
25
|
import { assessBackendTestPlanProtocol } from "../workflows/dag/backend-test-plan-protocol.js";
|
|
@@ -27,23 +27,221 @@ import { frontendTestLayoutFromSpec } from "../workflows/dag/frontend-test-layou
|
|
|
27
27
|
import { redactSecrets, truncateUtf8Preview } from "../shared/preview.js";
|
|
28
28
|
import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.js";
|
|
29
29
|
/**
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
* and produced zero attributed diff. This is a terminal, non-retryable
|
|
33
|
-
* diagnosis (the same prompt + model will hit the same budget wall); the
|
|
34
|
-
* recommendation is a model switch plus a fresh run. It must NOT mask a
|
|
35
|
-
* recoverable partial-write-set (incomplete-write-set) upgrade.
|
|
30
|
+
* Legacy diagnostic label retained for artifact compatibility. New executions
|
|
31
|
+
* classify this signal as output-limit so the node can retry incrementally.
|
|
36
32
|
*/
|
|
37
33
|
export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
|
|
38
34
|
/**
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
* typed facts with no assistant text. By the time this survives the segmented
|
|
42
|
-
* ladder the scope has already been degraded, so the durable fix is a
|
|
43
|
-
* thinking-capped or non-thinking model for the tier — not another replay of
|
|
44
|
-
* the same full-scope prompt.
|
|
35
|
+
* Legacy diagnostic label retained for artifact compatibility. New executions
|
|
36
|
+
* classify this signal as output-limit and preserve committed typed facts.
|
|
45
37
|
*/
|
|
46
38
|
export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
|
|
39
|
+
/**
|
|
40
|
+
* Parallel coverage shards namespace their verification target ids with
|
|
41
|
+
* `VT-SHARD-<shard>-` so the reducer can detect cross-shard conflicts. Once
|
|
42
|
+
* every shard's facts are known the namespace must be restored: the canonical
|
|
43
|
+
* contract has to carry the exact ids the task source froze (a surviving
|
|
44
|
+
* `VT-SHARD-N-…` id is a contract-requirement gap that design review
|
|
45
|
+
* rejects). Stripping happens per shard before adoption — after the strip,
|
|
46
|
+
* the existing identity-conflict check catches genuine cross-shard
|
|
47
|
+
* duplicate base ids and fails the merge closed.
|
|
48
|
+
*/
|
|
49
|
+
/**
|
|
50
|
+
* Deterministic pre-reduce for parallel coverage shard facts. Shards partition
|
|
51
|
+
* requirements, but a frozen verification target can legitimately be derived
|
|
52
|
+
* by several shards (one VT covers multiple ACs across shard boundaries), so
|
|
53
|
+
* same-id verification targets are MERGED: requirementIds and uiStates union,
|
|
54
|
+
* while divergent file/commandId is a real conflict. Everything else passes
|
|
55
|
+
* through unchanged.
|
|
56
|
+
*/
|
|
57
|
+
export function reduceParallelCoverageShardRecords(records) {
|
|
58
|
+
// Resolve replacements inside each source shard before comparing shards.
|
|
59
|
+
// A correction is an event-log operation, not a second independent target;
|
|
60
|
+
// retaining both records makes a valid same-shard correction look like a
|
|
61
|
+
// cross-shard conflict during the later reduction.
|
|
62
|
+
const byShard = new Map();
|
|
63
|
+
for (const record of records) {
|
|
64
|
+
const shard = byShard.get(record.attemptId) ?? [];
|
|
65
|
+
shard.push(record);
|
|
66
|
+
byShard.set(record.attemptId, shard);
|
|
67
|
+
}
|
|
68
|
+
const effectiveRecords = [];
|
|
69
|
+
for (const shardRecords of byShard.values()) {
|
|
70
|
+
const effective = [];
|
|
71
|
+
for (const record of shardRecords) {
|
|
72
|
+
const fact = record.fact;
|
|
73
|
+
const entry = fact?.entry;
|
|
74
|
+
const kind = typeof fact?.kind === "string" ? fact.kind : "";
|
|
75
|
+
const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
|
|
76
|
+
typeof entry?.id === "string"
|
|
77
|
+
? `${kind}:${entry.id}`
|
|
78
|
+
: undefined;
|
|
79
|
+
if (!identity) {
|
|
80
|
+
effective.push(record);
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
const replaces = typeof fact?.replaces === "string" ? `${kind}:${fact.replaces}` : undefined;
|
|
84
|
+
if (replaces) {
|
|
85
|
+
for (let index = effective.length - 1; index >= 0; index -= 1) {
|
|
86
|
+
const prior = effective[index];
|
|
87
|
+
const priorFact = prior.fact;
|
|
88
|
+
const priorEntry = priorFact?.entry;
|
|
89
|
+
const priorIdentity = typeof priorFact?.kind === "string" &&
|
|
90
|
+
typeof priorEntry?.id === "string"
|
|
91
|
+
? `${priorFact.kind}:${priorEntry.id}`
|
|
92
|
+
: undefined;
|
|
93
|
+
if (priorIdentity === replaces)
|
|
94
|
+
effective.splice(index, 1);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
// Keep the source record immutable. The replacement is represented by
|
|
98
|
+
// the latest event and its own payload hash.
|
|
99
|
+
effective.push(record);
|
|
100
|
+
}
|
|
101
|
+
effectiveRecords.push(...effective);
|
|
102
|
+
}
|
|
103
|
+
const verificationTargets = new Map();
|
|
104
|
+
const merged = [];
|
|
105
|
+
for (const record of effectiveRecords) {
|
|
106
|
+
const fact = record.fact;
|
|
107
|
+
if (!fact || typeof fact !== "object")
|
|
108
|
+
continue;
|
|
109
|
+
if (fact.kind !== "plan-verification-target") {
|
|
110
|
+
merged.push(record);
|
|
111
|
+
continue;
|
|
112
|
+
}
|
|
113
|
+
const entry = fact.entry;
|
|
114
|
+
if (!entry || typeof entry.id !== "string")
|
|
115
|
+
continue;
|
|
116
|
+
const existing = verificationTargets.get(entry.id);
|
|
117
|
+
if (!existing) {
|
|
118
|
+
verificationTargets.set(entry.id, {
|
|
119
|
+
record, entry: { ...entry },
|
|
120
|
+
sources: [`${record.attemptId}:${record.eventId}:${record.payloadSha256}`],
|
|
121
|
+
});
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
// Never mutate the entry held by the source record. The reducer creates
|
|
125
|
+
// a new merged entry and recomputes the payload hash for that derived
|
|
126
|
+
// fact, leaving source-event replay integrity intact.
|
|
127
|
+
const baseEntry = { ...existing.entry };
|
|
128
|
+
for (const field of ["commandId", "commandLabel", "file", "scope"]) {
|
|
129
|
+
if (entry[field] !== baseEntry[field]) {
|
|
130
|
+
throw new Error(`frontend plan coverage shard reduce conflict: verification target ${entry.id} has divergent ${field} "${String(entry[field])}" vs "${String(baseEntry[field])}"`);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
const unionSorted = (a, b) => {
|
|
134
|
+
const left = Array.isArray(a) ? a : [];
|
|
135
|
+
const right = Array.isArray(b) ? b : [];
|
|
136
|
+
return [...new Set([...left, ...right])].sort();
|
|
137
|
+
};
|
|
138
|
+
baseEntry.requirementIds = unionSorted(baseEntry.requirementIds, entry.requirementIds);
|
|
139
|
+
baseEntry.uiStates = unionSorted(baseEntry.uiStates, entry.uiStates);
|
|
140
|
+
const mergedFact = { ...fact, entry: { ...baseEntry } };
|
|
141
|
+
existing.entry = baseEntry;
|
|
142
|
+
existing.sources.push(`${record.attemptId}:${record.eventId}:${record.payloadSha256}`);
|
|
143
|
+
existing.record = {
|
|
144
|
+
...existing.record,
|
|
145
|
+
fact: mergedFact,
|
|
146
|
+
payloadSha256: typedEventPayloadSha256(mergedFact),
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
merged.push(...[...verificationTargets.values()].map((item) => {
|
|
150
|
+
// The reducer owns every VT revision, including a one-shard partial
|
|
151
|
+
// result. Content-addressed revisions replay idempotently, while an
|
|
152
|
+
// expanded/replaced source set appends an explicit replacement rather
|
|
153
|
+
// than masquerading as a changed source event or deleting audit facts.
|
|
154
|
+
const fact = {
|
|
155
|
+
...item.record.fact,
|
|
156
|
+
replaces: item.entry.id,
|
|
157
|
+
entry: {
|
|
158
|
+
...item.entry,
|
|
159
|
+
requirementIds: [...new Set(planFactStringList(item.entry.requirementIds))].sort(),
|
|
160
|
+
uiStates: [...new Set(planFactStringList(item.entry.uiStates))].sort(),
|
|
161
|
+
},
|
|
162
|
+
};
|
|
163
|
+
const payloadSha256 = typedEventPayloadSha256(fact);
|
|
164
|
+
const revisionId = createHash("sha256")
|
|
165
|
+
.update(JSON.stringify({ sources: [...new Set(item.sources)].sort(), payloadSha256 }))
|
|
166
|
+
.digest("hex");
|
|
167
|
+
return {
|
|
168
|
+
...item.record,
|
|
169
|
+
attemptId: "frontend-plan-coverage-reducer",
|
|
170
|
+
eventId: `coverage-reduced:${revisionId}`,
|
|
171
|
+
requestId: `coverage-reduced:${revisionId}`,
|
|
172
|
+
fact, payloadSha256,
|
|
173
|
+
};
|
|
174
|
+
}));
|
|
175
|
+
return merged;
|
|
176
|
+
}
|
|
177
|
+
export function normalizeParallelCoverageShardRecords(records, shardNumber) {
|
|
178
|
+
const prefix = `VT-SHARD-${shardNumber}-`;
|
|
179
|
+
// Models sometimes drop the `VT-` stem when applying the shard namespace
|
|
180
|
+
// (observed: frozen `VT-X` became `VT-SHARD-2-X`), so restoring the
|
|
181
|
+
// canonical id requires re-adding the stem after the strip.
|
|
182
|
+
const stripId = (id) => {
|
|
183
|
+
if (typeof id !== "string" || !id.startsWith(prefix))
|
|
184
|
+
return id;
|
|
185
|
+
const base = id.slice(prefix.length);
|
|
186
|
+
return base.startsWith("VT-") ? base : `VT-${base}`;
|
|
187
|
+
};
|
|
188
|
+
const stripIdList = (ids) => Array.isArray(ids) ? ids.map((id) => stripId(id)) : ids;
|
|
189
|
+
return records.map((record) => {
|
|
190
|
+
const fact = record.fact;
|
|
191
|
+
if (!fact || typeof fact !== "object")
|
|
192
|
+
return record;
|
|
193
|
+
const kind = fact.kind;
|
|
194
|
+
const entry = fact.entry;
|
|
195
|
+
if (!entry || typeof entry !== "object")
|
|
196
|
+
return record;
|
|
197
|
+
let rewritten;
|
|
198
|
+
const replaces = kind === "plan-verification-target" ? stripId(fact.replaces) : fact.replaces;
|
|
199
|
+
if (kind === "plan-verification-target") {
|
|
200
|
+
const id = stripId(entry.id);
|
|
201
|
+
if (id !== entry.id || replaces !== fact.replaces)
|
|
202
|
+
rewritten = { ...entry, id };
|
|
203
|
+
}
|
|
204
|
+
else if (kind === "plan-requirement") {
|
|
205
|
+
const verificationTargetIds = stripIdList(entry.verificationTargetIds);
|
|
206
|
+
if (verificationTargetIds !== entry.verificationTargetIds) {
|
|
207
|
+
rewritten = { ...entry, verificationTargetIds };
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
else if (kind === "state-flow") {
|
|
211
|
+
const rewriteBoundTargets = (item) => {
|
|
212
|
+
if (!item || typeof item !== "object")
|
|
213
|
+
return item;
|
|
214
|
+
return {
|
|
215
|
+
...item,
|
|
216
|
+
verificationTargetIds: stripIdList(item.verificationTargetIds),
|
|
217
|
+
};
|
|
218
|
+
};
|
|
219
|
+
const uiStates = Array.isArray(entry.uiStates)
|
|
220
|
+
? entry.uiStates.map(rewriteBoundTargets)
|
|
221
|
+
: entry.uiStates;
|
|
222
|
+
const interactions = Array.isArray(entry.interactions)
|
|
223
|
+
? entry.interactions.map(rewriteBoundTargets)
|
|
224
|
+
: entry.interactions;
|
|
225
|
+
if (uiStates !== entry.uiStates || interactions !== entry.interactions) {
|
|
226
|
+
rewritten = { ...entry, uiStates, interactions };
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
return rewritten
|
|
230
|
+
? {
|
|
231
|
+
...record,
|
|
232
|
+
fact: { ...fact, ...(replaces !== undefined ? { replaces } : {}), entry: rewritten },
|
|
233
|
+
// Keep the integrity hash consistent with the rewritten
|
|
234
|
+
// payload, or the adoption replay check reports the
|
|
235
|
+
// normalized record as a tampered source event.
|
|
236
|
+
payloadSha256: typedEventPayloadSha256({
|
|
237
|
+
...fact,
|
|
238
|
+
...(replaces !== undefined ? { replaces } : {}),
|
|
239
|
+
entry: rewritten,
|
|
240
|
+
}),
|
|
241
|
+
}
|
|
242
|
+
: record;
|
|
243
|
+
});
|
|
244
|
+
}
|
|
47
245
|
export function isPlannerThinkingExhausted(result, committedAnyFacts) {
|
|
48
246
|
if (result.ok)
|
|
49
247
|
return false;
|
|
@@ -985,6 +1183,36 @@ async function resolveFrontendDeclaredUiStateIds(input) {
|
|
|
985
1183
|
return [];
|
|
986
1184
|
}
|
|
987
1185
|
}
|
|
1186
|
+
/**
|
|
1187
|
+
* Resolve the frozen canonical behavior verification-target ids the PRD
|
|
1188
|
+
* declares. The requirement text is the same authority the design review reads
|
|
1189
|
+
* when it rejects `VT-SHARD-N-…` / variant ids as a contract-requirement gap,
|
|
1190
|
+
* so extracting `VT-…` tokens from the frozen contract requirement facts makes
|
|
1191
|
+
* that authority deterministic instead of prose-only. An empty result (no PRD
|
|
1192
|
+
* declared any behavior target id) disables the canonical check and keeps the
|
|
1193
|
+
* historical free-form path.
|
|
1194
|
+
*/
|
|
1195
|
+
export async function resolveFrontendCanonicalVerificationTargetIds(input) {
|
|
1196
|
+
try {
|
|
1197
|
+
const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
|
|
1198
|
+
const ids = new Set();
|
|
1199
|
+
for (const record of records) {
|
|
1200
|
+
const fact = record.fact;
|
|
1201
|
+
if (record.phase !== "committed" || fact.kind !== "requirement") {
|
|
1202
|
+
continue;
|
|
1203
|
+
}
|
|
1204
|
+
const text = typeof fact.text === "string" ? fact.text : "";
|
|
1205
|
+
for (const match of text.matchAll(/\bVT-[A-Z0-9][A-Z0-9_-]*\b/g)) {
|
|
1206
|
+
ids.add(match[0]);
|
|
1207
|
+
}
|
|
1208
|
+
}
|
|
1209
|
+
return [...ids].sort();
|
|
1210
|
+
}
|
|
1211
|
+
catch {
|
|
1212
|
+
// No contract facts (or unreadable): fall back to no canonical set.
|
|
1213
|
+
return [];
|
|
1214
|
+
}
|
|
1215
|
+
}
|
|
988
1216
|
/**
|
|
989
1217
|
* A+B: `frontend-plan-pi` records its decision ledger through incremental
|
|
990
1218
|
* `record_*` tools (origin=plan) and closes with exactly one
|
|
@@ -1717,6 +1945,15 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1717
1945
|
});
|
|
1718
1946
|
}
|
|
1719
1947
|
}
|
|
1948
|
+
if (id &&
|
|
1949
|
+
activeRequirementScope.length > 0 &&
|
|
1950
|
+
!activeRequirementScope.includes(id)) {
|
|
1951
|
+
return planToolReceipt({
|
|
1952
|
+
ok: false,
|
|
1953
|
+
kind: "plan-requirement",
|
|
1954
|
+
error: `record_plan_requirement id "${id}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
|
|
1955
|
+
});
|
|
1956
|
+
}
|
|
1720
1957
|
const result = await adoptPlanFact("plan-requirement", `${attemptId}:record_plan_requirement:${randomUUID()}`, {
|
|
1721
1958
|
kind: "plan-requirement",
|
|
1722
1959
|
origin: "plan",
|
|
@@ -1729,7 +1966,11 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1729
1966
|
const recordPlanVerificationTargetTool = defineTool({
|
|
1730
1967
|
name: "record_plan_verification_target",
|
|
1731
1968
|
label: "record_plan_verification_target",
|
|
1732
|
-
description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}"
|
|
1969
|
+
description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}" +
|
|
1970
|
+
(input.canonicalVerificationTargetIds &&
|
|
1971
|
+
input.canonicalVerificationTargetIds.length > 0
|
|
1972
|
+
? ` Frozen canonical behavior target ids (use exactly for behavior targets): ${input.canonicalVerificationTargetIds.join(", ")}.`
|
|
1973
|
+
: ""),
|
|
1733
1974
|
promptSnippet: "Commit 1-4 plan verification target entries (up to 4 per message).",
|
|
1734
1975
|
parameters: Type.Object({
|
|
1735
1976
|
entry: verificationTargetSchema,
|
|
@@ -1780,6 +2021,24 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1780
2021
|
error: `record_plan_verification_target verification-target-phase-mismatch: behavior command "${directoryEntry.label}" (${directoryEntry.commandId}) must bind a test file (__tests__/, tests?/, e2e/, cypress/, *.test.*, *.spec.*, *.cy.*); received file "${rawEntry.file}"`,
|
|
1781
2022
|
});
|
|
1782
2023
|
}
|
|
2024
|
+
// Canonical-identity check for behavior targets: the PRD freezes
|
|
2025
|
+
// the exact behavior verification-target ids (e.g.
|
|
2026
|
+
// VT-SMOKE-COUNTER-BEHAVIOR). A committed non-canonical id is
|
|
2027
|
+
// immutable and design review rejects it as a
|
|
2028
|
+
// contract-requirement gap, so reject invented ids here.
|
|
2029
|
+
if (directoryEntry.mode === "behavior" &&
|
|
2030
|
+
input.canonicalVerificationTargetIds &&
|
|
2031
|
+
input.canonicalVerificationTargetIds.length > 0) {
|
|
2032
|
+
const canonicalTargetId = typeof rawEntry.id === "string" ? rawEntry.id.trim() : "";
|
|
2033
|
+
if (canonicalTargetId &&
|
|
2034
|
+
!input.canonicalVerificationTargetIds.includes(canonicalTargetId)) {
|
|
2035
|
+
return planToolReceipt({
|
|
2036
|
+
ok: false,
|
|
2037
|
+
kind: "plan-verification-target",
|
|
2038
|
+
error: `record_plan_verification_target id "${canonicalTargetId}" is not a frozen canonical behavior verification target; canonical ids are: ${input.canonicalVerificationTargetIds.join(", ")}`,
|
|
2039
|
+
});
|
|
2040
|
+
}
|
|
2041
|
+
}
|
|
1783
2042
|
}
|
|
1784
2043
|
// Duplicate-id rejection: committed typed facts are immutable, so
|
|
1785
2044
|
// re-recording the same VT id would deadlock the compile by default.
|
|
@@ -1849,6 +2108,15 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1849
2108
|
: [];
|
|
1850
2109
|
}));
|
|
1851
2110
|
const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
|
|
2111
|
+
const outOfScopeRequirementIds = stringList(rawEntry.requirementIds).filter((id) => activeRequirementScope.length > 0 &&
|
|
2112
|
+
!activeRequirementScope.includes(id));
|
|
2113
|
+
if (outOfScopeRequirementIds.length > 0) {
|
|
2114
|
+
return planToolReceipt({
|
|
2115
|
+
ok: false,
|
|
2116
|
+
kind: "plan-verification-target",
|
|
2117
|
+
error: `record_plan_verification_target references requirements outside this session's scope [${activeRequirementScope.join(", ")}]: ${outOfScopeRequirementIds.join(", ")}`,
|
|
2118
|
+
});
|
|
2119
|
+
}
|
|
1852
2120
|
if (unknownRequirementIds.length > 0) {
|
|
1853
2121
|
return planToolReceipt({
|
|
1854
2122
|
ok: false,
|
|
@@ -1895,6 +2163,18 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1895
2163
|
error: "record_plan_evidence_gap requires a non-empty description describing the gap",
|
|
1896
2164
|
});
|
|
1897
2165
|
}
|
|
2166
|
+
const requirementId = typeof entry.requirementId === "string"
|
|
2167
|
+
? entry.requirementId
|
|
2168
|
+
: undefined;
|
|
2169
|
+
if (requirementId &&
|
|
2170
|
+
activeRequirementScope.length > 0 &&
|
|
2171
|
+
!activeRequirementScope.includes(requirementId)) {
|
|
2172
|
+
return planToolReceipt({
|
|
2173
|
+
ok: false,
|
|
2174
|
+
kind: "plan-evidence-gap",
|
|
2175
|
+
error: `record_plan_evidence_gap requirementId "${requirementId}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
|
|
2176
|
+
});
|
|
2177
|
+
}
|
|
1898
2178
|
const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
|
|
1899
2179
|
return planToolReceipt(result);
|
|
1900
2180
|
},
|
|
@@ -2088,6 +2368,66 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
2088
2368
|
adoptStagedFactTool,
|
|
2089
2369
|
finalizePlanTool,
|
|
2090
2370
|
],
|
|
2371
|
+
adoptCommittedFacts: async (records) => {
|
|
2372
|
+
for (const record of records) {
|
|
2373
|
+
if (record.phase !== "committed")
|
|
2374
|
+
continue;
|
|
2375
|
+
const fact = record.fact;
|
|
2376
|
+
if (!fact || typeof fact !== "object" || Array.isArray(fact))
|
|
2377
|
+
continue;
|
|
2378
|
+
const kind = typeof fact.kind === "string"
|
|
2379
|
+
? fact.kind
|
|
2380
|
+
: "plan-fact";
|
|
2381
|
+
const entry = fact.entry;
|
|
2382
|
+
const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
|
|
2383
|
+
typeof entry?.id === "string"
|
|
2384
|
+
? `${kind}:${entry.id}`
|
|
2385
|
+
: undefined;
|
|
2386
|
+
const sourceShardPrefix = `${attemptId}:parallel-merge:${record.attemptId}:`;
|
|
2387
|
+
const requestId = `${sourceShardPrefix}${record.eventId}`;
|
|
2388
|
+
const committed = readCommittedEvents(store, attemptId);
|
|
2389
|
+
const replay = committed.find((candidate) => candidate.requestId === requestId);
|
|
2390
|
+
if (replay) {
|
|
2391
|
+
if (replay.payloadSha256 !== record.payloadSha256) {
|
|
2392
|
+
throw new Error(`frontend plan shard fact merge replay conflict: source event ${record.eventId} changed payload`);
|
|
2393
|
+
}
|
|
2394
|
+
continue;
|
|
2395
|
+
}
|
|
2396
|
+
const explicitReplacement = typeof fact.replaces === "string" &&
|
|
2397
|
+
fact.replaces === entry?.id;
|
|
2398
|
+
const conflicting = committed.find((candidate) => {
|
|
2399
|
+
const candidateFact = candidate.fact;
|
|
2400
|
+
const candidateEntry = candidateFact.entry;
|
|
2401
|
+
const sameSourceShard = candidate.requestId.startsWith(sourceShardPrefix);
|
|
2402
|
+
return (candidateFact.kind === kind &&
|
|
2403
|
+
typeof candidateEntry?.id === "string" &&
|
|
2404
|
+
`${kind}:${candidateEntry.id}` === identity &&
|
|
2405
|
+
!(sameSourceShard && explicitReplacement));
|
|
2406
|
+
});
|
|
2407
|
+
if (conflicting) {
|
|
2408
|
+
// Two shards may honestly emit the same fact (e.g. both
|
|
2409
|
+
// derive the same frozen verification target). Identical
|
|
2410
|
+
// payloads are duplicates to skip; divergent payloads are
|
|
2411
|
+
// a real conflict the ladder must resolve.
|
|
2412
|
+
if (conflicting.payloadSha256 === record.payloadSha256) {
|
|
2413
|
+
continue;
|
|
2414
|
+
}
|
|
2415
|
+
const fromSameSourceShard = conflicting.requestId.startsWith(sourceShardPrefix);
|
|
2416
|
+
if (fromSameSourceShard) {
|
|
2417
|
+
throw new Error(`frontend plan shard fact merge conflict: ${identity} was already committed by another shard with a different payload`);
|
|
2418
|
+
}
|
|
2419
|
+
// Divergent payload from a different source attempt means a
|
|
2420
|
+
// ladder retry re-derived the coverage phase: the fresh
|
|
2421
|
+
// derivation supersedes the stored record.
|
|
2422
|
+
const storeWithRecords = store;
|
|
2423
|
+
storeWithRecords.records = storeWithRecords.records.filter((candidate) => candidate.eventId !== conflicting.eventId);
|
|
2424
|
+
}
|
|
2425
|
+
const result = await adoptPlanFact(kind, requestId, fact);
|
|
2426
|
+
if (!result.ok) {
|
|
2427
|
+
throw new Error(`frontend plan shard fact merge failed: ${result.error}`);
|
|
2428
|
+
}
|
|
2429
|
+
}
|
|
2430
|
+
},
|
|
2091
2431
|
setActiveRequirementScope: (requirementIds) => {
|
|
2092
2432
|
activeRequirementScope = [
|
|
2093
2433
|
...new Set(requirementIds.filter((id) => id.trim().length > 0)),
|
|
@@ -2717,6 +3057,22 @@ export async function createFrontendScoutEvidenceTools(input) {
|
|
|
2717
3057
|
});
|
|
2718
3058
|
return {
|
|
2719
3059
|
customTools: [recordTargetSurfaceTool, recordDesignEvidenceTool],
|
|
3060
|
+
committedFacts: () => readCommittedEvents(store, attemptId),
|
|
3061
|
+
adoptCommittedFacts: async (records) => {
|
|
3062
|
+
for (const record of records) {
|
|
3063
|
+
if (record.phase !== "committed")
|
|
3064
|
+
continue;
|
|
3065
|
+
const fact = record.fact;
|
|
3066
|
+
if (!fact || typeof fact !== "object" || Array.isArray(fact))
|
|
3067
|
+
continue;
|
|
3068
|
+
const result = await adoptScoutFact(typeof fact.kind === "string"
|
|
3069
|
+
? fact.kind
|
|
3070
|
+
: "scout-fact", fact);
|
|
3071
|
+
if (!result.ok) {
|
|
3072
|
+
throw new Error(`frontend scout shard fact merge failed: ${result.error}`);
|
|
3073
|
+
}
|
|
3074
|
+
}
|
|
3075
|
+
},
|
|
2720
3076
|
flush: async () => {
|
|
2721
3077
|
const committed = readCommittedEvents(store, attemptId);
|
|
2722
3078
|
await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "scout-typed-facts.jsonl"), committed);
|
|
@@ -3109,6 +3465,7 @@ const FRONTEND_PLAN_SEGMENTS = [
|
|
|
3109
3465
|
instruction: [
|
|
3110
3466
|
"PLAN PHASE — requirement coverage only.",
|
|
3111
3467
|
"Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target facts (verification target bound to requirement ids and files). Group related requirements under one non-static behavior target when one observable test behavior proves them together; do not mechanically create one target per requirement. A non-static target id is the stable machine trace token; never submit prose as a symbol. Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
|
|
3468
|
+
"A coverage session is complete only when EVERY requirement assigned to this session (the full inventory, or the exact COVERAGE BATCH / shard list when present) has committed coverage facts: a record_plan_requirement entry plus verification targets, or a committed evidence gap. Keep committing in batches of up to 4 record_* calls per assistant message until then; do not write a concluding summary while any assigned requirement is still uncommitted — an early stop strands the remainder into a MISSING-FACT repair session and doubles the sessions needed.",
|
|
3112
3469
|
"If a requirement genuinely cannot have a verification target, record a non-empty record_plan_evidence_gap. Do not call finalize_plan; it is not available in this phase.",
|
|
3113
3470
|
].join(" "),
|
|
3114
3471
|
},
|
|
@@ -3126,11 +3483,12 @@ const FRONTEND_PLAN_SEGMENTS = [
|
|
|
3126
3483
|
toolNames: new Set([
|
|
3127
3484
|
"record_component_choice",
|
|
3128
3485
|
"record_state_flow",
|
|
3486
|
+
"record_plan_verification_target",
|
|
3129
3487
|
"adopt_staged_fact",
|
|
3130
3488
|
]),
|
|
3131
3489
|
instruction: [
|
|
3132
3490
|
"PLAN PHASE — global UX decisions.",
|
|
3133
|
-
"Requirements and verification targets are already committed in the ledger
|
|
3491
|
+
"Requirements and verification targets are already committed in the ledger. Review the complete requirement set and the committed global UX registry together, then record each component choice, UI state and interaction. Bind each applicable state to its verificationTargetIds; the runtime derives the reverse VT.uiStates relation. If a VT requires correction, record_plan_verification_target with replace:true is available after declaring its states; preserve its requirement coverage. Multiple requirements describing one behavior share one registry name and state-flow entry. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
|
|
3134
3492
|
"Do not call finalize_plan; it is not available in this phase.",
|
|
3135
3493
|
].join(" "),
|
|
3136
3494
|
},
|
|
@@ -3374,12 +3732,28 @@ export function batchFrontendPlanRequirements(input) {
|
|
|
3374
3732
|
return batches;
|
|
3375
3733
|
}
|
|
3376
3734
|
const FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS = 10;
|
|
3377
|
-
|
|
3378
|
-
//
|
|
3379
|
-
//
|
|
3735
|
+
const FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY = 4;
|
|
3736
|
+
// A large requirement set creates parallel coverage shards; UX remains one
|
|
3737
|
+
// global decision session so behavior names and component/state facts are not
|
|
3738
|
+
// reinvented per AC batch. Keep a safety bound for adaptive coverage retries
|
|
3739
|
+
// without letting the old 32-session ceiling skip finalize.
|
|
3380
3740
|
const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
|
|
3381
3741
|
const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
|
|
3382
3742
|
const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
|
|
3743
|
+
async function mapWithConcurrency(items, limit, worker) {
|
|
3744
|
+
const results = new Array(items.length);
|
|
3745
|
+
let nextIndex = 0;
|
|
3746
|
+
const workerCount = Math.min(Math.max(1, limit), items.length);
|
|
3747
|
+
await Promise.all(Array.from({ length: workerCount }, async () => {
|
|
3748
|
+
while (true) {
|
|
3749
|
+
const index = nextIndex++;
|
|
3750
|
+
if (index >= items.length)
|
|
3751
|
+
return;
|
|
3752
|
+
results[index] = await worker(items[index], index);
|
|
3753
|
+
}
|
|
3754
|
+
}));
|
|
3755
|
+
return results;
|
|
3756
|
+
}
|
|
3383
3757
|
function compactPromptString(value, maxChars) {
|
|
3384
3758
|
if (typeof value !== "string" || value.trim().length === 0)
|
|
3385
3759
|
return undefined;
|
|
@@ -3827,7 +4201,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3827
4201
|
const buildCoveragePrompt = (slice, missing = []) => [
|
|
3828
4202
|
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
|
|
3829
4203
|
coverageSegment.instruction,
|
|
3830
|
-
`COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
|
|
4204
|
+
`COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them. Finish the whole list before concluding: commit every listed requirement's coverage facts (verification targets or an evidence gap), batching up to 4 record_* calls per message; an early stop re-queues the remainder as a MISSING-FACT repair session.`,
|
|
3831
4205
|
...(missing.length > 0
|
|
3832
4206
|
? [
|
|
3833
4207
|
"MISSING-FACT QUEUE: the previous session did not establish complete coverage. Repair ONLY these items, then re-check the slice:",
|
|
@@ -3927,8 +4301,8 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3927
4301
|
const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
|
|
3928
4302
|
? {
|
|
3929
4303
|
...r,
|
|
3930
|
-
failureCategory:
|
|
3931
|
-
stderr: `${r.stderr}\n${
|
|
4304
|
+
failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
4305
|
+
stderr: `${r.stderr}\n${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before the next typed fact; preserve committed facts and retry only the unfinished phase`.trim(),
|
|
3932
4306
|
}
|
|
3933
4307
|
: r;
|
|
3934
4308
|
// An empty list means "ledger unreadable / unknown" and falls back to one
|
|
@@ -3943,6 +4317,29 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3943
4317
|
: [];
|
|
3944
4318
|
const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
|
|
3945
4319
|
const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
|
|
4320
|
+
if (input.parallelCoverageOnly === true &&
|
|
4321
|
+
requirementIdsProvided &&
|
|
4322
|
+
coverageWorkIds.length === 0) {
|
|
4323
|
+
// A retry may reopen a shard whose committed facts are already complete.
|
|
4324
|
+
// Treat that shard as an idempotent no-op; returning the normal empty
|
|
4325
|
+
// session failure would make a partially failed map impossible to resume.
|
|
4326
|
+
return {
|
|
4327
|
+
ok: true,
|
|
4328
|
+
assistantText: "",
|
|
4329
|
+
command: [],
|
|
4330
|
+
durationMs: 0,
|
|
4331
|
+
exitCode: 0,
|
|
4332
|
+
failureCategory: "success",
|
|
4333
|
+
modelDisplay: "reused-coverage-facts",
|
|
4334
|
+
parsedEvents: 0,
|
|
4335
|
+
stderr: "",
|
|
4336
|
+
stdout: "",
|
|
4337
|
+
timedOut: false,
|
|
4338
|
+
attemptedModels: [],
|
|
4339
|
+
fallbackUsed: false,
|
|
4340
|
+
tokensUsed: 0,
|
|
4341
|
+
};
|
|
4342
|
+
}
|
|
3946
4343
|
const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
|
|
3947
4344
|
const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
|
|
3948
4345
|
// Small, single-surface requests do not benefit from six isolated Pi
|
|
@@ -4004,7 +4401,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4004
4401
|
if (useCompactSmallPlan) {
|
|
4005
4402
|
// Compact mode already queued both sessions above.
|
|
4006
4403
|
}
|
|
4007
|
-
else {
|
|
4404
|
+
else if (!input.parallelCoverageOnly) {
|
|
4008
4405
|
if (requirementIdsProvided) {
|
|
4009
4406
|
queue.push({
|
|
4010
4407
|
id: uxRegistrySegment.id,
|
|
@@ -4041,7 +4438,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4041
4438
|
prompt: buildPhasePrompt(finalizeSegment),
|
|
4042
4439
|
});
|
|
4043
4440
|
}
|
|
4044
|
-
let
|
|
4441
|
+
let accumulated;
|
|
4045
4442
|
let index = 0;
|
|
4046
4443
|
let invocationCount = 0;
|
|
4047
4444
|
while (index < queue.length) {
|
|
@@ -4092,8 +4489,10 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4092
4489
|
}
|
|
4093
4490
|
}
|
|
4094
4491
|
const committedBefore = input.committedFactCount();
|
|
4095
|
-
input.setActiveRequirementScope?.(session.
|
|
4096
|
-
|
|
4492
|
+
input.setActiveRequirementScope?.(session.coverageOnly ||
|
|
4493
|
+
session.id === "compact-local" ||
|
|
4494
|
+
session.id.startsWith("ux-local-")
|
|
4495
|
+
? session.coverageSlice ?? session.requirementSlice ?? []
|
|
4097
4496
|
: []);
|
|
4098
4497
|
if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
|
|
4099
4498
|
break;
|
|
@@ -4111,7 +4510,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4111
4510
|
}
|
|
4112
4511
|
: {}),
|
|
4113
4512
|
});
|
|
4114
|
-
|
|
4513
|
+
accumulated = accumulated
|
|
4514
|
+
? combineSequentialPiResults(accumulated, result)
|
|
4515
|
+
: result;
|
|
4115
4516
|
try {
|
|
4116
4517
|
await input.flushLedger();
|
|
4117
4518
|
}
|
|
@@ -4220,9 +4621,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4220
4621
|
continue;
|
|
4221
4622
|
}
|
|
4222
4623
|
return {
|
|
4223
|
-
...
|
|
4624
|
+
...accumulated,
|
|
4224
4625
|
ok: false,
|
|
4225
|
-
stderr: `${
|
|
4626
|
+
stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
|
|
4226
4627
|
failureCategory: "invalid-output",
|
|
4227
4628
|
};
|
|
4228
4629
|
}
|
|
@@ -4239,9 +4640,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4239
4640
|
}
|
|
4240
4641
|
if (missingPhaseFacts.length > 0) {
|
|
4241
4642
|
return {
|
|
4242
|
-
...
|
|
4643
|
+
...accumulated,
|
|
4243
4644
|
ok: false,
|
|
4244
|
-
stderr:
|
|
4645
|
+
stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
|
|
4245
4646
|
failureCategory: "invalid-output",
|
|
4246
4647
|
};
|
|
4247
4648
|
}
|
|
@@ -4332,11 +4733,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4332
4733
|
};
|
|
4333
4734
|
continue;
|
|
4334
4735
|
}
|
|
4335
|
-
return mapPlannerExhaustion(
|
|
4736
|
+
return mapPlannerExhaustion(accumulated, committedAfter > committedBefore);
|
|
4336
4737
|
}
|
|
4337
4738
|
if (index < queue.length) {
|
|
4338
4739
|
return {
|
|
4339
|
-
...(
|
|
4740
|
+
...(accumulated ?? {
|
|
4340
4741
|
ok: false,
|
|
4341
4742
|
stdout: "",
|
|
4342
4743
|
stderr: "",
|
|
@@ -4353,11 +4754,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4353
4754
|
tokensUsed: 0,
|
|
4354
4755
|
}),
|
|
4355
4756
|
ok: false,
|
|
4356
|
-
stderr: `${
|
|
4757
|
+
stderr: `${accumulated?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
|
|
4357
4758
|
failureCategory: "invalid-output",
|
|
4358
4759
|
};
|
|
4359
4760
|
}
|
|
4360
|
-
return mapPlannerExhaustion(
|
|
4761
|
+
return mapPlannerExhaustion(accumulated ?? {
|
|
4361
4762
|
ok: false,
|
|
4362
4763
|
stdout: "",
|
|
4363
4764
|
stderr: "frontend plan segmentation produced no session",
|
|
@@ -4365,6 +4766,184 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
4365
4766
|
durationMs: 0,
|
|
4366
4767
|
}, false);
|
|
4367
4768
|
}
|
|
4769
|
+
/**
|
|
4770
|
+
* Run the two independent Scout evidence surfaces concurrently while keeping
|
|
4771
|
+
* their typed-event stores isolated. The main Scout store is the only store
|
|
4772
|
+
* visible to Plan; shard facts are merged in completion-order-independent
|
|
4773
|
+
* order after both sessions settle. This gives discovery real parallelism
|
|
4774
|
+
* without allowing sibling models to race a shared revision counter.
|
|
4775
|
+
*/
|
|
4776
|
+
function aggregateParallelPiResults(results) {
|
|
4777
|
+
const first = results[0];
|
|
4778
|
+
const failed = results.find((result) => !result.ok);
|
|
4779
|
+
const representative = failed ?? first;
|
|
4780
|
+
return {
|
|
4781
|
+
...representative,
|
|
4782
|
+
ok: failed === undefined,
|
|
4783
|
+
failureCategory: failed?.failureCategory ?? "success",
|
|
4784
|
+
durationMs: Math.max(...results.map((result) => result.durationMs), 0),
|
|
4785
|
+
exitCode: failed ? failed.exitCode : 0,
|
|
4786
|
+
stderr: results.map((result) => result.stderr).filter(Boolean).join("\n"),
|
|
4787
|
+
tokensUsed: results.reduce((total, result) => total + result.tokensUsed, 0),
|
|
4788
|
+
parsedEvents: results.reduce((total, result) => total + result.parsedEvents, 0),
|
|
4789
|
+
attemptedModels: [
|
|
4790
|
+
...new Set(results.flatMap((result) => result.attemptedModels)),
|
|
4791
|
+
],
|
|
4792
|
+
fallbackUsed: results.some((result) => result.fallbackUsed),
|
|
4793
|
+
timedOut: results.some((result) => result.timedOut),
|
|
4794
|
+
};
|
|
4795
|
+
}
|
|
4796
|
+
function combineSequentialPiResults(first, second) {
|
|
4797
|
+
return {
|
|
4798
|
+
...second,
|
|
4799
|
+
durationMs: first.durationMs + second.durationMs,
|
|
4800
|
+
stderr: [first.stderr, second.stderr].filter(Boolean).join("\n"),
|
|
4801
|
+
tokensUsed: first.tokensUsed + second.tokensUsed,
|
|
4802
|
+
parsedEvents: first.parsedEvents + second.parsedEvents,
|
|
4803
|
+
attemptedModels: [
|
|
4804
|
+
...new Set([...first.attemptedModels, ...second.attemptedModels]),
|
|
4805
|
+
],
|
|
4806
|
+
fallbackUsed: first.fallbackUsed || second.fallbackUsed,
|
|
4807
|
+
timedOut: first.timedOut || second.timedOut,
|
|
4808
|
+
};
|
|
4809
|
+
}
|
|
4810
|
+
async function runFrontendScoutParallelSessions(input) {
|
|
4811
|
+
const [{ createTypedEventStore }] = await Promise.all([
|
|
4812
|
+
import("../workflows/dag/frontend-typed-event-store.js"),
|
|
4813
|
+
]);
|
|
4814
|
+
const shards = [
|
|
4815
|
+
{
|
|
4816
|
+
id: "surface",
|
|
4817
|
+
toolName: "record_target_surface",
|
|
4818
|
+
instruction: [
|
|
4819
|
+
"PARALLEL SCOUT SHARD — target surface only.",
|
|
4820
|
+
"Inspect routes, entrypoints, implementation ownership, data source, and applicable test paths.",
|
|
4821
|
+
"Call record_target_surface exactly once with the complete runtime-evidenced surface. Do not call record_design_evidence.",
|
|
4822
|
+
].join(" "),
|
|
4823
|
+
},
|
|
4824
|
+
{
|
|
4825
|
+
id: "design",
|
|
4826
|
+
toolName: "record_design_evidence",
|
|
4827
|
+
instruction: [
|
|
4828
|
+
"PARALLEL SCOUT SHARD — design evidence only.",
|
|
4829
|
+
"Inspect the frontend framework, styling/theme conventions, reusable components, and relevant design/spec files.",
|
|
4830
|
+
"Call record_design_evidence for the evidence you actually read. Do not call record_target_surface.",
|
|
4831
|
+
].join(" "),
|
|
4832
|
+
},
|
|
4833
|
+
];
|
|
4834
|
+
const outcomes = await Promise.all(shards.map(async (shard) => {
|
|
4835
|
+
let shardTools;
|
|
4836
|
+
let shardResult;
|
|
4837
|
+
try {
|
|
4838
|
+
const store = createTypedEventStore();
|
|
4839
|
+
const shardNodeId = `${input.nodeId}/parallel/${shard.id}`;
|
|
4840
|
+
shardTools = await createFrontendScoutEvidenceTools({
|
|
4841
|
+
attemptId: `${input.attemptId}:parallel:${shard.id}`,
|
|
4842
|
+
store,
|
|
4843
|
+
runDir: input.runDir,
|
|
4844
|
+
nodeId: shardNodeId,
|
|
4845
|
+
workspaceRoot: input.workspaceRoot,
|
|
4846
|
+
sourceDeclaredPaths: input.sourceDeclaredPaths,
|
|
4847
|
+
});
|
|
4848
|
+
const customTools = shardTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
4849
|
+
tool !== null &&
|
|
4850
|
+
tool.name === shard.toolName);
|
|
4851
|
+
const sessionOptions = {
|
|
4852
|
+
...input.sessionOptions,
|
|
4853
|
+
sessionEventsPath: path.join(input.runDir, shardNodeId, "session-events.jsonl"),
|
|
4854
|
+
};
|
|
4855
|
+
shardResult = await input.piStepFn({
|
|
4856
|
+
...sessionOptions,
|
|
4857
|
+
prompt: `${input.basePrompt}\n\n${shard.instruction}`,
|
|
4858
|
+
writerToolPolicy: {
|
|
4859
|
+
requireSdk: true,
|
|
4860
|
+
customTools: [...customTools, ...(input.readBudgetTools ?? [])],
|
|
4861
|
+
},
|
|
4862
|
+
});
|
|
4863
|
+
await shardTools.flush();
|
|
4864
|
+
return { shard, result: shardResult, tools: shardTools };
|
|
4865
|
+
}
|
|
4866
|
+
catch (error) {
|
|
4867
|
+
const crashMessage = `frontend scout parallel shard ${shard.id} crashed: ${error instanceof Error ? error.message : String(error)}`;
|
|
4868
|
+
return {
|
|
4869
|
+
shard,
|
|
4870
|
+
tools: shardTools,
|
|
4871
|
+
result: shardResult
|
|
4872
|
+
? {
|
|
4873
|
+
...shardResult,
|
|
4874
|
+
ok: false,
|
|
4875
|
+
failureCategory: shardResult.ok
|
|
4876
|
+
? "invalid-output"
|
|
4877
|
+
: shardResult.failureCategory,
|
|
4878
|
+
stderr: [shardResult.stderr, crashMessage]
|
|
4879
|
+
.filter(Boolean)
|
|
4880
|
+
.join("\n"),
|
|
4881
|
+
}
|
|
4882
|
+
: {
|
|
4883
|
+
ok: false,
|
|
4884
|
+
assistantText: "",
|
|
4885
|
+
command: [],
|
|
4886
|
+
durationMs: 0,
|
|
4887
|
+
exitCode: null,
|
|
4888
|
+
failureCategory: "tool-policy",
|
|
4889
|
+
modelDisplay: "unknown",
|
|
4890
|
+
parsedEvents: 0,
|
|
4891
|
+
stderr: crashMessage,
|
|
4892
|
+
stdout: "",
|
|
4893
|
+
timedOut: false,
|
|
4894
|
+
attemptedModels: [],
|
|
4895
|
+
fallbackUsed: false,
|
|
4896
|
+
tokensUsed: 0,
|
|
4897
|
+
},
|
|
4898
|
+
};
|
|
4899
|
+
}
|
|
4900
|
+
}));
|
|
4901
|
+
const ordered = [...outcomes].sort((left, right) => left.shard.id.localeCompare(right.shard.id));
|
|
4902
|
+
try {
|
|
4903
|
+
for (const outcome of ordered) {
|
|
4904
|
+
if (outcome.result.ok && outcome.tools) {
|
|
4905
|
+
await input.mainTools.adoptCommittedFacts(outcome.tools.committedFacts());
|
|
4906
|
+
}
|
|
4907
|
+
}
|
|
4908
|
+
}
|
|
4909
|
+
catch (error) {
|
|
4910
|
+
const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4911
|
+
return {
|
|
4912
|
+
...aggregate,
|
|
4913
|
+
ok: false,
|
|
4914
|
+
failureCategory: "invalid-output",
|
|
4915
|
+
stderr: [
|
|
4916
|
+
aggregate.stderr,
|
|
4917
|
+
`frontend scout parallel fact merge failed: ${error instanceof Error ? error.message : String(error)}`,
|
|
4918
|
+
]
|
|
4919
|
+
.filter(Boolean)
|
|
4920
|
+
.join("\n"),
|
|
4921
|
+
};
|
|
4922
|
+
}
|
|
4923
|
+
const failed = ordered.find((outcome) => !outcome.result.ok);
|
|
4924
|
+
if (failed) {
|
|
4925
|
+
const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4926
|
+
return {
|
|
4927
|
+
...aggregate,
|
|
4928
|
+
ok: false,
|
|
4929
|
+
stderr: `${aggregate.stderr}\nfrontend scout parallel shard failed: ${failed.shard.id}`.trim(),
|
|
4930
|
+
};
|
|
4931
|
+
}
|
|
4932
|
+
const committedKinds = new Set(ordered.flatMap((outcome) => (outcome.tools?.committedFacts() ?? [])
|
|
4933
|
+
.map((record) => record.fact?.kind)
|
|
4934
|
+
.filter((kind) => typeof kind === "string")));
|
|
4935
|
+
const missing = ["target-surface", "design-evidence"].filter((kind) => !committedKinds.has(kind));
|
|
4936
|
+
if (missing.length > 0) {
|
|
4937
|
+
const fallback = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4938
|
+
return {
|
|
4939
|
+
...fallback,
|
|
4940
|
+
ok: false,
|
|
4941
|
+
failureCategory: "invalid-output",
|
|
4942
|
+
stderr: `${fallback.stderr}\nfrontend scout parallel shards committed no ${missing.join(" or ")} fact`.trim(),
|
|
4943
|
+
};
|
|
4944
|
+
}
|
|
4945
|
+
return aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4946
|
+
}
|
|
4368
4947
|
export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
|
|
4369
4948
|
const started = Date.now();
|
|
4370
4949
|
const persona = resolveDagPiPersona(input.task);
|
|
@@ -4459,6 +5038,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4459
5038
|
let planLedgerTools;
|
|
4460
5039
|
let contractTools;
|
|
4461
5040
|
let scoutEvidenceTools;
|
|
5041
|
+
let scoutSourceDeclaredPaths;
|
|
4462
5042
|
let readBudgetTools;
|
|
4463
5043
|
const commandPolicy = resolveDagCommandPolicy(input.task.commandPolicy);
|
|
4464
5044
|
const allowsPlaywrightCli = dagCommandPolicyAllows(input.task.commandPolicy, "playwright-cli");
|
|
@@ -4626,16 +5206,17 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4626
5206
|
try {
|
|
4627
5207
|
const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
4628
5208
|
const store = createTypedEventStore();
|
|
5209
|
+
scoutSourceDeclaredPaths = await resolveFrontendScoutSourceDeclaredPaths({
|
|
5210
|
+
cwd: input.cwd,
|
|
5211
|
+
spec: meta.spec,
|
|
5212
|
+
});
|
|
4629
5213
|
scoutEvidenceTools = await createFrontendScoutEvidenceTools({
|
|
4630
5214
|
attemptId: `${meta.runId}:${input.task.id}`,
|
|
4631
5215
|
store,
|
|
4632
5216
|
runDir: meta.runDir,
|
|
4633
5217
|
nodeId: input.task.id,
|
|
4634
5218
|
workspaceRoot: input.cwd,
|
|
4635
|
-
sourceDeclaredPaths:
|
|
4636
|
-
cwd: input.cwd,
|
|
4637
|
-
spec: meta.spec,
|
|
4638
|
-
}),
|
|
5219
|
+
sourceDeclaredPaths: scoutSourceDeclaredPaths,
|
|
4639
5220
|
});
|
|
4640
5221
|
writerToolPolicy = {
|
|
4641
5222
|
requireSdk: true,
|
|
@@ -4671,6 +5252,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4671
5252
|
declaredUiStateIds: await resolveFrontendDeclaredUiStateIds({
|
|
4672
5253
|
runDir: meta.runDir,
|
|
4673
5254
|
}),
|
|
5255
|
+
canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
|
|
5256
|
+
runDir: meta.runDir,
|
|
5257
|
+
}),
|
|
4674
5258
|
workspaceRoot: input.cwd,
|
|
4675
5259
|
});
|
|
4676
5260
|
writerToolPolicy = {
|
|
@@ -4811,11 +5395,27 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4811
5395
|
}
|
|
4812
5396
|
: undefined,
|
|
4813
5397
|
};
|
|
4814
|
-
if (
|
|
4815
|
-
|
|
4816
|
-
|
|
4817
|
-
|
|
4818
|
-
|
|
5398
|
+
if (isFrontendScoutEvidenceNode(input.task) &&
|
|
5399
|
+
scoutEvidenceTools &&
|
|
5400
|
+
input.task.complexity !== "LOW") {
|
|
5401
|
+
result = await runFrontendScoutParallelSessions({
|
|
5402
|
+
piStepFn,
|
|
5403
|
+
sessionOptions: piSessionOptions,
|
|
5404
|
+
basePrompt: input.prompt,
|
|
5405
|
+
mainTools: scoutEvidenceTools,
|
|
5406
|
+
runDir: meta.runDir,
|
|
5407
|
+
nodeId: input.task.id,
|
|
5408
|
+
attemptId: `${meta.runId}:${input.task.id}`,
|
|
5409
|
+
workspaceRoot: input.cwd,
|
|
5410
|
+
sourceDeclaredPaths: scoutSourceDeclaredPaths,
|
|
5411
|
+
readBudgetTools: readBudgetTools?.customTools,
|
|
5412
|
+
});
|
|
5413
|
+
}
|
|
5414
|
+
else if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
|
|
5415
|
+
// Frontend-only split: independent coverage map sessions feed a single
|
|
5416
|
+
// reducer (UX decisions -> global policy -> finalize), mirroring the
|
|
5417
|
+
// backend-test template's module sharding. Small plans normally have one
|
|
5418
|
+
// coverage batch and retain the compact path below.
|
|
4819
5419
|
let planRequirementIds = [];
|
|
4820
5420
|
const planRequirementCosts = new Map();
|
|
4821
5421
|
const behaviorRequiredRequirementIds = [];
|
|
@@ -4842,26 +5442,189 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4842
5442
|
catch {
|
|
4843
5443
|
// Unreadable ledger falls back to a single coverage session.
|
|
4844
5444
|
}
|
|
4845
|
-
|
|
5445
|
+
// Independent requirement-coverage batches are map workers. Each worker
|
|
5446
|
+
// owns an isolated typed-event store; only after all workers settle do we
|
|
5447
|
+
// merge facts into the main Plan ledger and run the single UX/global/
|
|
5448
|
+
// finalize reducer. This avoids revision races while shortening the
|
|
5449
|
+
// longest coverage phase for large plans.
|
|
5450
|
+
const runPlanSessions = (options) => runFrontendPlanSegmentedSessions({
|
|
4846
5451
|
piStepFn,
|
|
4847
|
-
sessionOptions:
|
|
4848
|
-
basePrompt: input.prompt,
|
|
5452
|
+
sessionOptions: options.sessionOptions,
|
|
5453
|
+
basePrompt: options.basePrompt ?? input.prompt,
|
|
4849
5454
|
attempt: input.attempt ?? 1,
|
|
4850
|
-
committedFactCount: () =>
|
|
4851
|
-
requirementIds: planRequirementIds,
|
|
5455
|
+
committedFactCount: () => options.ledgerTools.committedFactCount(),
|
|
5456
|
+
requirementIds: options.requirementIds ?? planRequirementIds,
|
|
4852
5457
|
requirementCosts: planRequirementCosts,
|
|
4853
|
-
|
|
4854
|
-
|
|
4855
|
-
|
|
5458
|
+
...(options.parallelCoverageOnly !== undefined
|
|
5459
|
+
? { parallelCoverageOnly: options.parallelCoverageOnly }
|
|
5460
|
+
: {}),
|
|
5461
|
+
...(options.compactSmallPlan !== undefined
|
|
5462
|
+
? { compactSmallPlan: options.compactSmallPlan }
|
|
5463
|
+
: {}),
|
|
5464
|
+
committedRequirementIds: () => options.ledgerTools.committedRequirementIds(),
|
|
5465
|
+
committedFacts: () => options.ledgerTools.committedFacts(),
|
|
4856
5466
|
behaviorRequiredRequirementIds,
|
|
4857
|
-
setActiveRequirementScope: (requirementIds) =>
|
|
5467
|
+
setActiveRequirementScope: (requirementIds) => options.ledgerTools.setActiveRequirementScope(requirementIds),
|
|
4858
5468
|
segmentCustomTools: (toolNames) => toolNames === null
|
|
4859
|
-
?
|
|
4860
|
-
:
|
|
5469
|
+
? options.ledgerTools.customTools
|
|
5470
|
+
: options.ledgerTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
4861
5471
|
tool !== null &&
|
|
4862
5472
|
toolNames.has(tool.name)),
|
|
4863
|
-
flushLedger: () =>
|
|
5473
|
+
flushLedger: () => options.ledgerTools.flush(),
|
|
4864
5474
|
});
|
|
5475
|
+
if (planRequirementIds.length > 1) {
|
|
5476
|
+
const estimatedPlanCalls = planRequirementIds.reduce((total, id) => total + Math.max(1, planRequirementCosts.get(id) ?? 2), 0);
|
|
5477
|
+
const compactEligibleBeforeSharding = planRequirementIds.length <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
|
|
5478
|
+
estimatedPlanCalls > 12 &&
|
|
5479
|
+
estimatedPlanCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
|
|
5480
|
+
countFrontendPlanTargetSurfaces(input.prompt) === 1;
|
|
5481
|
+
if (compactEligibleBeforeSharding) {
|
|
5482
|
+
// Decide the small-plan topology before creating coverage shards.
|
|
5483
|
+
// Sharding first would make the compact two-session path unreachable.
|
|
5484
|
+
result = await runPlanSessions({
|
|
5485
|
+
ledgerTools: planLedgerTools,
|
|
5486
|
+
sessionOptions: piSessionOptions,
|
|
5487
|
+
compactSmallPlan: true,
|
|
5488
|
+
});
|
|
5489
|
+
}
|
|
5490
|
+
else {
|
|
5491
|
+
const coverageBatches = batchFrontendPlanRequirements({
|
|
5492
|
+
requirementIds: planRequirementIds,
|
|
5493
|
+
maxEstimatedRecordCalls: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS,
|
|
5494
|
+
maxRequirements: 4,
|
|
5495
|
+
requirementCosts: planRequirementCosts,
|
|
5496
|
+
});
|
|
5497
|
+
if (coverageBatches.length > 1) {
|
|
5498
|
+
const shardResults = await mapWithConcurrency(coverageBatches, FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY, async (slice, index) => {
|
|
5499
|
+
let shardTools;
|
|
5500
|
+
let shardResult;
|
|
5501
|
+
try {
|
|
5502
|
+
const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
5503
|
+
shardTools = await createFrontendPlanLedgerTools({
|
|
5504
|
+
attemptId: `${meta.runId}:${input.task.id}:parallel:${index + 1}`,
|
|
5505
|
+
store: createTypedEventStore(),
|
|
5506
|
+
runDir: meta.runDir,
|
|
5507
|
+
nodeId: `${input.task.id}/parallel/coverage-${index + 1}`,
|
|
5508
|
+
skeleton: input.task.structuredContractOutput?.skeleton,
|
|
5509
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
5510
|
+
requirementIds: slice,
|
|
5511
|
+
writeSetPatterns: input.task.writeSet,
|
|
5512
|
+
canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
|
|
5513
|
+
runDir: meta.runDir,
|
|
5514
|
+
}),
|
|
5515
|
+
componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
|
|
5516
|
+
cwd: input.cwd,
|
|
5517
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
5518
|
+
}),
|
|
5519
|
+
});
|
|
5520
|
+
const shardSessionOptions = {
|
|
5521
|
+
...piSessionOptions,
|
|
5522
|
+
sessionEventsPath: path.join(meta.runDir, input.task.id, "parallel", `coverage-${index + 1}`, "session-events.jsonl"),
|
|
5523
|
+
};
|
|
5524
|
+
shardResult = await runPlanSessions({
|
|
5525
|
+
ledgerTools: shardTools,
|
|
5526
|
+
sessionOptions: shardSessionOptions,
|
|
5527
|
+
requirementIds: slice,
|
|
5528
|
+
basePrompt: `${input.prompt}\n\nPARALLEL COVERAGE SHARD ${index + 1}: use the frozen canonical behavior verification-target ids listed in the record_plan_verification_target tool description — do not prefix ids with a shard namespace or invent variant ids; identical cross-shard targets are deduped, divergent ones fail the merge.`,
|
|
5529
|
+
parallelCoverageOnly: true,
|
|
5530
|
+
});
|
|
5531
|
+
await shardTools.flush();
|
|
5532
|
+
return { index, result: shardResult, tools: shardTools };
|
|
5533
|
+
}
|
|
5534
|
+
catch (error) {
|
|
5535
|
+
const crashMessage = `frontend plan coverage shard ${index + 1} crashed: ${error instanceof Error ? error.message : String(error)}`;
|
|
5536
|
+
return {
|
|
5537
|
+
index,
|
|
5538
|
+
tools: shardTools,
|
|
5539
|
+
result: shardResult
|
|
5540
|
+
? {
|
|
5541
|
+
...shardResult,
|
|
5542
|
+
ok: false,
|
|
5543
|
+
failureCategory: shardResult.ok
|
|
5544
|
+
? "invalid-output"
|
|
5545
|
+
: shardResult.failureCategory,
|
|
5546
|
+
stderr: [shardResult.stderr, crashMessage]
|
|
5547
|
+
.filter(Boolean)
|
|
5548
|
+
.join("\n"),
|
|
5549
|
+
}
|
|
5550
|
+
: {
|
|
5551
|
+
ok: false,
|
|
5552
|
+
assistantText: "",
|
|
5553
|
+
command: [],
|
|
5554
|
+
durationMs: 0,
|
|
5555
|
+
exitCode: null,
|
|
5556
|
+
failureCategory: "tool-policy",
|
|
5557
|
+
modelDisplay: "unknown",
|
|
5558
|
+
parsedEvents: 0,
|
|
5559
|
+
stderr: crashMessage,
|
|
5560
|
+
stdout: "",
|
|
5561
|
+
timedOut: false,
|
|
5562
|
+
attemptedModels: [],
|
|
5563
|
+
fallbackUsed: false,
|
|
5564
|
+
tokensUsed: 0,
|
|
5565
|
+
},
|
|
5566
|
+
};
|
|
5567
|
+
}
|
|
5568
|
+
});
|
|
5569
|
+
const coverageResult = aggregateParallelPiResults(shardResults.map((shard) => shard.result));
|
|
5570
|
+
let mergeFailure;
|
|
5571
|
+
try {
|
|
5572
|
+
// Flatten every shard's facts FIRST, then pre-reduce: shards
|
|
5573
|
+
// partition requirements but a frozen verification target can
|
|
5574
|
+
// span shards, so same-id targets must merge across shards
|
|
5575
|
+
// before adoption, not per shard.
|
|
5576
|
+
await planLedgerTools.adoptCommittedFacts(reduceParallelCoverageShardRecords(shardResults
|
|
5577
|
+
.filter((item) => item.result.ok && item.tools)
|
|
5578
|
+
.flatMap((shard) => normalizeParallelCoverageShardRecords(shard.tools.committedFacts(), shard.index + 1))));
|
|
5579
|
+
}
|
|
5580
|
+
catch (error) {
|
|
5581
|
+
mergeFailure = error;
|
|
5582
|
+
}
|
|
5583
|
+
const failedShard = shardResults.find((shard) => !shard.result.ok);
|
|
5584
|
+
if (mergeFailure) {
|
|
5585
|
+
result = {
|
|
5586
|
+
...coverageResult,
|
|
5587
|
+
ok: false,
|
|
5588
|
+
failureCategory: "invalid-output",
|
|
5589
|
+
stderr: [
|
|
5590
|
+
coverageResult.stderr,
|
|
5591
|
+
`frontend plan coverage shard merge failed: ${mergeFailure instanceof Error ? mergeFailure.message : String(mergeFailure)}`,
|
|
5592
|
+
]
|
|
5593
|
+
.filter(Boolean)
|
|
5594
|
+
.join("\n"),
|
|
5595
|
+
};
|
|
5596
|
+
}
|
|
5597
|
+
else if (failedShard) {
|
|
5598
|
+
result = {
|
|
5599
|
+
...coverageResult,
|
|
5600
|
+
ok: false,
|
|
5601
|
+
stderr: `${coverageResult.stderr}\nfrontend plan coverage shard ${failedShard.index + 1} failed before reduce`.trim(),
|
|
5602
|
+
};
|
|
5603
|
+
}
|
|
5604
|
+
else {
|
|
5605
|
+
const reducerResult = await runPlanSessions({
|
|
5606
|
+
ledgerTools: planLedgerTools,
|
|
5607
|
+
sessionOptions: piSessionOptions,
|
|
5608
|
+
});
|
|
5609
|
+
result = combineSequentialPiResults(coverageResult, reducerResult);
|
|
5610
|
+
}
|
|
5611
|
+
}
|
|
5612
|
+
else {
|
|
5613
|
+
result = await runPlanSessions({
|
|
5614
|
+
ledgerTools: planLedgerTools,
|
|
5615
|
+
sessionOptions: piSessionOptions,
|
|
5616
|
+
compactSmallPlan: true,
|
|
5617
|
+
});
|
|
5618
|
+
}
|
|
5619
|
+
}
|
|
5620
|
+
}
|
|
5621
|
+
else {
|
|
5622
|
+
result = await runPlanSessions({
|
|
5623
|
+
ledgerTools: planLedgerTools,
|
|
5624
|
+
sessionOptions: piSessionOptions,
|
|
5625
|
+
compactSmallPlan: true,
|
|
5626
|
+
});
|
|
5627
|
+
}
|
|
4865
5628
|
}
|
|
4866
5629
|
else {
|
|
4867
5630
|
result = await piStepFn({
|
|
@@ -5408,7 +6171,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
5408
6171
|
runDir: meta.runDir,
|
|
5409
6172
|
progress,
|
|
5410
6173
|
attempt: input.attempt ?? 1,
|
|
5411
|
-
maxAttempts: input.task.retryPolicy?.maxAttempts ??
|
|
6174
|
+
maxAttempts: input.task.retryPolicy?.maxAttempts ?? 5,
|
|
5412
6175
|
});
|
|
5413
6176
|
if (progress.status !== "PASS") {
|
|
5414
6177
|
const classified = classifyBackendTestWriterCompletenessFailure(progress);
|
|
@@ -5458,7 +6221,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
5458
6221
|
stderrParts.push(completenessFailure.detail);
|
|
5459
6222
|
}
|
|
5460
6223
|
if (writerThinkingExhausted) {
|
|
5461
|
-
stderrParts.push(`${
|
|
6224
|
+
stderrParts.push(`${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before any write tool call; retry from the existing workspace and complete only unfinished targets`);
|
|
5462
6225
|
}
|
|
5463
6226
|
if (meta.writeGuardAttribution === "best-effort") {
|
|
5464
6227
|
stderrParts.push("write guard note: concurrent rank writers use best-effort per-node attribution; keep same-rank writeSet entries disjoint");
|
|
@@ -5531,14 +6294,11 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
5531
6294
|
rawFailureCategory === "context-overflow" &&
|
|
5532
6295
|
(changeManifestChangedFiles?.length ?? 0) > 0
|
|
5533
6296
|
? "partial-success-with-context-overflow"
|
|
5534
|
-
: //
|
|
5535
|
-
//
|
|
5536
|
-
//
|
|
5537
|
-
// the empty-output base category so provider/transport failures keep
|
|
5538
|
-
// their original category, and only when the completeness gate did not
|
|
5539
|
-
// upgrade to incomplete-write-set above.
|
|
6297
|
+
: // output-limit: a length-stopped attempt is capacity truncation, not
|
|
6298
|
+
// empty output. Preserve the raw category for diagnostics and let the
|
|
6299
|
+
// node retry from committed facts/current workspace state.
|
|
5540
6300
|
writerThinkingExhausted
|
|
5541
|
-
?
|
|
6301
|
+
? OUTPUT_LIMIT_RETRY_CATEGORY
|
|
5542
6302
|
: writerCleanTimeout
|
|
5543
6303
|
? WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY
|
|
5544
6304
|
: // writer-budget-exhausted: the provider session consumed an
|