@tea-agent/loop-agent 0.44.0-next.10 → 0.44.0-next.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/build-stamp.json +3 -3
- package/dist/executors/dag-pi/sessions/index.js +6 -0
- package/dist/executors/dag-pi/sessions/plan-batches.js +280 -0
- package/dist/executors/dag-pi/sessions/plan-prompts.js +367 -0
- package/dist/executors/dag-pi/sessions/scout-parallel.js +197 -0
- package/dist/executors/dag-pi/sessions/segmented-plan.js +1184 -0
- package/dist/executors/dag-pi/sessions/writer-evidence.js +60 -0
- package/dist/executors/dag-pi-executor.js +12 -2075
- package/dist/workflows/dag/checkpoint.js +686 -0
- package/dist/workflows/dag/hybrid/sources.js +1812 -0
- package/dist/workflows/dag/hybrid/templates/backend-test.js +1807 -0
- package/dist/workflows/dag/hybrid/templates/frontend-test.js +702 -0
- package/dist/workflows/dag/hybrid/templates/frontend.js +1529 -0
- package/dist/workflows/dag/hybrid/templates/index.js +8 -0
- package/dist/workflows/dag/hybrid/templates/kg-bootstrap.js +449 -0
- package/dist/workflows/dag/hybrid/templates/knowledge-sync.js +495 -0
- package/dist/workflows/dag/hybrid/templates/shared.js +101 -0
- package/dist/workflows/dag/hybrid/templates/standard.js +265 -0
- package/dist/workflows/dag/hybrid/types.js +32 -0
- package/dist/workflows/dag/init-hybrid.js +42 -7162
- package/dist/workflows/dag/runner.js +11 -1178
- package/dist/workflows/dag/terminal-status.js +502 -0
- package/docs/architecture/dag-execution.md +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,1184 @@
|
|
|
1
|
+
/** DAG Pi plan 分段会话执行器:顺序执行、预算与缺失事实修复次序、terminal 行为。副作用集中于此。 */
|
|
2
|
+
import { classifyFrontendPlanRecovery } from "../../../workflows/dag/frontend-plan-recovery-policy.js";
|
|
3
|
+
import { collectFrontendExecutionGroups } from "../../../workflows/dag/frontend-execution-groups.js";
|
|
4
|
+
import { FRONTEND_SCOPE_TARGET_BYTES, packFrontendInputUnits, parseFrontendInputBlock, projectFrontendContractPrompt, projectFrontendInputScope } from "../../../workflows/dag/frontend-input-projection.js";
|
|
5
|
+
import { observeFrontendFinalization, observeFrontendSession } from "../../../workflows/dag/frontend-session-budget.js";
|
|
6
|
+
import { collectFrontendPlanMissingFacts, collectFrontendPlanPhaseMissingFacts, committedFactFromPlanRecord, planFactStringList } from "../plan/facts.js";
|
|
7
|
+
import { OUTPUT_LIMIT_RETRY_CATEGORY } from "../../../workflows/dag/retry-policy.js";
|
|
8
|
+
import { FRONTEND_PLAN_BATCH_MAX_SESSIONS, FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS, FRONTEND_PLAN_SEGMENTS, buildFrontendPlanWorkload, collectFrontendVerificationCommandFiles } from "./plan-batches.js";
|
|
9
|
+
import { compactFrontendPlanLedgerContext, compactFrontendPlanPromptForRequirementSlice } from "./plan-prompts.js";
|
|
10
|
+
import { isPlannerThinkingExhausted, readWriterThinkingExhaustionEvidence } from "./writer-evidence.js";
|
|
11
|
+
import { combineSequentialPiResults } from "./scout-parallel.js";
|
|
12
|
+
export async function runFrontendReviewSegmentedSessions(input) {
|
|
13
|
+
const protocol = input.tools.scopeProtocol;
|
|
14
|
+
const terminalKinds = input.phase === "review" ? ["approve_review", "request_review_changes"] : ["approve_design", "request_design_changes"];
|
|
15
|
+
const terminal = () => protocol.committedFacts().some(r => terminalKinds.includes(String(r.fact.kind)));
|
|
16
|
+
const targetBytes = input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES;
|
|
17
|
+
const queue = packFrontendInputUnits(input.inventory.scopes.filter(s => !protocol.completedScopeIds().has(s.id)), { targetBytes, maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits }).map(scopes => ({ scopes, repairs: 0 }));
|
|
18
|
+
if (!queue.length)
|
|
19
|
+
queue.push({ scopes: [], repairs: 0 });
|
|
20
|
+
let calls = 0;
|
|
21
|
+
let last = { ok: true, assistantText: "", command: [], durationMs: 0, exitCode: 0, failureCategory: "success", modelDisplay: "unknown", parsedEvents: 0, stderr: "", stdout: "", timedOut: false, attemptedModels: [], fallbackUsed: false, tokensUsed: 0 };
|
|
22
|
+
for (let index = 0; index < queue.length; index++) {
|
|
23
|
+
await input.tools.flush();
|
|
24
|
+
await input.inventory.validate();
|
|
25
|
+
if (terminal()) {
|
|
26
|
+
protocol.assertComplete();
|
|
27
|
+
return last;
|
|
28
|
+
}
|
|
29
|
+
const item = queue[index];
|
|
30
|
+
const scopes = item.scopes.filter(s => !protocol.completedScopeIds().has(s.id));
|
|
31
|
+
protocol.setActiveScope(scopes.map(s => s.id));
|
|
32
|
+
const finalScope = input.inventory.scopes.every(s => protocol.completedScopeIds().has(s.id) || scopes.some(current => current.id === s.id));
|
|
33
|
+
const customTools = finalScope ? input.customTools : input.customTools.filter(t => !terminalKinds.includes(String(t.name)));
|
|
34
|
+
const prompt = `${input.basePrompt}\n<frontend_review_authority>\nThe Contract acceptance criteria, constraints, required deliverables, UI-state declarations, and verification expectations in the supplied input are authoritative. Review the actual diff and evidence against those facts; do not replace them with a Plan-derived interpretation.\n</frontend_review_authority>\n<frontend_review_scope>\n${JSON.stringify({ semantics: "full", inventoryDigest: input.inventory.digest, scopes, previouslyCompleted: [...protocol.completedScopeIds()], savedFindings: protocol.committedFacts().filter(r => String(r.fact.kind).endsWith("-finding")).map(r => ({ id: r.fact.id, finding: r.fact.finding })) })}\n</frontend_review_scope>\nReview the complete supplied scopes, saving each finding immediately. Call complete_review_scope for each exact id only after all its independent permissions, thresholds, errors and evidence have been checked. ${finalScope ? "After every scope is complete, make one independent overall approve/request_changes decision; persisted findings cannot be omitted." : "More scopes remain. Do not finalize or reread already completed scopes unless resolving a cross-scope issue."}`;
|
|
35
|
+
if (Buffer.byteLength(prompt) > targetBytes && scopes.length > 1) {
|
|
36
|
+
const at = Math.ceil(scopes.length / 2);
|
|
37
|
+
queue.splice(index, 1, { scopes: scopes.slice(0, at), repairs: 0 }, { scopes: scopes.slice(at), repairs: 0 });
|
|
38
|
+
index--;
|
|
39
|
+
continue;
|
|
40
|
+
}
|
|
41
|
+
const completionInstruction = `${item.repairs ? "REPAIR: the prior session did not commit all required checkpoints/verdict. Do not repeat the review narrative. " : ""}Complete these exact runtime scope IDs using complete_review_scope: ${JSON.stringify(scopes.map(scope => scope.id))}. These are scope IDs, not node IDs. ${finalScope ? `After those checkpoints succeed, call exactly one terminal tool: ${terminalKinds.join(" or ")}. A prose conclusion is not a committed verdict.` : "More scopes remain; do not submit an overall verdict yet."}`;
|
|
42
|
+
const userMessage = `${input.sessionOptions.userMessage}\n\n${completionInstruction}`;
|
|
43
|
+
if (++calls > 128)
|
|
44
|
+
return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_REVIEW_RECOVERY_EXHAUSTED: session quota reached" };
|
|
45
|
+
last = await observeFrontendSession({ ...input.observation, phase: `${input.phase}/scope`, scopeIds: scopes.map(s => s.id), prompt, userMessage, customTools,
|
|
46
|
+
artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${calls}.json` : undefined,
|
|
47
|
+
committedCount: () => protocol.committedFacts().length, durableCommittedCount: () => protocol.committedFacts().length,
|
|
48
|
+
}, observer => input.piStepFn({ ...input.sessionOptions, prompt, userMessage, onAttemptObservation: observer, writerToolPolicy: { requireSdk: true, customTools } }));
|
|
49
|
+
await input.tools.flush();
|
|
50
|
+
await input.inventory.validate();
|
|
51
|
+
if (last.timedOut || /interrupt|termination-unconfirmed|budget_breach/.test(last.failureCategory))
|
|
52
|
+
return { ...last, ok: false };
|
|
53
|
+
const capacity = readWriterThinkingExhaustionEvidence(last).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(last.failureCategory);
|
|
54
|
+
const missing = scopes.filter(s => !protocol.completedScopeIds().has(s.id));
|
|
55
|
+
if (capacity && (missing.length || !terminal() && finalScope)) {
|
|
56
|
+
if (missing.length === 1 && scopes.length === 1 || !missing.length && !scopes.length)
|
|
57
|
+
return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_INPUT_UNIT_TOO_LARGE: complete review unit or terminal exhausted" };
|
|
58
|
+
const at = Math.ceil(missing.length / 2);
|
|
59
|
+
const smaller = missing.length ? [missing.slice(0, at), missing.slice(at)].filter(s => s.length) : [[]];
|
|
60
|
+
queue.splice(index, 1, ...smaller.map(scopes => ({ scopes, repairs: 0 })));
|
|
61
|
+
index--;
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
if (!last.ok && !capacity) {
|
|
65
|
+
// Terminal-only completion: this protocol finishes through durable
|
|
66
|
+
// tools (complete_review_scope -> approve_review/request_review_changes),
|
|
67
|
+
// so the model can end its turn with no closing prose and the step
|
|
68
|
+
// classifies as empty-output even though the review is complete
|
|
69
|
+
// (smoke r27/r31/r32 committed the terminal and still reported
|
|
70
|
+
// empty-output; the node then replayed the committed fact in an
|
|
71
|
+
// extra attempt). Accept the segment here instead, so a completed
|
|
72
|
+
// review neither spends a replay attempt nor depends on the retry
|
|
73
|
+
// ladder, and a misclassified category (smoke r28 read a finding id
|
|
74
|
+
// containing UNAUTHORIZED as an auth error, which is not retryable)
|
|
75
|
+
// can no longer turn a committed review into a node failure.
|
|
76
|
+
if (last.failureCategory === "empty-output" && terminal() && !missing.length) {
|
|
77
|
+
try {
|
|
78
|
+
await input.tools.flush();
|
|
79
|
+
protocol.assertComplete();
|
|
80
|
+
return { ...last, ok: true, failureCategory: "success" };
|
|
81
|
+
}
|
|
82
|
+
catch {
|
|
83
|
+
// Unfinished scope checkpoints still fail this segment.
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
return last;
|
|
87
|
+
}
|
|
88
|
+
if (missing.length || finalScope && !terminal()) {
|
|
89
|
+
if (item.repairs >= 1)
|
|
90
|
+
return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_REVIEW_SCOPE_INCOMPLETE: required checkpoint or verdict missing" };
|
|
91
|
+
queue.splice(index, 1, { scopes: missing, repairs: item.repairs + 1 });
|
|
92
|
+
index--;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
protocol.assertComplete();
|
|
96
|
+
return terminal() ? { ...last, ok: true, failureCategory: "success" } : { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "FRONTEND_REVIEW_SCOPE_INCOMPLETE: missing independent verdict" };
|
|
97
|
+
}
|
|
98
|
+
export async function runFrontendScoutSegmentedSessions(input) {
|
|
99
|
+
const inventory = parseFrontendInputBlock(input.basePrompt, "scout");
|
|
100
|
+
if (!inventory)
|
|
101
|
+
throw Error("FRONTEND_INPUT_MISSING: Scout compiled inventory unavailable");
|
|
102
|
+
const units = collectFrontendExecutionGroups(inventory.payload.requirements).map(group => ({ ...group, id: `${group.kind === "unclassified" ? "requirement" : "group"}:${group.id}`, requirements: inventory.payload.requirements.filter(r => group.requirementIds.includes(r.id)) }));
|
|
103
|
+
const targetBytes = input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES;
|
|
104
|
+
const queue = packFrontendInputUnits(units, { targetBytes, maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits }).map(batch => ({ groups: batch, repairs: 0 }));
|
|
105
|
+
let last = { ok: true, assistantText: "", command: [], durationMs: 0, exitCode: 0, failureCategory: "success", modelDisplay: input.sessionOptions.modelConfig?.model ?? "unknown", parsedEvents: 0, stderr: "", stdout: "", timedOut: false, attemptedModels: [], fallbackUsed: false, tokensUsed: 0 };
|
|
106
|
+
let calls = 0;
|
|
107
|
+
for (let index = 0; index < queue.length; index++) {
|
|
108
|
+
await input.tools.flush();
|
|
109
|
+
const item = queue[index];
|
|
110
|
+
const completed = input.tools.completedRequirementIds();
|
|
111
|
+
const groups = item.groups.filter(g => g.requirementIds.some(id => !completed.has(id)));
|
|
112
|
+
if (!groups.length)
|
|
113
|
+
continue;
|
|
114
|
+
const ids = groups.flatMap(g => g.requirementIds);
|
|
115
|
+
const scopeId = input.tools.setActiveScope(ids);
|
|
116
|
+
const prompt = input.basePrompt.replace(inventory.block, `<frontend_scout_input>\n${JSON.stringify(projectFrontendInputScope(inventory.payload, ids))}\n</frontend_scout_input>`) +
|
|
117
|
+
`\nSCOUT SCOPE ${scopeId}: discover the related surfaces for exactly these complete obligations: ${ids.join(", ")}. Reuse proven paths from completed scope navigation; do not reread their content unless relevant new evidence is needed. Submit record_target_surface with scopeId="${scopeId}" after all discovery/design evidence for this scope. completeness=complete closes only this scope; runtime merges all scopes.\n` +
|
|
118
|
+
JSON.stringify({ completedScopePaths: input.tools.completedScopeFacts().map(f => ({ id: f.id, requirementIds: f.requirementIds, paths: f.surface?.implementationPaths })) });
|
|
119
|
+
const envelopeBytes = Buffer.byteLength(prompt + input.sessionOptions.userMessage + JSON.stringify(input.customTools.map(t => { const tool = t; return { name: tool.name, description: tool.description, parameters: tool.parameters }; })));
|
|
120
|
+
if (envelopeBytes > targetBytes && groups.length > 1) {
|
|
121
|
+
const at = Math.ceil(groups.length / 2);
|
|
122
|
+
queue.splice(index, 1, { groups: groups.slice(0, at), repairs: 0 }, { groups: groups.slice(at), repairs: 0 });
|
|
123
|
+
index--;
|
|
124
|
+
continue;
|
|
125
|
+
}
|
|
126
|
+
if (++calls > 128)
|
|
127
|
+
return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "SCOUT_RECOVERY_EXHAUSTED: shared session quota reached" };
|
|
128
|
+
last = await observeFrontendSession({ ...input.observation, phase: "scout/scope", scopeIds: ids, prompt, userMessage: input.sessionOptions.userMessage, customTools: input.customTools,
|
|
129
|
+
artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${calls}.json` : undefined,
|
|
130
|
+
committedCount: () => input.tools.committedFacts().length, durableCommittedCount: () => input.tools.committedFacts().length,
|
|
131
|
+
}, observer => input.piStepFn({ ...input.sessionOptions, prompt, onAttemptObservation: observer, writerToolPolicy: { requireSdk: true, customTools: input.customTools } }));
|
|
132
|
+
await input.tools.flush();
|
|
133
|
+
const missing = groups.filter(g => g.requirementIds.some(id => !input.tools.completedRequirementIds().has(id)));
|
|
134
|
+
if (last.timedOut || /interrupt|termination-unconfirmed|budget_breach/.test(last.failureCategory))
|
|
135
|
+
return { ...last, ok: false };
|
|
136
|
+
const capacity = readWriterThinkingExhaustionEvidence(last).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(last.failureCategory);
|
|
137
|
+
if (capacity && missing.length) {
|
|
138
|
+
if (missing.length === 1 && groups.length === 1)
|
|
139
|
+
return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: `${last.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: Scout ${missing[0].id}; refine the complete source unit; unchanged retries disabled` };
|
|
140
|
+
const at = Math.ceil(missing.length / 2);
|
|
141
|
+
queue.splice(index, 1, ...[missing.slice(0, at), missing.slice(at)].filter(batch => batch.length).map(batch => ({ groups: batch, repairs: 0 })));
|
|
142
|
+
index--;
|
|
143
|
+
continue;
|
|
144
|
+
}
|
|
145
|
+
if (!last.ok && !capacity)
|
|
146
|
+
return last;
|
|
147
|
+
if (missing.length) {
|
|
148
|
+
if (item.repairs >= 1)
|
|
149
|
+
return { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: "SCOUT_SCOPE_INCOMPLETE: unresolved discovery remains after local correction" };
|
|
150
|
+
queue.splice(index, 1, { groups: missing, repairs: item.repairs + 1 });
|
|
151
|
+
index--;
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
await input.tools.flush();
|
|
156
|
+
const { readCompleteScoutTargetSurface } = await import("../../../workflows/dag/frontend-committed-facts.js");
|
|
157
|
+
const closure = readCompleteScoutTargetSurface(input.tools.committedFacts());
|
|
158
|
+
return closure.ok ? { ...last, ok: true, failureCategory: "success" } : { ...last, ok: false, failureCategory: "frontend-input-exhausted", stderr: closure.reason };
|
|
159
|
+
}
|
|
160
|
+
export async function runFrontendContractSegmentedSessions(input) {
|
|
161
|
+
// Requirement identity, text, and source fragments are frozen in the
|
|
162
|
+
// runtime ledger. Seed those facts once; model sessions spend their budget
|
|
163
|
+
// on execution grouping, evidence expectations, and genuine decisions.
|
|
164
|
+
await input.tools.seedCanonicalRequirements();
|
|
165
|
+
// Build from the frozen runtime inventory if the caller has not rendered it yet.
|
|
166
|
+
const basePrompt = parseFrontendInputBlock(input.basePrompt, "contract") ? input.basePrompt : input.basePrompt +
|
|
167
|
+
`\n<frontend_contract_input>\nFrozen complete source obligations.\n${JSON.stringify({ requirements: input.tools.inputRequirements() })}\n</frontend_contract_input>`;
|
|
168
|
+
const inventory = parseFrontendInputBlock(basePrompt, "contract");
|
|
169
|
+
const pending = inventory.payload.requirements.filter(r => !input.tools.completedScopeRequirementIds().has(r.id));
|
|
170
|
+
const batches = packFrontendInputUnits(pending, { targetBytes: input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes, maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits });
|
|
171
|
+
if (!batches.length)
|
|
172
|
+
batches.push([]);
|
|
173
|
+
let last;
|
|
174
|
+
let invocation = 0;
|
|
175
|
+
const maxSessions = batches.length * 3 + 2;
|
|
176
|
+
const terminal = () => input.tools.committedFacts().some(r => r.fact.kind === "contract-finalized");
|
|
177
|
+
try {
|
|
178
|
+
for (let index = 0; index < batches.length; index += 1) {
|
|
179
|
+
let scopeIds = batches[index].map(r => r.id);
|
|
180
|
+
const finalScope = index === batches.length - 1;
|
|
181
|
+
for (let repair = 0; repair < 2; repair += 1) {
|
|
182
|
+
// Confirmed IDs remain visible until the model commits a scope checkpoint.
|
|
183
|
+
const completed = input.tools.completedScopeRequirementIds();
|
|
184
|
+
scopeIds = scopeIds.filter(id => !completed.has(id));
|
|
185
|
+
input.tools.setActiveRequirementScope(scopeIds);
|
|
186
|
+
const groupIndex = collectFrontendExecutionGroups(input.tools.committedFacts().filter(r => r.fact.kind === "requirement").map(r => ({ id: String(r.fact.id), execution: r.fact.execution })));
|
|
187
|
+
const shared = input.tools.committedFacts().filter(r => !["requirement", "contract-finalized", "contract-scope-completed"].includes(String(r.fact.kind))).map(r => r.fact);
|
|
188
|
+
const prompt = projectFrontendContractPrompt(basePrompt, scopeIds) +
|
|
189
|
+
`\nCONTRACT SCOPE: analyze only ${scopeIds.join(", ") || "(all scopes complete; verify global facts and terminal)"}. Requirement identity/text/source fragments are already committed by runtime; do not call record_requirement. Use record_requirement_execution for model-owned execution grouping, then submit decisions and evidence records. Call complete_contract_scope after ALL decisions for this scope. ` +
|
|
190
|
+
(finalScope ? "After complete scope coverage and source-bound deliverables, call finalize_contract. Correct rejected calls and retry." : "Do not finalize; subsequent complete scopes remain.") +
|
|
191
|
+
`\n<committed_contract_facts>\n${JSON.stringify({ facts: shared, executionGroups: groupIndex })}\n</committed_contract_facts>`;
|
|
192
|
+
const customTools = input.tools.customTools.filter(t => finalScope || t.name !== "finalize_contract");
|
|
193
|
+
invocation += 1;
|
|
194
|
+
if (invocation > maxSessions)
|
|
195
|
+
return { ...last, ok: false, failureCategory: "invalid-output", stderr: "CONTRACT_RECOVERY_EXHAUSTED: session quota exceeded" };
|
|
196
|
+
const committedBefore = input.tools.committedFacts().length;
|
|
197
|
+
last = await observeFrontendSession({
|
|
198
|
+
...input.observation, phase: "contract/scope", scopeIds, prompt, userMessage: input.sessionOptions.userMessage, customTools,
|
|
199
|
+
artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${invocation}.json` : undefined,
|
|
200
|
+
committedCount: () => input.tools.committedFacts().length, durableCommittedCount: () => input.tools.committedFacts().length,
|
|
201
|
+
}, observer => input.piStepFn({ ...input.sessionOptions, prompt, onAttemptObservation: observer, writerToolPolicy: { requireSdk: true, customTools } }));
|
|
202
|
+
await input.tools.flush();
|
|
203
|
+
if (last.timedOut || /interrupt|termination-unconfirmed|budget_breach/.test(last.failureCategory))
|
|
204
|
+
return { ...last, ok: false };
|
|
205
|
+
const capacityExhausted = readWriterThinkingExhaustionEvidence(last).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(last.failureCategory);
|
|
206
|
+
if (capacityExhausted && !last.timedOut) {
|
|
207
|
+
const missing = scopeIds.filter(id => !input.tools.completedScopeRequirementIds().has(id));
|
|
208
|
+
if (!missing.length) {
|
|
209
|
+
if (finalScope && !terminal()) {
|
|
210
|
+
if (!scopeIds.length)
|
|
211
|
+
return { ...last, ok: false, failureCategory: "invalid-output", stderr: `${last.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: contract terminal exhausted; unchanged retry disabled` };
|
|
212
|
+
batches.push([]);
|
|
213
|
+
}
|
|
214
|
+
break;
|
|
215
|
+
}
|
|
216
|
+
if (missing.length === 1 && scopeIds.length > 1) {
|
|
217
|
+
batches.splice(index, 1, inventory.payload.requirements.filter(r => missing.includes(r.id)));
|
|
218
|
+
index -= 1;
|
|
219
|
+
break;
|
|
220
|
+
}
|
|
221
|
+
if (missing.length <= 1)
|
|
222
|
+
return { ...last, ok: false, stderr: `${last.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${missing[0] ?? "contract terminal"}; atom could not complete; refine the source without dropping conditions` };
|
|
223
|
+
const smaller = packFrontendInputUnits(inventory.payload.requirements.filter(r => missing.includes(r.id)), { maxUnits: Math.ceil(missing.length / 2) });
|
|
224
|
+
batches.splice(index, 1, ...smaller);
|
|
225
|
+
index -= 1;
|
|
226
|
+
break;
|
|
227
|
+
}
|
|
228
|
+
// A session that ends with blank assistant text but committed new
|
|
229
|
+
// facts is not a node failure: small-output models legitimately
|
|
230
|
+
// stop after their tool calls. Continue so the scope/repair checks
|
|
231
|
+
// below decide, instead of burning a full node retry.
|
|
232
|
+
const committedFactsOnlySuccess = !(last.assistantText ?? "").trim() &&
|
|
233
|
+
!last.stderr.trim() &&
|
|
234
|
+
input.tools.committedFacts().length > committedBefore;
|
|
235
|
+
if (!last.ok && !committedFactsOnlySuccess)
|
|
236
|
+
return last;
|
|
237
|
+
if (committedFactsOnlySuccess)
|
|
238
|
+
last = { ...last, ok: true, failureCategory: "success" };
|
|
239
|
+
const missing = scopeIds.filter(id => !input.tools.completedScopeRequirementIds().has(id));
|
|
240
|
+
if (!missing.length && (!finalScope || terminal()))
|
|
241
|
+
break;
|
|
242
|
+
if (repair === 1)
|
|
243
|
+
return { ...last, ok: false, failureCategory: "invalid-output", stderr: `CONTRACT_SCOPE_INCOMPLETE: ${missing.join(", ") || "missing finalize_contract"}` };
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
return last;
|
|
247
|
+
}
|
|
248
|
+
finally {
|
|
249
|
+
input.tools.setActiveRequirementScope(null);
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
export async function runFrontendPlanSegmentedSessions(input) {
|
|
253
|
+
const queue = [];
|
|
254
|
+
const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
|
|
255
|
+
const uxRegistrySegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-registry");
|
|
256
|
+
const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
|
|
257
|
+
const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
|
|
258
|
+
const allRequirementIds = input.requirementIds ?? [];
|
|
259
|
+
const buildPhasePrompt = (segment, missing = [], scopeIds = allRequirementIds) => {
|
|
260
|
+
const compact = allRequirementIds.length > 0
|
|
261
|
+
? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, scopeIds, {
|
|
262
|
+
includeRequirementText: segment.id === "global-mock-data" || segment.id === "finalize",
|
|
263
|
+
includeVerificationTargets: false,
|
|
264
|
+
includeDesignEvidence: segment.id === "global-dependency-deviation" || segment.id === "finalize",
|
|
265
|
+
includeChecklist: false,
|
|
266
|
+
})
|
|
267
|
+
: input.basePrompt;
|
|
268
|
+
const kinds = segment.id === "global-route"
|
|
269
|
+
? ["target-surface"]
|
|
270
|
+
: segment.id === "global-mock-data"
|
|
271
|
+
? ["plan-requirement", "plan-verification-target", "state-flow", "data-flow", "mock-api"]
|
|
272
|
+
: segment.id === "global-dependency-deviation"
|
|
273
|
+
? ["dependency", "design-deviation"]
|
|
274
|
+
: ["plan-requirement", "plan-verification-target", "state-registry", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
|
|
275
|
+
const ledger = input.committedFacts
|
|
276
|
+
? compactFrontendPlanLedgerContext({
|
|
277
|
+
committedFacts: input.committedFacts(),
|
|
278
|
+
requirementIds: scopeIds,
|
|
279
|
+
kinds,
|
|
280
|
+
...(segment.id === "global-mock-data" ? { scopedKinds: ["state-flow"] } : {}),
|
|
281
|
+
})
|
|
282
|
+
: "";
|
|
283
|
+
return [
|
|
284
|
+
compact,
|
|
285
|
+
segment.instruction,
|
|
286
|
+
ledger,
|
|
287
|
+
...(missing.length > 0
|
|
288
|
+
? [
|
|
289
|
+
"MISSING-FACT QUEUE: repair ONLY these items, then re-check the phase:",
|
|
290
|
+
...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
|
|
291
|
+
]
|
|
292
|
+
: []),
|
|
293
|
+
].filter(Boolean).join("\n\n");
|
|
294
|
+
};
|
|
295
|
+
const buildCoveragePrompt = (slice, missing = []) => [
|
|
296
|
+
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
|
|
297
|
+
coverageSegment.instruction,
|
|
298
|
+
`COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them. Finish the whole list before concluding: commit every listed requirement's coverage facts (verification targets or an evidence gap), batching up to 4 record_* calls per message; an early stop re-queues the remainder as a MISSING-FACT repair session.`,
|
|
299
|
+
...(missing.length > 0
|
|
300
|
+
? [
|
|
301
|
+
"MISSING-FACT QUEUE: the previous session did not establish complete coverage. Repair ONLY these items, then re-check the slice:",
|
|
302
|
+
...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
|
|
303
|
+
]
|
|
304
|
+
: []),
|
|
305
|
+
].join("\n\n");
|
|
306
|
+
const buildUxPrompt = (slice, missing = []) => {
|
|
307
|
+
const ledger = input.committedFacts
|
|
308
|
+
? compactFrontendPlanLedgerContext({
|
|
309
|
+
committedFacts: input.committedFacts(),
|
|
310
|
+
requirementIds: slice,
|
|
311
|
+
kinds: [
|
|
312
|
+
"plan-requirement",
|
|
313
|
+
"plan-verification-target",
|
|
314
|
+
"state-registry",
|
|
315
|
+
"component-choice",
|
|
316
|
+
"state-flow",
|
|
317
|
+
],
|
|
318
|
+
// Include shared state bindings from other scopes; page full values before correcting.
|
|
319
|
+
})
|
|
320
|
+
: "";
|
|
321
|
+
return [
|
|
322
|
+
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
|
|
323
|
+
uxSegment.instruction,
|
|
324
|
+
`UX SCOPE: process these complete behavior groups together: ${slice.join(", ")}. Reuse shared registry/component ownership; do not rename a behavior by requirement id. Constraint/exclusion groups do not create UI; preserve genuine verification gaps.`,
|
|
325
|
+
ledger,
|
|
326
|
+
...(missing.length > 0
|
|
327
|
+
? [
|
|
328
|
+
"MISSING-FACT QUEUE: repair ONLY these items, then re-check this UX slice:",
|
|
329
|
+
...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
|
|
330
|
+
]
|
|
331
|
+
: []),
|
|
332
|
+
].filter(Boolean).join("\n\n");
|
|
333
|
+
};
|
|
334
|
+
const buildUxRegistryPrompt = (missing = []) => {
|
|
335
|
+
const ledger = input.committedFacts
|
|
336
|
+
? compactFrontendPlanLedgerContext({
|
|
337
|
+
committedFacts: input.committedFacts(),
|
|
338
|
+
requirementIds: allRequirementIds,
|
|
339
|
+
kinds: [
|
|
340
|
+
"plan-requirement",
|
|
341
|
+
"plan-verification-target",
|
|
342
|
+
"state-registry",
|
|
343
|
+
],
|
|
344
|
+
})
|
|
345
|
+
: "";
|
|
346
|
+
const compact = allRequirementIds.length > 0
|
|
347
|
+
? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds, {
|
|
348
|
+
includeRequirementText: false,
|
|
349
|
+
includeVerificationTargets: false,
|
|
350
|
+
includeDesignEvidence: false,
|
|
351
|
+
includeChecklist: false,
|
|
352
|
+
})
|
|
353
|
+
: input.basePrompt;
|
|
354
|
+
return [
|
|
355
|
+
compact,
|
|
356
|
+
uxRegistrySegment.instruction,
|
|
357
|
+
ledger,
|
|
358
|
+
...(missing.length > 0
|
|
359
|
+
? [
|
|
360
|
+
"MISSING-FACT QUEUE: commit the global UX registry, then re-check this phase:",
|
|
361
|
+
...missing.map((item) => `- ${item.reason}`),
|
|
362
|
+
]
|
|
363
|
+
: []),
|
|
364
|
+
].filter(Boolean).join("\n\n");
|
|
365
|
+
};
|
|
366
|
+
const buildCompactLocalPrompt = (missing = []) => {
|
|
367
|
+
const ledger = input.committedFacts
|
|
368
|
+
? compactFrontendPlanLedgerContext({
|
|
369
|
+
committedFacts: input.committedFacts(),
|
|
370
|
+
requirementIds: allRequirementIds,
|
|
371
|
+
kinds: [
|
|
372
|
+
"plan-requirement",
|
|
373
|
+
"plan-verification-target",
|
|
374
|
+
"state-registry",
|
|
375
|
+
"component-choice",
|
|
376
|
+
"state-flow",
|
|
377
|
+
],
|
|
378
|
+
})
|
|
379
|
+
: "";
|
|
380
|
+
return [
|
|
381
|
+
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds),
|
|
382
|
+
"PLAN PHASE — compact local planning for a small frontend request.",
|
|
383
|
+
"Review all listed requirements together and record_state_registry first with one global UX vocabulary. Then record every plan-requirement and verification-target fact, followed by component/state-flow facts and the minimal route, Mock/data, dependency, and design-deviation policy facts needed by the observable behavior. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
|
|
384
|
+
"TOOL-FIRST: your first assistant actions must be record_* tool calls, at most 2-3 facts per message. Do not draft the whole analysis before recording; if a fact is uncertain, record it with an evidence gap instead of reasoning longer.",
|
|
385
|
+
ledger,
|
|
386
|
+
...(missing.length > 0
|
|
387
|
+
? [
|
|
388
|
+
"MISSING-FACT QUEUE: repair ONLY these items, then re-check the compact local phase:",
|
|
389
|
+
...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
|
|
390
|
+
]
|
|
391
|
+
: []),
|
|
392
|
+
].filter(Boolean).join("\n\n");
|
|
393
|
+
};
|
|
394
|
+
const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
|
|
395
|
+
? {
|
|
396
|
+
...r,
|
|
397
|
+
failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
398
|
+
stderr: `${r.stderr}\n${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before the next typed fact; preserve committed facts and retry only the unfinished phase`.trim(),
|
|
399
|
+
}
|
|
400
|
+
: r;
|
|
401
|
+
// An empty list means "ledger unreadable / unknown" and falls back to one
|
|
402
|
+
// unscoped coverage session; a non-empty list enables deterministic sharding.
|
|
403
|
+
const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
|
|
404
|
+
const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
405
|
+
const initialMissing = requirementIdsProvided && input.committedFacts
|
|
406
|
+
? collectFrontendPlanMissingFacts({
|
|
407
|
+
requirementIds: input.requirementIds,
|
|
408
|
+
committedFacts: input.committedFacts(),
|
|
409
|
+
})
|
|
410
|
+
: [];
|
|
411
|
+
const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
|
|
412
|
+
const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
|
|
413
|
+
if (input.parallelCoverageOnly === true &&
|
|
414
|
+
requirementIdsProvided &&
|
|
415
|
+
coverageWorkIds.length === 0) {
|
|
416
|
+
// A retry may reopen a shard whose committed facts are already complete.
|
|
417
|
+
// Treat that shard as an idempotent no-op; returning the normal empty
|
|
418
|
+
// session failure would make a partially failed map impossible to resume.
|
|
419
|
+
return {
|
|
420
|
+
ok: true,
|
|
421
|
+
assistantText: "",
|
|
422
|
+
command: [],
|
|
423
|
+
durationMs: 0,
|
|
424
|
+
exitCode: 0,
|
|
425
|
+
failureCategory: "success",
|
|
426
|
+
modelDisplay: "reused-coverage-facts",
|
|
427
|
+
parsedEvents: 0,
|
|
428
|
+
stderr: "",
|
|
429
|
+
stdout: "",
|
|
430
|
+
timedOut: false,
|
|
431
|
+
attemptedModels: [],
|
|
432
|
+
fallbackUsed: false,
|
|
433
|
+
tokensUsed: 0,
|
|
434
|
+
};
|
|
435
|
+
}
|
|
436
|
+
const compiledInput = parseFrontendInputBlock(input.basePrompt, "plan")?.payload;
|
|
437
|
+
const fullUnits = new Map(compiledInput?.requirements.map(r => [r.id, r]) ?? []);
|
|
438
|
+
const workCost = (ids) => 1 + ids.reduce((total, id) => total + Math.max(1, (input.requirementCosts?.get(id) ?? 2) - 1), 0);
|
|
439
|
+
const { workGroups, buildWorkBatches, compactEligible } = buildFrontendPlanWorkload({
|
|
440
|
+
basePrompt: input.basePrompt, requirementIds: allRequirementIds,
|
|
441
|
+
requirementCosts: input.requirementCosts, sessionOptions: input.sessionOptions,
|
|
442
|
+
});
|
|
443
|
+
// Data decisions are grouped by observable data domain. Requirements that
|
|
444
|
+
// mention the same endpoint/resource or execution group share one session;
|
|
445
|
+
// unrelated domains remain isolated. This prevents the old requirement-by-
|
|
446
|
+
// requirement repetition while preserving the packer's size bound.
|
|
447
|
+
const buildDataBatches = (ids) => {
|
|
448
|
+
const selected = new Set(ids);
|
|
449
|
+
const domains = new Map();
|
|
450
|
+
for (const [index, group] of workGroups.entries()) {
|
|
451
|
+
const members = group.requirementIds.filter((id) => selected.has(id));
|
|
452
|
+
if (!members.length)
|
|
453
|
+
continue;
|
|
454
|
+
const texts = members.map((id) => String(fullUnits.get(id)?.text ?? ""));
|
|
455
|
+
const endpoint = texts
|
|
456
|
+
.map((text) => text.match(/\b(?:GET|POST|PUT|PATCH|DELETE)\s+(\/[^\s,;.)]+)/i)?.[1])
|
|
457
|
+
.find(Boolean);
|
|
458
|
+
const resource = endpoint
|
|
459
|
+
? endpoint.split("/").filter(Boolean).slice(0, 2).join("/")
|
|
460
|
+
: undefined;
|
|
461
|
+
const domain = resource ? `endpoint:${resource}` : `group:${group.id}`;
|
|
462
|
+
const current = domains.get(domain);
|
|
463
|
+
if (current) {
|
|
464
|
+
current.requirementIds.push(...members);
|
|
465
|
+
current.requirements.push(...members.map((id) => fullUnits.get(id) ?? { id }));
|
|
466
|
+
current.estimatedCalls = workCost(current.requirementIds);
|
|
467
|
+
}
|
|
468
|
+
else {
|
|
469
|
+
domains.set(domain, {
|
|
470
|
+
id: `${index}:${domain}`,
|
|
471
|
+
requirementIds: [...members],
|
|
472
|
+
requirements: members.map((id) => fullUnits.get(id) ?? { id }),
|
|
473
|
+
estimatedCalls: workCost(members),
|
|
474
|
+
});
|
|
475
|
+
}
|
|
476
|
+
}
|
|
477
|
+
const work = [...domains.values()];
|
|
478
|
+
// Work-unit packing and full provider envelope capacity have separate budgets.
|
|
479
|
+
return packFrontendInputUnits(work, {
|
|
480
|
+
targetBytes: input.sessionOptions.frontendExecutionPolicy?.scopeTargetBytes ?? FRONTEND_SCOPE_TARGET_BYTES,
|
|
481
|
+
maxUnits: input.sessionOptions.frontendExecutionPolicy?.maxScopeUnits ?? 4,
|
|
482
|
+
maxCost: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS,
|
|
483
|
+
cost: (group) => group.estimatedCalls,
|
|
484
|
+
}).map((batch) => batch.flatMap((group) => group.requirementIds));
|
|
485
|
+
};
|
|
486
|
+
// Dependency/deviation is optional policy. For a pure local UI request with
|
|
487
|
+
// no dependency, package, design-conflict, or OpenSpec signal, opening a
|
|
488
|
+
// dedicated model session only produces an empty policy fact. Keep the
|
|
489
|
+
// session when the prompt or ledger contains any such signal so this is a
|
|
490
|
+
// conservative skip, not a blanket removal of the gate.
|
|
491
|
+
const dependencyDeviationNeeded = /(?:dependenc|package\.json|npm\s+(?:install|i)|yarn\s+add|pnpm\s+add|openspec|design\s+conflict|规范冲突|依赖)/i.test(input.basePrompt) ||
|
|
492
|
+
Boolean(input.committedFacts?.().some((record) => {
|
|
493
|
+
const fact = committedFactFromPlanRecord(record);
|
|
494
|
+
return fact?.kind === "dependency" || fact?.kind === "design-deviation";
|
|
495
|
+
}));
|
|
496
|
+
const globalPolicyToolNames = new Set([
|
|
497
|
+
"record_route_selection",
|
|
498
|
+
"record_dependency",
|
|
499
|
+
"record_design_deviation",
|
|
500
|
+
"adopt_staged_fact",
|
|
501
|
+
]);
|
|
502
|
+
const globalPolicyPrompt = [
|
|
503
|
+
"PLAN PHASE — global implementation policy.",
|
|
504
|
+
"Use the Scout target surface to record the selected route(s), then record dependency policy and design-evidence conflicts only when they are relevant. Do not record requirement-local UX or Mock/data facts. Do not call finalize_plan.",
|
|
505
|
+
dependencyDeviationNeeded
|
|
506
|
+
? "Dependency/design policy signals are present; inspect them and commit the minimal policy facts needed."
|
|
507
|
+
: "No dependency/design-conflict signal was found; do not invent a policy fact.",
|
|
508
|
+
].join(" ");
|
|
509
|
+
let globalPolicyQueued = false;
|
|
510
|
+
// Small, single-surface requests do not benefit from six isolated Pi
|
|
511
|
+
// sessions. Keep the typed ledger as the authority, but let one local
|
|
512
|
+
// session establish requirement/UX facts and one final session establish
|
|
513
|
+
// cross-cutting policy + finalize. The old sharded ladder remains available
|
|
514
|
+
// for larger plans and for the unscoped compatibility path.
|
|
515
|
+
const useCompactSmallPlan = requirementIdsProvided &&
|
|
516
|
+
input.compactSmallPlan === true &&
|
|
517
|
+
compactEligible;
|
|
518
|
+
if (useCompactSmallPlan) {
|
|
519
|
+
const compactLocalTools = new Set([
|
|
520
|
+
"record_plan_requirement",
|
|
521
|
+
"record_plan_group_coverage",
|
|
522
|
+
"record_plan_verification_target",
|
|
523
|
+
"record_plan_evidence_gap",
|
|
524
|
+
"record_state_registry",
|
|
525
|
+
"record_component_choice",
|
|
526
|
+
"record_state_flow",
|
|
527
|
+
"record_route_selection",
|
|
528
|
+
"record_data_flow",
|
|
529
|
+
"record_mock_api",
|
|
530
|
+
"record_mock_endpoint",
|
|
531
|
+
"record_dependency",
|
|
532
|
+
"record_design_deviation",
|
|
533
|
+
"adopt_staged_fact",
|
|
534
|
+
]);
|
|
535
|
+
queue.push({
|
|
536
|
+
id: "compact-local",
|
|
537
|
+
toolNames: compactLocalTools,
|
|
538
|
+
requirementSlice: [...allRequirementIds],
|
|
539
|
+
prompt: buildCompactLocalPrompt(),
|
|
540
|
+
});
|
|
541
|
+
}
|
|
542
|
+
else if (!requirementIdsProvided) {
|
|
543
|
+
queue.push({
|
|
544
|
+
id: "coverage",
|
|
545
|
+
toolNames: coverageSegment.toolNames,
|
|
546
|
+
prompt: `${input.basePrompt}\n\n${coverageSegment.instruction}`,
|
|
547
|
+
});
|
|
548
|
+
}
|
|
549
|
+
else if (coverageWorkIds.length > 0) {
|
|
550
|
+
const completeBatches = buildWorkBatches(coverageWorkIds);
|
|
551
|
+
completeBatches.forEach((slice, batchIndex) => queue.push({
|
|
552
|
+
id: `coverage-batch-${batchIndex + 1}`,
|
|
553
|
+
toolNames: coverageSegment.toolNames,
|
|
554
|
+
coverageSlice: slice,
|
|
555
|
+
coverageOnly: true,
|
|
556
|
+
prompt: buildCoveragePrompt(slice),
|
|
557
|
+
}));
|
|
558
|
+
}
|
|
559
|
+
if (useCompactSmallPlan) {
|
|
560
|
+
// Compact mode already queued both sessions above.
|
|
561
|
+
}
|
|
562
|
+
else if (!input.parallelCoverageOnly) {
|
|
563
|
+
if (requirementIdsProvided) {
|
|
564
|
+
queue.push({
|
|
565
|
+
id: uxRegistrySegment.id,
|
|
566
|
+
toolNames: uxRegistrySegment.toolNames,
|
|
567
|
+
prompt: buildUxRegistryPrompt(),
|
|
568
|
+
});
|
|
569
|
+
}
|
|
570
|
+
const uxWorkIds = workGroups.filter(g => !["constraint", "exclusion"].includes(g.kind) || g.requirementIds.some(id => input.behaviorRequiredRequirementIds?.includes(id))).flatMap(g => g.requirementIds);
|
|
571
|
+
const uxBatches = requirementIdsProvided ? buildWorkBatches(uxWorkIds) : [[]];
|
|
572
|
+
uxBatches.forEach((slice, batchIndex) => queue.push({
|
|
573
|
+
id: `ux-local-${batchIndex + 1}`,
|
|
574
|
+
toolNames: requirementIdsProvided ? uxSegment.toolNames : new Set([...(uxSegment.toolNames ?? []), "record_state_registry"]),
|
|
575
|
+
...(slice.length ? { requirementSlice: slice } : {}),
|
|
576
|
+
prompt: slice.length ? buildUxPrompt(slice) : buildPhasePrompt(uxSegment),
|
|
577
|
+
}));
|
|
578
|
+
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
579
|
+
if (["coverage", "ux-registry", "ux-local"].includes(segment.id))
|
|
580
|
+
continue;
|
|
581
|
+
if (segment.id === "global-dependency-deviation" && !dependencyDeviationNeeded)
|
|
582
|
+
continue;
|
|
583
|
+
if (segment.id === "global-route") {
|
|
584
|
+
if (globalPolicyQueued)
|
|
585
|
+
continue;
|
|
586
|
+
globalPolicyQueued = true;
|
|
587
|
+
queue.push({
|
|
588
|
+
id: "global-policy",
|
|
589
|
+
toolNames: globalPolicyToolNames,
|
|
590
|
+
prompt: `${buildPhasePrompt(segment)}\n\n${globalPolicyPrompt}`,
|
|
591
|
+
});
|
|
592
|
+
continue;
|
|
593
|
+
}
|
|
594
|
+
if (segment.id === "global-dependency-deviation")
|
|
595
|
+
continue;
|
|
596
|
+
if (segment.id === "global-mock-data" && requirementIdsProvided) {
|
|
597
|
+
buildDataBatches(allRequirementIds).forEach((slice, i) => queue.push({ id: `global-mock-data-${i + 1}`, toolNames: segment.toolNames, requirementSlice: slice, prompt: buildPhasePrompt(segment, [], slice) }));
|
|
598
|
+
}
|
|
599
|
+
else {
|
|
600
|
+
queue.push({
|
|
601
|
+
id: segment.id,
|
|
602
|
+
toolNames: segment.toolNames,
|
|
603
|
+
prompt: buildPhasePrompt(segment),
|
|
604
|
+
});
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
let accumulated;
|
|
609
|
+
let index = 0;
|
|
610
|
+
let invocationCount = 0;
|
|
611
|
+
let lastDurableCount = input.committedFactCount();
|
|
612
|
+
while (index < queue.length) {
|
|
613
|
+
const session = queue[index];
|
|
614
|
+
if (!session)
|
|
615
|
+
break;
|
|
616
|
+
if (input.attempt > 1 && input.committedFacts && session.id === "ux-registry" &&
|
|
617
|
+
collectFrontendPlanPhaseMissingFacts({ phase: "ux-registry", requirementIds: allRequirementIds, committedFacts: input.committedFacts() }).length === 0) {
|
|
618
|
+
index += 1;
|
|
619
|
+
continue;
|
|
620
|
+
}
|
|
621
|
+
// Reuse complete UX scopes on node retry. Require explicit per-requirement
|
|
622
|
+
// ownership; an unrelated/global fact must not prove a slice complete.
|
|
623
|
+
if (input.attempt > 1 && input.committedFacts && session.id.startsWith("ux-local-") && session.requirementSlice?.length) {
|
|
624
|
+
const facts = input.committedFacts();
|
|
625
|
+
const ownedIds = new Set(facts.flatMap(value => {
|
|
626
|
+
const fact = committedFactFromPlanRecord(value);
|
|
627
|
+
return fact ? planFactStringList(fact.scopeRequirementIds) : [];
|
|
628
|
+
}));
|
|
629
|
+
const complete = session.requirementSlice.every(id => ownedIds.has(id)) &&
|
|
630
|
+
collectFrontendPlanPhaseMissingFacts({ phase: "ux-local", requirementIds: session.requirementSlice,
|
|
631
|
+
committedFacts: facts, behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds }).length === 0;
|
|
632
|
+
if (complete) {
|
|
633
|
+
index += 1;
|
|
634
|
+
continue;
|
|
635
|
+
}
|
|
636
|
+
}
|
|
637
|
+
const remaining = (session.coverageOnly ? session.coverageSlice ?? [] : []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
638
|
+
const preexistingMissing = session.coverageOnly && session.coverageSlice && input.committedFacts
|
|
639
|
+
? collectFrontendPlanMissingFacts({
|
|
640
|
+
requirementIds: session.coverageSlice,
|
|
641
|
+
committedFacts: input.committedFacts(),
|
|
642
|
+
})
|
|
643
|
+
: [];
|
|
644
|
+
// Resume/earlier-batch commits may already cover this slice.
|
|
645
|
+
if (session.coverageOnly &&
|
|
646
|
+
session.coverageSlice &&
|
|
647
|
+
remaining.length === 0 &&
|
|
648
|
+
preexistingMissing.length === 0) {
|
|
649
|
+
index += 1;
|
|
650
|
+
continue;
|
|
651
|
+
}
|
|
652
|
+
// Resume only unfinished UX scopes. The registry and local decisions are
|
|
653
|
+
// durable facts; replaying a completed scope wastes a model session and
|
|
654
|
+
// can make a previously valid shared decision look like a duplicate.
|
|
655
|
+
if (session.id === "ux-registry" &&
|
|
656
|
+
input.committedFacts &&
|
|
657
|
+
!collectFrontendPlanPhaseMissingFacts({
|
|
658
|
+
phase: "ux-registry",
|
|
659
|
+
requirementIds: allRequirementIds,
|
|
660
|
+
committedFacts: input.committedFacts(),
|
|
661
|
+
}).length) {
|
|
662
|
+
index += 1;
|
|
663
|
+
continue;
|
|
664
|
+
}
|
|
665
|
+
if (session.id.startsWith("ux-local-") &&
|
|
666
|
+
session.requirementSlice &&
|
|
667
|
+
(input.behaviorRequiredRequirementIds?.length ?? 0) > 0 &&
|
|
668
|
+
input.committedFacts &&
|
|
669
|
+
!collectFrontendPlanPhaseMissingFacts({
|
|
670
|
+
phase: "ux-local",
|
|
671
|
+
requirementIds: session.requirementSlice,
|
|
672
|
+
committedFacts: input.committedFacts(),
|
|
673
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
674
|
+
}).length) {
|
|
675
|
+
index += 1;
|
|
676
|
+
continue;
|
|
677
|
+
}
|
|
678
|
+
let prompt = session.prompt;
|
|
679
|
+
if (session.coverageOnly && session.coverageSlice) {
|
|
680
|
+
const promptSlice = [...new Set([...remaining, ...preexistingMissing.flatMap(item => item.requirementIds)])];
|
|
681
|
+
prompt = buildCoveragePrompt(promptSlice, session.missingFacts ?? preexistingMissing);
|
|
682
|
+
}
|
|
683
|
+
else if (session.id === "compact-local") {
|
|
684
|
+
prompt = buildCompactLocalPrompt(session.missingFacts);
|
|
685
|
+
}
|
|
686
|
+
else if (session.id === "ux-registry") {
|
|
687
|
+
prompt = buildUxRegistryPrompt(session.missingFacts);
|
|
688
|
+
}
|
|
689
|
+
else if (session.id.startsWith("ux-local-")) {
|
|
690
|
+
// Build this at execution time: coverage facts are committed by the
|
|
691
|
+
// preceding sessions and must be visible to the UX-local model.
|
|
692
|
+
prompt = session.requirementSlice
|
|
693
|
+
? buildUxPrompt(session.requirementSlice, session.missingFacts)
|
|
694
|
+
: buildPhasePrompt(uxSegment, session.missingFacts);
|
|
695
|
+
}
|
|
696
|
+
else {
|
|
697
|
+
// Global phases consume the latest committed ledger;
|
|
698
|
+
// constructing their prompt only when the session starts prevents a
|
|
699
|
+
// stale queue entry from dropping facts written by earlier phases.
|
|
700
|
+
const segment = FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === session.id || (candidate.id === "global-mock-data" && session.id.startsWith("global-mock-data-")));
|
|
701
|
+
if (session.id === "global-policy") {
|
|
702
|
+
const routePrompt = buildPhasePrompt(FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === "global-route"));
|
|
703
|
+
prompt = `${routePrompt}\n\n${globalPolicyPrompt}`;
|
|
704
|
+
}
|
|
705
|
+
else if (segment) {
|
|
706
|
+
prompt = buildPhasePrompt(segment, session.missingFacts, session.requirementSlice ?? allRequirementIds);
|
|
707
|
+
}
|
|
708
|
+
}
|
|
709
|
+
const atomicFocus = session.atomicRecovery ? session.missingFacts?.[0] : undefined;
|
|
710
|
+
if (atomicFocus) {
|
|
711
|
+
prompt = [
|
|
712
|
+
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, atomicFocus.requirementIds, {
|
|
713
|
+
includeChecklist: false,
|
|
714
|
+
...(atomicFocus.kind === "plan-verification-target" && atomicFocus.id
|
|
715
|
+
? { verificationTargetIds: [atomicFocus.id] } : {}),
|
|
716
|
+
}),
|
|
717
|
+
`ATOMIC FACT: ${atomicFocus.kind}${atomicFocus.id ? ` ${atomicFocus.id}` : ""}`,
|
|
718
|
+
atomicFocus.reason,
|
|
719
|
+
"Commit ONLY this missing fact with one record_* call, then end this session. Other facts are queued separately. Preserve canonical IDs and shared bindings; use read_plan_facts for existing values. Do not summarize or plan the entire requirement. Capacity exhaustion is not evidence of a requirement gap.",
|
|
720
|
+
].join("\n\n");
|
|
721
|
+
}
|
|
722
|
+
const committedBefore = input.committedFactCount();
|
|
723
|
+
input.setActiveRequirementScope?.(session.coverageOnly ? session.coverageSlice ?? [] : session.requirementSlice ?? []);
|
|
724
|
+
if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
|
|
725
|
+
break;
|
|
726
|
+
invocationCount += 1;
|
|
727
|
+
const atomicTools = {
|
|
728
|
+
"plan-requirement": "record_plan_requirement", "plan-verification-target": "record_plan_verification_target",
|
|
729
|
+
"state-registry": "record_state_registry", "component-choice": "record_component_choice",
|
|
730
|
+
"state-flow": "record_state_flow", "data-flow": "record_data_flow", "mock-api": "record_mock_api",
|
|
731
|
+
};
|
|
732
|
+
const customTools = input.segmentCustomTools(atomicFocus ? new Set([atomicTools[atomicFocus.kind]]) : session.toolNames);
|
|
733
|
+
const result = await observeFrontendSession({
|
|
734
|
+
...input.observation, phase: `plan/${session.id}`, dispatchReason: (session.retryCount ?? 0) > 0 || !!session.missingFacts?.length || /capacity|recovery|split/.test(session.id) ? "correction" : "initial", scopeIds: session.requirementSlice ?? session.coverageSlice ?? allRequirementIds,
|
|
735
|
+
prompt, userMessage: input.sessionOptions.userMessage, customTools, committedCount: input.committedFactCount, durableCommittedCount: () => lastDurableCount,
|
|
736
|
+
artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-${invocationCount}.json` : undefined,
|
|
737
|
+
}, async (observer) => {
|
|
738
|
+
const result = await input.piStepFn({
|
|
739
|
+
...input.sessionOptions,
|
|
740
|
+
onAttemptObservation: observer,
|
|
741
|
+
prompt,
|
|
742
|
+
...(customTools.length > 0
|
|
743
|
+
? {
|
|
744
|
+
writerToolPolicy: {
|
|
745
|
+
requireSdk: true,
|
|
746
|
+
customTools,
|
|
747
|
+
},
|
|
748
|
+
}
|
|
749
|
+
: {}),
|
|
750
|
+
});
|
|
751
|
+
try {
|
|
752
|
+
await input.flushLedger();
|
|
753
|
+
lastDurableCount = input.committedFactCount();
|
|
754
|
+
}
|
|
755
|
+
catch {
|
|
756
|
+
// best-effort: the node-level flush runs again after the attempt
|
|
757
|
+
}
|
|
758
|
+
return result;
|
|
759
|
+
});
|
|
760
|
+
accumulated = accumulated
|
|
761
|
+
? combineSequentialPiResults(accumulated, result)
|
|
762
|
+
: result;
|
|
763
|
+
const committedAfter = input.committedFactCount();
|
|
764
|
+
const committedFactsOnlySuccess = !(result.assistantText ?? "").trim() &&
|
|
765
|
+
!result.stderr.trim() &&
|
|
766
|
+
!result.timedOut &&
|
|
767
|
+
committedAfter > committedBefore;
|
|
768
|
+
const missingCoverage = session.coverageOnly && session.coverageSlice && input.committedFacts
|
|
769
|
+
? collectFrontendPlanMissingFacts({
|
|
770
|
+
requirementIds: session.coverageSlice,
|
|
771
|
+
committedFacts: input.committedFacts(),
|
|
772
|
+
})
|
|
773
|
+
: [];
|
|
774
|
+
const isCompactLocalSession = session.id === "compact-local";
|
|
775
|
+
const isUxRegistrySession = session.id === "ux-registry";
|
|
776
|
+
const isUxLocalSession = session.id.startsWith("ux-local-");
|
|
777
|
+
const missingPhase = isCompactLocalSession && input.committedFacts
|
|
778
|
+
? [
|
|
779
|
+
...collectFrontendPlanMissingFacts({
|
|
780
|
+
requirementIds: allRequirementIds,
|
|
781
|
+
committedFacts: input.committedFacts(),
|
|
782
|
+
}),
|
|
783
|
+
...collectFrontendPlanPhaseMissingFacts({
|
|
784
|
+
phase: "ux-registry",
|
|
785
|
+
requirementIds: allRequirementIds,
|
|
786
|
+
committedFacts: input.committedFacts(),
|
|
787
|
+
}),
|
|
788
|
+
...collectFrontendPlanPhaseMissingFacts({
|
|
789
|
+
phase: "ux-local",
|
|
790
|
+
requirementIds: allRequirementIds,
|
|
791
|
+
committedFacts: input.committedFacts(),
|
|
792
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
793
|
+
}),
|
|
794
|
+
]
|
|
795
|
+
: isUxRegistrySession && input.committedFacts
|
|
796
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
797
|
+
phase: "ux-registry",
|
|
798
|
+
requirementIds: allRequirementIds,
|
|
799
|
+
committedFacts: input.committedFacts(),
|
|
800
|
+
})
|
|
801
|
+
: isUxLocalSession && session.requirementSlice && input.committedFacts
|
|
802
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
803
|
+
phase: "ux-local",
|
|
804
|
+
requirementIds: session.requirementSlice,
|
|
805
|
+
committedFacts: input.committedFacts(),
|
|
806
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
807
|
+
})
|
|
808
|
+
: session.id.startsWith("global-mock-data") && allRequirementIds.length > 0 && input.committedFacts
|
|
809
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
810
|
+
phase: "global-mock-data",
|
|
811
|
+
requirementIds: session.requirementSlice ?? allRequirementIds,
|
|
812
|
+
committedFacts: input.committedFacts(),
|
|
813
|
+
})
|
|
814
|
+
: [];
|
|
815
|
+
const missingPhaseFacts = [...missingCoverage, ...missingPhase];
|
|
816
|
+
// Frozen verification commands must operate on files the planner has
|
|
817
|
+
// committed verification targets for; otherwise the admission writeSet
|
|
818
|
+
// cannot authorize the file and verify-shell deterministically fails.
|
|
819
|
+
const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.basePrompt);
|
|
820
|
+
if (verificationCommandFiles.length > 0 && allRequirementIds.length > 0 && input.committedFacts) {
|
|
821
|
+
const coveredVerificationFiles = new Set(input
|
|
822
|
+
.committedFacts()
|
|
823
|
+
.map(committedFactFromPlanRecord)
|
|
824
|
+
.filter((fact) => Boolean(fact && fact.origin === "plan" && fact.kind === "plan-verification-target"))
|
|
825
|
+
.map((fact) => fact.entry?.file)
|
|
826
|
+
.filter((file) => typeof file === "string"));
|
|
827
|
+
for (const file of verificationCommandFiles) {
|
|
828
|
+
if (coveredVerificationFiles.has(file))
|
|
829
|
+
continue;
|
|
830
|
+
missingPhaseFacts.push({
|
|
831
|
+
kind: "plan-verification-target",
|
|
832
|
+
id: file,
|
|
833
|
+
requirementIds: allRequirementIds.slice(0, 1),
|
|
834
|
+
reason: `frozen verification command references ${file} but no committed verification target covers it; record a static verification target for this file so it joins the writeSet`,
|
|
835
|
+
});
|
|
836
|
+
}
|
|
837
|
+
}
|
|
838
|
+
const promptForMissingPhase = (missing) => session.coverageOnly
|
|
839
|
+
? buildCoveragePrompt(session.coverageSlice ?? [], missing)
|
|
840
|
+
: isCompactLocalSession
|
|
841
|
+
? buildCompactLocalPrompt(missing)
|
|
842
|
+
: isUxRegistrySession
|
|
843
|
+
? buildUxRegistryPrompt(missing)
|
|
844
|
+
: isUxLocalSession
|
|
845
|
+
? buildUxPrompt(session.requirementSlice ?? [], missing)
|
|
846
|
+
: buildPhasePrompt(globalMockDataSegment, missing, session.requirementSlice ?? allRequirementIds);
|
|
847
|
+
const recovery = classifyFrontendPlanRecovery({ ...result, stopReason: readWriterThinkingExhaustionEvidence(result).stopReason });
|
|
848
|
+
if (recovery === "stop")
|
|
849
|
+
return { ...result, ok: false };
|
|
850
|
+
if ((recovery === "output" || (session.atomicRecovery && result.ok)) && input.committedFacts &&
|
|
851
|
+
!(missingPhaseFacts.length === 1 && missingPhaseFacts[0]?.kind === "plan-requirement") &&
|
|
852
|
+
(session.coverageOnly || isUxRegistrySession || isUxLocalSession || session.id.startsWith("global-mock-data"))) {
|
|
853
|
+
if (missingPhaseFacts.length === 0) {
|
|
854
|
+
// A complete validated slice does not need a successful prose turn.
|
|
855
|
+
// This never accepts finalize or bypasses the final contract validator.
|
|
856
|
+
accumulated = { ...accumulated, ok: true, failureCategory: "success" };
|
|
857
|
+
index += 1;
|
|
858
|
+
continue;
|
|
859
|
+
}
|
|
860
|
+
const missingIds = [...new Set(missingPhaseFacts.flatMap(item => item.requirementIds))];
|
|
861
|
+
const scope = session.coverageSlice ?? session.requirementSlice;
|
|
862
|
+
if (scope && missingIds.length < scope.length) {
|
|
863
|
+
queue[index] = { ...session, missingFacts: missingPhaseFacts,
|
|
864
|
+
...(session.coverageOnly ? { coverageSlice: missingIds } : { requirementSlice: missingIds }),
|
|
865
|
+
retryCount: 0 };
|
|
866
|
+
continue;
|
|
867
|
+
}
|
|
868
|
+
if (scope && missingIds.length > 1 && (session.coverageOnly || isUxLocalSession)) {
|
|
869
|
+
const half = Math.ceil(missingIds.length / 2);
|
|
870
|
+
queue.splice(index, 1, ...[missingIds.slice(0, half), missingIds.slice(half)].map(ids => ({
|
|
871
|
+
...session, retryCount: 0,
|
|
872
|
+
...(session.coverageOnly ? { coverageSlice: ids } : { requirementSlice: ids }),
|
|
873
|
+
missingFacts: missingPhaseFacts.filter(item => item.requirementIds.some(id => ids.includes(id))),
|
|
874
|
+
})));
|
|
875
|
+
continue;
|
|
876
|
+
}
|
|
877
|
+
const focusStillMissing = atomicFocus && missingPhaseFacts.some(item => item.kind === atomicFocus.kind && item.id === atomicFocus.id &&
|
|
878
|
+
item.requirementIds.join("\0") === atomicFocus.requirementIds.join("\0"));
|
|
879
|
+
const retries = focusStillMissing ? (session.retryCount ?? 0) + 1 : 0;
|
|
880
|
+
if (retries < 2) {
|
|
881
|
+
queue[index] = { ...session, atomicRecovery: true, missingFacts: missingPhaseFacts, retryCount: retries };
|
|
882
|
+
continue;
|
|
883
|
+
}
|
|
884
|
+
return { ...accumulated, ok: false, failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
885
|
+
stderr: `${accumulated.stderr}\nfrontend plan capacity recovery exhausted: preserve committed facts; phase=${session.id}; scope=${missingIds.join(",")}; strategy=atomic-fact; missing=${JSON.stringify(missingPhaseFacts)}`.trim() };
|
|
886
|
+
}
|
|
887
|
+
const capacityExhausted = recovery === "output" || recovery === "context";
|
|
888
|
+
if (capacityExhausted && !result.timedOut) {
|
|
889
|
+
const failure = () => mapPlannerExhaustion({ ...result, ok: false, stderr: `${result.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${session.id}; no smaller complete scope can finish; unchanged retries are disabled` }, committedAfter > committedBefore);
|
|
890
|
+
const missingCoverageIds = input.committedFacts ? [...new Set(collectFrontendPlanMissingFacts({ requirementIds: session.coverageSlice ?? session.requirementSlice ?? allRequirementIds, committedFacts: input.committedFacts() }).flatMap(f => f.requirementIds))] : [...(session.coverageSlice ?? session.requirementSlice ?? allRequirementIds)];
|
|
891
|
+
const splitScope = (ids, allowMemberSplit) => {
|
|
892
|
+
const groups = workGroups.map(g => g.requirementIds.filter(id => ids.includes(id))).filter(g => g.length);
|
|
893
|
+
if (groups.length > 1) {
|
|
894
|
+
const half = Math.ceil(groups.length / 2);
|
|
895
|
+
return [groups.slice(0, half).flat(), groups.slice(half).flat()];
|
|
896
|
+
}
|
|
897
|
+
if (allowMemberSplit && ids.length > 1) {
|
|
898
|
+
const half = Math.ceil(ids.length / 2);
|
|
899
|
+
return [ids.slice(0, half), ids.slice(half)];
|
|
900
|
+
}
|
|
901
|
+
return [];
|
|
902
|
+
};
|
|
903
|
+
if (isCompactLocalSession) {
|
|
904
|
+
let scopes = splitScope(missingCoverageIds, true);
|
|
905
|
+
if (!scopes.length && missingCoverageIds.length) {
|
|
906
|
+
if (missingCoverageIds.length === allRequirementIds.length && committedAfter === committedBefore)
|
|
907
|
+
return failure();
|
|
908
|
+
scopes = [missingCoverageIds];
|
|
909
|
+
}
|
|
910
|
+
// UX ownership is independent of coverage completion. Preserve all groups.
|
|
911
|
+
const recovery = scopes.map((slice, i) => ({ id: `coverage-capacity-${invocationCount}-${i}`, coverageOnly: true, coverageSlice: slice, toolNames: coverageSegment.toolNames, prompt: buildCoveragePrompt(slice) }));
|
|
912
|
+
recovery.push({ id: "ux-registry", toolNames: uxRegistrySegment.toolNames, prompt: buildUxRegistryPrompt() });
|
|
913
|
+
recovery.push(...buildWorkBatches([...allRequirementIds]).map((slice, i) => ({ id: `ux-local-capacity-${invocationCount}-${i}`, requirementSlice: slice, toolNames: uxSegment.toolNames, prompt: buildUxPrompt(slice) })));
|
|
914
|
+
queue.splice(index, 1, ...recovery);
|
|
915
|
+
continue;
|
|
916
|
+
}
|
|
917
|
+
if (session.coverageOnly || isUxLocalSession) {
|
|
918
|
+
const missingIds = session.coverageOnly ? missingCoverageIds : [...new Set(missingPhase.flatMap(f => f.requirementIds))];
|
|
919
|
+
if (!missingIds.length) {
|
|
920
|
+
index += 1;
|
|
921
|
+
continue;
|
|
922
|
+
}
|
|
923
|
+
const scopes = splitScope(missingIds, Boolean(session.coverageOnly));
|
|
924
|
+
if (scopes.length) {
|
|
925
|
+
queue.splice(index, 1, ...scopes.map(slice => ({ ...session, ...(session.coverageOnly ? { coverageSlice: slice } : { requirementSlice: slice }), missingFacts: missingPhaseFacts.filter(f => f.requirementIds.some(id => slice.includes(id))), retryCount: 0 })));
|
|
926
|
+
continue;
|
|
927
|
+
}
|
|
928
|
+
const originalIds = session.coverageSlice ?? session.requirementSlice ?? [];
|
|
929
|
+
if ((missingIds.length < originalIds.length || committedAfter > committedBefore) && (session.retryCount ?? 0) < 2) {
|
|
930
|
+
queue[index] = { ...session, ...(session.coverageOnly ? { coverageSlice: missingIds } : { requirementSlice: missingIds }), missingFacts: missingPhaseFacts, retryCount: (session.retryCount ?? 0) + 1 };
|
|
931
|
+
continue;
|
|
932
|
+
}
|
|
933
|
+
}
|
|
934
|
+
return failure();
|
|
935
|
+
}
|
|
936
|
+
if (result.ok) {
|
|
937
|
+
if (missingPhaseFacts.length === 0) {
|
|
938
|
+
index += 1;
|
|
939
|
+
continue;
|
|
940
|
+
}
|
|
941
|
+
if ((session.retryCount ?? 0) < 2) {
|
|
942
|
+
queue[index] = {
|
|
943
|
+
...session,
|
|
944
|
+
retryCount: (session.retryCount ?? 0) + 1,
|
|
945
|
+
missingFacts: missingPhaseFacts,
|
|
946
|
+
prompt: promptForMissingPhase(missingPhaseFacts),
|
|
947
|
+
};
|
|
948
|
+
continue;
|
|
949
|
+
}
|
|
950
|
+
return {
|
|
951
|
+
...accumulated,
|
|
952
|
+
ok: false,
|
|
953
|
+
stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
|
|
954
|
+
failureCategory: "invalid-output",
|
|
955
|
+
};
|
|
956
|
+
}
|
|
957
|
+
if (committedFactsOnlySuccess) {
|
|
958
|
+
// The session died but banked facts: keep the progress and move on.
|
|
959
|
+
if (missingPhaseFacts.length > 0 && (session.retryCount ?? 0) < 2) {
|
|
960
|
+
queue[index] = {
|
|
961
|
+
...session,
|
|
962
|
+
retryCount: (session.retryCount ?? 0) + 1,
|
|
963
|
+
missingFacts: missingPhaseFacts,
|
|
964
|
+
prompt: promptForMissingPhase(missingPhaseFacts),
|
|
965
|
+
};
|
|
966
|
+
continue;
|
|
967
|
+
}
|
|
968
|
+
if (missingPhaseFacts.length > 0) {
|
|
969
|
+
return {
|
|
970
|
+
...accumulated,
|
|
971
|
+
ok: false,
|
|
972
|
+
stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
|
|
973
|
+
failureCategory: "invalid-output",
|
|
974
|
+
};
|
|
975
|
+
}
|
|
976
|
+
index += 1;
|
|
977
|
+
continue;
|
|
978
|
+
}
|
|
979
|
+
// Option 5: a multi-requirement coverage batch that failed with ZERO
|
|
980
|
+
// new facts and no provider stderr is the upfront-reasoning burn —
|
|
981
|
+
// halve the slice and retry instead of failing the attempt. The compact
|
|
982
|
+
// local session participates through its requirementSlice so a
|
|
983
|
+
// thinking-burned small plan degrades into smaller batched sessions
|
|
984
|
+
// instead of replaying one full-scope prompt until the repair budget
|
|
985
|
+
// runs out.
|
|
986
|
+
const coverageSlice = session.coverageOnly
|
|
987
|
+
? session.coverageSlice
|
|
988
|
+
: isCompactLocalSession
|
|
989
|
+
? session.requirementSlice
|
|
990
|
+
: undefined;
|
|
991
|
+
const zeroProgressBurn = coverageSlice !== undefined &&
|
|
992
|
+
coverageSlice.length > 1 &&
|
|
993
|
+
(capacityExhausted || (["empty-output", "unknown"].includes(result.failureCategory) && committedAfter === committedBefore && !(result.assistantText ?? "").trim() && !result.stderr.trim())) &&
|
|
994
|
+
!result.timedOut;
|
|
995
|
+
if (zeroProgressBurn && coverageSlice) {
|
|
996
|
+
const missingIds = input.committedFacts ? new Set(collectFrontendPlanMissingFacts({ requirementIds: coverageSlice, committedFacts: input.committedFacts() }).flatMap(f => f.requirementIds)) : undefined;
|
|
997
|
+
const unfinished = coverageSlice.filter(id => !missingIds || missingIds.has(id));
|
|
998
|
+
if (unfinished.length <= 1)
|
|
999
|
+
return mapPlannerExhaustion({ ...result, ok: false, stderr: `${result.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${unfinished[0] ?? session.id}; no smaller complete scope can finish` }, committedAfter > committedBefore);
|
|
1000
|
+
const half = Math.ceil(unfinished.length / 2);
|
|
1001
|
+
const firstSlice = unfinished.slice(0, half);
|
|
1002
|
+
const secondSlice = unfinished.slice(half);
|
|
1003
|
+
if (isCompactLocalSession) {
|
|
1004
|
+
queue.splice(index + 1, 0, { id: "ux-registry", toolNames: uxRegistrySegment.toolNames, prompt: buildUxRegistryPrompt() }, ...buildWorkBatches([...firstSlice, ...secondSlice]).map((slice, i) => ({ id: `ux-local-recovery-${i}`, toolNames: uxSegment.toolNames, requirementSlice: slice, prompt: buildUxPrompt(slice) })));
|
|
1005
|
+
// The split halves leave compact mode: continue them as ordinary
|
|
1006
|
+
// coverage sessions so every downstream ladder branch applies.
|
|
1007
|
+
queue.splice(index, 1, {
|
|
1008
|
+
...session,
|
|
1009
|
+
id: `coverage-compact-split-1`,
|
|
1010
|
+
toolNames: coverageSegment.toolNames, retryCount: 0,
|
|
1011
|
+
coverageOnly: true,
|
|
1012
|
+
coverageSlice: firstSlice,
|
|
1013
|
+
requirementSlice: firstSlice,
|
|
1014
|
+
prompt: buildCoveragePrompt(firstSlice),
|
|
1015
|
+
}, {
|
|
1016
|
+
...session,
|
|
1017
|
+
id: `coverage-compact-split-2`,
|
|
1018
|
+
toolNames: coverageSegment.toolNames, retryCount: 0,
|
|
1019
|
+
coverageOnly: true,
|
|
1020
|
+
coverageSlice: secondSlice,
|
|
1021
|
+
requirementSlice: secondSlice,
|
|
1022
|
+
prompt: buildCoveragePrompt(secondSlice),
|
|
1023
|
+
});
|
|
1024
|
+
continue;
|
|
1025
|
+
}
|
|
1026
|
+
queue.splice(index, 1, {
|
|
1027
|
+
...session,
|
|
1028
|
+
coverageSlice: firstSlice,
|
|
1029
|
+
prompt: buildCoveragePrompt(firstSlice),
|
|
1030
|
+
}, {
|
|
1031
|
+
...session,
|
|
1032
|
+
coverageSlice: secondSlice,
|
|
1033
|
+
prompt: buildCoveragePrompt(secondSlice),
|
|
1034
|
+
});
|
|
1035
|
+
continue;
|
|
1036
|
+
}
|
|
1037
|
+
if (readWriterThinkingExhaustionEvidence(result).stopReason === "length" || ["context-overflow", "context-budget-exhausted"].includes(result.failureCategory))
|
|
1038
|
+
return mapPlannerExhaustion({ ...result, ok: false, stderr: `${result.stderr}\nFRONTEND_INPUT_UNIT_TOO_LARGE: ${session.id}; refine the remaining complete scope; unchanged retries are disabled` }, committedAfter > committedBefore);
|
|
1039
|
+
if ((session.coverageOnly || isUxRegistrySession || isUxLocalSession || session.id.startsWith("global-mock-data")) &&
|
|
1040
|
+
missingPhaseFacts.length > 0 &&
|
|
1041
|
+
!result.stderr.trim() &&
|
|
1042
|
+
!result.timedOut &&
|
|
1043
|
+
(session.retryCount ?? 0) < 2) {
|
|
1044
|
+
queue[index] = {
|
|
1045
|
+
...session,
|
|
1046
|
+
retryCount: (session.retryCount ?? 0) + 1,
|
|
1047
|
+
missingFacts: missingPhaseFacts,
|
|
1048
|
+
prompt: promptForMissingPhase(missingPhaseFacts),
|
|
1049
|
+
};
|
|
1050
|
+
continue;
|
|
1051
|
+
}
|
|
1052
|
+
return mapPlannerExhaustion(accumulated, committedAfter > committedBefore);
|
|
1053
|
+
}
|
|
1054
|
+
if (index < queue.length) {
|
|
1055
|
+
return {
|
|
1056
|
+
...(accumulated ?? {
|
|
1057
|
+
ok: false,
|
|
1058
|
+
stdout: "",
|
|
1059
|
+
stderr: "",
|
|
1060
|
+
assistantText: "",
|
|
1061
|
+
command: [],
|
|
1062
|
+
durationMs: 0,
|
|
1063
|
+
exitCode: null,
|
|
1064
|
+
failureCategory: "invalid-output",
|
|
1065
|
+
modelDisplay: "unknown",
|
|
1066
|
+
parsedEvents: 0,
|
|
1067
|
+
timedOut: false,
|
|
1068
|
+
attemptedModels: [],
|
|
1069
|
+
fallbackUsed: false,
|
|
1070
|
+
tokensUsed: 0,
|
|
1071
|
+
}),
|
|
1072
|
+
ok: false,
|
|
1073
|
+
stderr: `${accumulated?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
|
|
1074
|
+
failureCategory: "invalid-output",
|
|
1075
|
+
};
|
|
1076
|
+
}
|
|
1077
|
+
if (!input.parallelCoverageOnly && input.finalizePlan) {
|
|
1078
|
+
// Preserve optional closeout context if an earlier correction or future
|
|
1079
|
+
// ledger producer committed it. The old model-only finalize session was
|
|
1080
|
+
// the only writer of these fields; runtime-owned finalization must not
|
|
1081
|
+
// silently erase them when they are already available.
|
|
1082
|
+
const optionalPlanFields = {};
|
|
1083
|
+
for (const record of input.committedFacts?.() ?? []) {
|
|
1084
|
+
const fact = committedFactFromPlanRecord(record);
|
|
1085
|
+
if (!fact)
|
|
1086
|
+
continue;
|
|
1087
|
+
if (Array.isArray(fact.residualRisks)) {
|
|
1088
|
+
optionalPlanFields.residualRisks = fact.residualRisks.filter((item) => typeof item === "string" && item.trim().length > 0);
|
|
1089
|
+
}
|
|
1090
|
+
if (typeof fact.realIntegrationGap === "string" && fact.realIntegrationGap.trim()) {
|
|
1091
|
+
optionalPlanFields.realIntegrationGap = fact.realIntegrationGap;
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
let finalizationCount = 0;
|
|
1095
|
+
const finalize = () => observeFrontendFinalization({
|
|
1096
|
+
...input.observation,
|
|
1097
|
+
artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-runtime-finalize-${++finalizationCount}.json` : undefined,
|
|
1098
|
+
}, () => input.finalizePlan(optionalPlanFields));
|
|
1099
|
+
let finalizeResult = await finalize();
|
|
1100
|
+
const finalizeDetails = finalizeResult && typeof finalizeResult === "object"
|
|
1101
|
+
? (finalizeResult.details ?? finalizeResult)
|
|
1102
|
+
: finalizeResult;
|
|
1103
|
+
if (!finalizeDetails ||
|
|
1104
|
+
typeof finalizeDetails !== "object" ||
|
|
1105
|
+
finalizeDetails.ok !== true) {
|
|
1106
|
+
// Normal plans never open a finalize model session. Keep the old
|
|
1107
|
+
// correction/recovery semantic only for a rejected deterministic
|
|
1108
|
+
// compile: give the planner one bounded repair pass, then retry the
|
|
1109
|
+
// same runtime authority. The correction prompt explicitly forbids
|
|
1110
|
+
// calling finalize_plan, so terminal ownership remains deterministic.
|
|
1111
|
+
const correctionPrompt = [
|
|
1112
|
+
input.basePrompt,
|
|
1113
|
+
"PLAN FINALIZE CORRECTION — the runtime compile rejected the committed ledger.",
|
|
1114
|
+
`Runtime error: ${String(finalizeDetails?.error ?? "plan finalize rejected")}`,
|
|
1115
|
+
"Repair only the reported facts with the typed record_* tools. Do not call finalize_plan; the runtime will retry it after this correction.",
|
|
1116
|
+
input.committedFacts ? compactFrontendPlanLedgerContext({ committedFacts: input.committedFacts(), requirementIds: input.requirementIds ?? [], kinds: ["plan-requirement", "plan-verification-target", "state-registry", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"] }) : "",
|
|
1117
|
+
].filter(Boolean).join("\n\n");
|
|
1118
|
+
input.setActiveRequirementScope?.([]);
|
|
1119
|
+
const correctionTools = input.segmentCustomTools(null);
|
|
1120
|
+
const correction = await observeFrontendSession({
|
|
1121
|
+
...input.observation, phase: "plan/finalize-correction", dispatchReason: "correction",
|
|
1122
|
+
scopeIds: input.requirementIds, prompt: correctionPrompt, userMessage: input.sessionOptions.userMessage,
|
|
1123
|
+
customTools: correctionTools, committedCount: input.committedFactCount, durableCommittedCount: () => lastDurableCount,
|
|
1124
|
+
artifactPath: input.observation?.artifactPath ? `${input.observation.artifactPath}-finalize-correction.json` : undefined,
|
|
1125
|
+
}, async (observer) => {
|
|
1126
|
+
const result = await input.piStepFn({
|
|
1127
|
+
...input.sessionOptions, onAttemptObservation: observer, prompt: correctionPrompt,
|
|
1128
|
+
writerToolPolicy: { requireSdk: true, customTools: correctionTools },
|
|
1129
|
+
});
|
|
1130
|
+
try {
|
|
1131
|
+
await input.flushLedger();
|
|
1132
|
+
lastDurableCount = input.committedFactCount();
|
|
1133
|
+
}
|
|
1134
|
+
catch { /* node-level flush retries below */ }
|
|
1135
|
+
return result;
|
|
1136
|
+
});
|
|
1137
|
+
accumulated = accumulated
|
|
1138
|
+
? combineSequentialPiResults(accumulated, correction)
|
|
1139
|
+
: correction;
|
|
1140
|
+
if (!correction.ok && !["invalid-output", "empty-output", "success"].includes(correction.failureCategory))
|
|
1141
|
+
return { ...accumulated, ok: false, failureCategory: correction.failureCategory };
|
|
1142
|
+
if (correction.ok) {
|
|
1143
|
+
finalizeResult = await finalize();
|
|
1144
|
+
}
|
|
1145
|
+
const retriedDetails = finalizeResult && typeof finalizeResult === "object"
|
|
1146
|
+
? (finalizeResult.details ?? finalizeResult)
|
|
1147
|
+
: finalizeResult;
|
|
1148
|
+
if (retriedDetails && typeof retriedDetails === "object" && retriedDetails.ok === true) {
|
|
1149
|
+
return mapPlannerExhaustion(accumulated ?? { ok: true, stdout: "", stderr: "", assistantText: "", command: [], durationMs: 0, exitCode: 0, failureCategory: "success", modelDisplay: "runtime-finalize", parsedEvents: 0, timedOut: false, attemptedModels: [], fallbackUsed: false, tokensUsed: 0 }, true);
|
|
1150
|
+
}
|
|
1151
|
+
const detail = retriedDetails && typeof retriedDetails === "object"
|
|
1152
|
+
? String(retriedDetails.error ?? "plan finalize rejected")
|
|
1153
|
+
: "plan finalize rejected";
|
|
1154
|
+
return {
|
|
1155
|
+
...(accumulated ?? {
|
|
1156
|
+
ok: false,
|
|
1157
|
+
stdout: "",
|
|
1158
|
+
stderr: "",
|
|
1159
|
+
assistantText: "",
|
|
1160
|
+
command: [],
|
|
1161
|
+
durationMs: 0,
|
|
1162
|
+
exitCode: null,
|
|
1163
|
+
failureCategory: "invalid-output",
|
|
1164
|
+
modelDisplay: "unknown",
|
|
1165
|
+
parsedEvents: 0,
|
|
1166
|
+
timedOut: false,
|
|
1167
|
+
attemptedModels: [],
|
|
1168
|
+
fallbackUsed: false,
|
|
1169
|
+
tokensUsed: 0,
|
|
1170
|
+
}),
|
|
1171
|
+
ok: false,
|
|
1172
|
+
stderr: `${accumulated?.stderr ?? ""}\nfrontend plan deterministic finalize failed: ${detail}`.trim(),
|
|
1173
|
+
failureCategory: "invalid-output",
|
|
1174
|
+
};
|
|
1175
|
+
}
|
|
1176
|
+
}
|
|
1177
|
+
return mapPlannerExhaustion(accumulated ?? {
|
|
1178
|
+
ok: false,
|
|
1179
|
+
stdout: "",
|
|
1180
|
+
stderr: "frontend plan segmentation produced no session",
|
|
1181
|
+
failureCategory: "empty-output",
|
|
1182
|
+
durationMs: 0,
|
|
1183
|
+
}, false);
|
|
1184
|
+
}
|